Skip to content

guidellm.benchmark.profiles

Orchestrate multi-strategy benchmark execution through configurable profiles.

Provides abstractions for coordinating sequential execution of scheduling strategies during benchmarking workflows. Profiles automatically generate strategies based on configuration parameters, manage runtime constraints, and track completion state across execution sequences. Each profile type implements a specific execution pattern (synchronous, concurrent, throughput-focused, rate-based async, or adaptive sweep) that determines how benchmark requests are scheduled and executed.

AsyncProfile

Bases: Profile

Schedule requests at specified rates using constant or Poisson patterns.

Schedules requests at specified rates using either constant interval or Poisson distribution patterns for realistic load simulation.

Source code in src/guidellm/benchmark/profiles/asynchronous.py
@ProfileFactory.register(["async", "constant", "poisson"])
class AsyncProfile(Profile):
    """
    Schedule requests at specified rates using constant or Poisson patterns.

    Schedules requests at specified rates using either constant interval or
    Poisson distribution patterns for realistic load simulation.
    """

    args: AsyncProfileArgs

    def __init__(
        self,
        args: AsyncProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args
        if args.kind in ("async", "constant"):
            self._strategy_type: Literal["constant", "poisson"] = "constant"
        elif args.kind == "poisson":
            self._strategy_type = "poisson"
        else:
            raise ValueError(f"Invalid profile kind: {args.kind}")

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Async strategy types for each configured rate
        """
        return [self._strategy_type] * len(self.args.rate)

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> AsyncConstantStrategy | AsyncPoissonStrategy | None:
        """
        Generate async strategy for next configured rate.

        If a previous rate was terminated by a constraint with
        stopping_scope='all', remaining rates are skipped.

        :param prev_strategy: Previously completed strategy
        :param prev_benchmark: Benchmark results from previous execution
        :return: AsyncConstantStrategy or AsyncPoissonStrategy for next rate,
            or None if all rates completed or escalation halted
        :raises ValueError: If strategy_type is neither 'constant' nor 'poisson'
        """
        _ = prev_strategy

        if len(self.completed_strategies) >= len(self.args.rate):
            return None

        if prev_benchmark is not None and self._should_stop_escalating(prev_benchmark):
            return None

        current_rate = self.args.rate[len(self.completed_strategies)]

        if self._strategy_type == "constant":
            return AsyncConstantStrategy(
                rate=current_rate,
                max_concurrency=self.args.max_concurrency,
                rampup_duration=self.args.rampup_duration,
            )
        if self._strategy_type == "poisson":
            return AsyncPoissonStrategy(
                rate=current_rate,
                max_concurrency=self.args.max_concurrency,
                random_seed=self.random_seed,
            )
        raise ValueError(f"Invalid strategy type: {self._strategy_type}")

strategy_types property

Returns:

Type Description
list[str]

Async strategy types for each configured rate

next_strategy(prev_strategy, prev_benchmark)

Generate async strategy for next configured rate.

If a previous rate was terminated by a constraint with stopping_scope='all', remaining rates are skipped.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy

required
prev_benchmark Benchmark | None

Benchmark results from previous execution

required

Returns:

Type Description
AsyncConstantStrategy | AsyncPoissonStrategy | None

AsyncConstantStrategy or AsyncPoissonStrategy for next rate, or None if all rates completed or escalation halted

Raises:

Type Description
ValueError

If strategy_type is neither 'constant' nor 'poisson'

Source code in src/guidellm/benchmark/profiles/asynchronous.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> AsyncConstantStrategy | AsyncPoissonStrategy | None:
    """
    Generate async strategy for next configured rate.

    If a previous rate was terminated by a constraint with
    stopping_scope='all', remaining rates are skipped.

    :param prev_strategy: Previously completed strategy
    :param prev_benchmark: Benchmark results from previous execution
    :return: AsyncConstantStrategy or AsyncPoissonStrategy for next rate,
        or None if all rates completed or escalation halted
    :raises ValueError: If strategy_type is neither 'constant' nor 'poisson'
    """
    _ = prev_strategy

    if len(self.completed_strategies) >= len(self.args.rate):
        return None

    if prev_benchmark is not None and self._should_stop_escalating(prev_benchmark):
        return None

    current_rate = self.args.rate[len(self.completed_strategies)]

    if self._strategy_type == "constant":
        return AsyncConstantStrategy(
            rate=current_rate,
            max_concurrency=self.args.max_concurrency,
            rampup_duration=self.args.rampup_duration,
        )
    if self._strategy_type == "poisson":
        return AsyncPoissonStrategy(
            rate=current_rate,
            max_concurrency=self.args.max_concurrency,
            random_seed=self.random_seed,
        )
    raise ValueError(f"Invalid strategy type: {self._strategy_type}")

ConcurrentProfile

Bases: Profile

Execute strategies with fixed concurrency levels for performance testing.

Executes requests with a fixed number of concurrent streams, useful for testing system performance under specific concurrency levels.

Source code in src/guidellm/benchmark/profiles/concurrent.py
@ProfileFactory.register("concurrent")
class ConcurrentProfile(Profile):
    """
    Execute strategies with fixed concurrency levels for performance testing.

    Executes requests with a fixed number of concurrent streams, useful for
    testing system performance under specific concurrency levels.
    """

    args: ConcurrentProfileArgs

    def __init__(
        self,
        args: ConcurrentProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Concurrent strategy types for each configured stream count
        """
        return [self.kind] * len(self.args.streams)

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> ConcurrentStrategy | None:
        """
        Generate concurrent strategy for next stream count.

        If a previous stream count was terminated by a constraint with
        stopping_scope='all', remaining stream counts are skipped.

        :param prev_strategy: Previously completed strategy
        :param prev_benchmark: Benchmark results from previous execution
        :return: ConcurrentStrategy with next stream count, or None if complete
            or escalation halted
        """
        _ = prev_strategy

        if len(self.completed_strategies) >= len(self.args.streams):
            return None

        if prev_benchmark is not None and self._should_stop_escalating(prev_benchmark):
            return None

        return ConcurrentStrategy(
            streams=self.args.streams[len(self.completed_strategies)],
            rampup_duration=self.args.rampup_duration,
        )

strategy_types property

Returns:

Type Description
list[str]

Concurrent strategy types for each configured stream count

next_strategy(prev_strategy, prev_benchmark)

Generate concurrent strategy for next stream count.

If a previous stream count was terminated by a constraint with stopping_scope='all', remaining stream counts are skipped.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy

required
prev_benchmark Benchmark | None

Benchmark results from previous execution

required

Returns:

Type Description
ConcurrentStrategy | None

ConcurrentStrategy with next stream count, or None if complete or escalation halted

Source code in src/guidellm/benchmark/profiles/concurrent.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> ConcurrentStrategy | None:
    """
    Generate concurrent strategy for next stream count.

    If a previous stream count was terminated by a constraint with
    stopping_scope='all', remaining stream counts are skipped.

    :param prev_strategy: Previously completed strategy
    :param prev_benchmark: Benchmark results from previous execution
    :return: ConcurrentStrategy with next stream count, or None if complete
        or escalation halted
    """
    _ = prev_strategy

    if len(self.completed_strategies) >= len(self.args.streams):
        return None

    if prev_benchmark is not None and self._should_stop_escalating(prev_benchmark):
        return None

    return ConcurrentStrategy(
        streams=self.args.streams[len(self.completed_strategies)],
        rampup_duration=self.args.rampup_duration,
    )

GoodputProfile

Bases: Profile

Locate the highest concurrency meeting configured latency objectives.

Doubles concurrency until a level fails its objectives, then bisects between the highest passing and lowest failing level. Concurrency is the control variable rather than request rate because every concurrency level has a well-defined steady state, whereas a rate above the server's capacity produces a growing backlog whose measurements describe the backlog rather than the server.

Each probe's pass or fail decision uses SLO attainment, the fraction of requests meeting every objective. Attainment is a ratio over the measured population, so unlike a rate it is unaffected by how much of the measurement window the server spent filling its pipeline.

Source code in src/guidellm/benchmark/profiles/goodput.py
@ProfileFactory.register("goodput")
class GoodputProfile(Profile):
    """
    Locate the highest concurrency meeting configured latency objectives.

    Doubles concurrency until a level fails its objectives, then bisects between
    the highest passing and lowest failing level. Concurrency is the control
    variable rather than request rate because every concurrency level has a
    well-defined steady state, whereas a rate above the server's capacity
    produces a growing backlog whose measurements describe the backlog rather
    than the server.

    Each probe's pass or fail decision uses SLO attainment, the fraction of
    requests meeting every objective. Attainment is a ratio over the measured
    population, so unlike a rate it is unaffected by how much of the measurement
    window the server spent filling its pipeline.
    """

    args: GoodputProfileArgs

    def __init__(
        self,
        args: GoodputProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args
        self.search = GoodputSearchState()
        self._next_streams: int | None = args.initial_streams

    @property
    def strategy_types(self) -> list[str]:
        """
        Declare the probe budget rather than the probes run so far.

        The progress display sizes its task list from this before the first
        strategy is generated, so reporting completed probes would leave it
        empty and render every run as complete.

        :return: Concurrent strategy types, one per probe the search may run
        """
        return ["concurrent"] * self.args.max_probes

    @property
    def conclusion(self) -> dict[str, Any] | None:
        """
        :return: The recorded search trace, bounds and stop reason
        """
        return self.search.model_dump()

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> ConcurrentStrategy | None:
        """
        Generate the next concurrency level to probe.

        :param prev_strategy: Previously completed strategy instance
        :param prev_benchmark: Benchmark results from the previous probe
        :return: ConcurrentStrategy for the next level, or None when the search
            has converged, exhausted its probe budget, or hit its stream ceiling
        """
        if prev_strategy is not None and prev_benchmark is not None:
            aborted = self._should_stop_escalating(prev_benchmark)
            self._record_probe(prev_strategy, prev_benchmark, aborted=aborted)
            if aborted:
                self._next_streams = None
                self.search.stop_reason = "constraint_stopped_escalation"
            else:
                self._advance()

        if self._next_streams is None:
            self._log_result()
            return None

        if len(self.completed_strategies) >= self.args.max_probes:
            self.search.stop_reason = "max_probes_exhausted"
            self._log_result()
            return None

        return ConcurrentStrategy(
            streams=self._next_streams,
            rampup_duration=self.args.rampup_duration,
        )

    def _record_probe(
        self,
        prev_strategy: SchedulingStrategy,
        prev_benchmark: Benchmark,
        aborted: bool = False,
    ) -> None:
        """
        Score the completed probe against the target attainment.

        An aborted probe is recorded but never becomes a bound. A constraint
        that stops the run mid-probe, such as enforced over-saturation, cancels
        active requests; those are excluded from attainment, so the completed
        remainder can look conforming and would otherwise report an unsafe
        concurrency as the highest passing level.

        :param prev_strategy: Strategy that produced the benchmark
        :param prev_benchmark: Benchmark results to score
        :param aborted: Whether a constraint stopped the run during this probe
        :raises RuntimeError: If no latency objectives were configured
        """
        if not isinstance(prev_strategy, ConcurrentStrategy):
            raise RuntimeError(
                "The goodput profile only issues concurrent strategies but was "
                f"given a {type(prev_strategy).__name__} to score."
            )
        if not isinstance(prev_benchmark, GenerativeBenchmark):
            raise RuntimeError(
                "The goodput profile requires generative benchmark results to "
                f"read latency objectives from, got {type(prev_benchmark).__name__}."
            )

        streams = prev_strategy.streams
        attainment = prev_benchmark.metrics.slo_attainment
        determined = prev_benchmark.metrics.slo_determined_requests

        if attainment is None:
            raise RuntimeError(
                "The goodput profile requires latency objectives that can be "
                "measured on this workload. No request produced a determined "
                "verdict; check that --metrics defines an slo and that the "
                "objectives it names are measurable, for example that ttft_ms "
                "and tpot_ms are only used with a streaming backend."
            )

        lower, upper = wilson_interval(
            successes=round(attainment * determined),
            trials=determined,
            confidence=self.args.confidence,
        )
        target = self.args.target_attainment
        # The probe resolves the question only when the whole interval sits on
        # one side of the target. Otherwise the run was too short to tell.
        resolved = lower >= target or upper < target
        passed = attainment >= target

        probe = GoodputProbe(
            streams=streams,
            attainment=attainment,
            attainment_lower=lower,
            attainment_upper=upper,
            determined_requests=determined,
            goodput=(
                prev_benchmark.metrics.request_goodput.successful.mean
                if prev_benchmark.metrics.request_goodput is not None
                else None
            ),
            passed=passed,
            resolved=resolved,
            aborted=aborted,
        )
        self.search.probes.append(probe)

        if aborted:
            return

        if not resolved:
            logger.warning(
                "Goodput probe at concurrency {} is unresolved: attainment "
                "{:.3f} with {:.0f}% interval [{:.3f}, {:.3f}] straddles the "
                "target {:.3f}. Increase the per-probe duration or request "
                "count to resolve it.",
                streams,
                attainment,
                self.args.confidence * 100,
                lower,
                upper,
                target,
            )

        if passed:
            if (
                self.search.best_passing_streams is None
                or streams > self.search.best_passing_streams
            ):
                self.search.best_passing_streams = streams
        elif (
            self.search.lowest_failing_streams is None
            or streams < self.search.lowest_failing_streams
        ):
            self.search.lowest_failing_streams = streams

    def _log_result(self) -> None:
        """
        Report the search outcome once no further probe will run.

        The per-benchmark config captures profile state before each run, so the
        final probe and the stop reason never reach the serialized report. This
        is where a user learns whether the answer is the objective's knee or
        merely as far as the search got.

        A converged, fully resolved search logs at info level. Any other outcome
        logs at warning, because the console log level defaults to warning and a
        result that is only a lower bound is worse to miss than to over-report.
        """
        best = self.search.best_passing_streams
        reason = self.search.stop_reason
        unresolved = [
            probe.streams for probe in self.search.probes if not probe.resolved
        ]

        if best is None:
            if reason == "indeterminate_at_minimum":
                logger.warning(
                    "Goodput search found no concurrency meeting attainment "
                    ">= {:.3f}, but probes at {} streams collected too few "
                    "requests to separate their attainment from the target. "
                    "Raise the per-probe duration or request count before "
                    "concluding the objectives cannot be met.",
                    self.args.target_attainment,
                    unresolved,
                )
                return

            logger.warning(
                "Goodput search found no concurrency meeting attainment >= "
                "{:.3f}; the objectives are not met even at a single stream "
                "({}).",
                self.args.target_attainment,
                reason,
            )
            return

        if reason == "converged" and not unresolved:
            logger.info(
                "Goodput search converged after {} probes: the highest "
                "concurrency meeting attainment >= {:.3f} is {} streams.",
                len(self.search.probes),
                self.args.target_attainment,
                best,
            )
            return

        caveats = []
        if reason != "converged":
            caveats.append(
                f"the search stopped early ({reason}), so {best} is a lower "
                "bound rather than the highest passing level"
            )
        if unresolved:
            caveats.append(
                f"probes at {unresolved} streams collected too few requests to "
                "separate their attainment from the target; raise the per-probe "
                "duration or request count"
            )
        logger.warning(
            "Goodput search finished after {} probes with attainment >= {:.3f} "
            "at {} streams, but {}.",
            len(self.search.probes),
            self.args.target_attainment,
            best,
            " and ".join(caveats),
        )

    def _advance(self) -> None:
        """Choose the next concurrency level, or stop when the search is done."""
        best = self.search.best_passing_streams
        worst = self.search.lowest_failing_streams

        if worst is None:
            # No failure seen yet: double until one appears or the ceiling is hit.
            current = best if best is not None else self.args.initial_streams
            if current >= self.args.max_streams:
                self._next_streams = None
                self.search.stop_reason = "max_streams_reached"
                return
            self._next_streams = min(current * 2, self.args.max_streams)
            return

        if best is None:
            # The lowest level tested already failed; walk down toward 1.
            if worst <= 1:
                self._next_streams = None
                # Each failure was decided on a point estimate. If any of those
                # intervals straddled the target, the descent is not evidence
                # that no concurrency meets the objectives, only that the
                # probes were too short to tell them apart.
                self.search.stop_reason = (
                    "objectives_unmet_at_minimum"
                    if all(probe.resolved for probe in self.search.probes)
                    else "indeterminate_at_minimum"
                )
                return
            self._next_streams = worst // 2
            return

        if worst - best <= max(1, self.args.tolerance * best):
            self._next_streams = None
            self.search.stop_reason = "converged"
            return

        # worst - best >= 2 here, so the midpoint is strictly inside the bracket.
        self._next_streams = (best + worst) // 2

conclusion property

Returns:

Type Description
dict[str, Any] | None

The recorded search trace, bounds and stop reason

strategy_types property

Declare the probe budget rather than the probes run so far.

The progress display sizes its task list from this before the first strategy is generated, so reporting completed probes would leave it empty and render every run as complete.

Returns:

Type Description
list[str]

Concurrent strategy types, one per probe the search may run

next_strategy(prev_strategy, prev_benchmark)

Generate the next concurrency level to probe.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy instance

required
prev_benchmark Benchmark | None

Benchmark results from the previous probe

required

Returns:

Type Description
ConcurrentStrategy | None

ConcurrentStrategy for the next level, or None when the search has converged, exhausted its probe budget, or hit its stream ceiling

Source code in src/guidellm/benchmark/profiles/goodput.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> ConcurrentStrategy | None:
    """
    Generate the next concurrency level to probe.

    :param prev_strategy: Previously completed strategy instance
    :param prev_benchmark: Benchmark results from the previous probe
    :return: ConcurrentStrategy for the next level, or None when the search
        has converged, exhausted its probe budget, or hit its stream ceiling
    """
    if prev_strategy is not None and prev_benchmark is not None:
        aborted = self._should_stop_escalating(prev_benchmark)
        self._record_probe(prev_strategy, prev_benchmark, aborted=aborted)
        if aborted:
            self._next_streams = None
            self.search.stop_reason = "constraint_stopped_escalation"
        else:
            self._advance()

    if self._next_streams is None:
        self._log_result()
        return None

    if len(self.completed_strategies) >= self.args.max_probes:
        self.search.stop_reason = "max_probes_exhausted"
        self._log_result()
        return None

    return ConcurrentStrategy(
        streams=self._next_streams,
        rampup_duration=self.args.rampup_duration,
    )

Profile

Bases: ABC

Coordinate multi-strategy benchmark execution with automatic strategy generation.

Manages sequential execution of scheduling strategies with automatic strategy generation, constraint management, and completion tracking. Subclasses define specific execution patterns like synchronous, concurrent, throughput-focused, rate-based async, or adaptive sweep profiles.

Example: :: @Profile.register("synchronous") class SynchronousProfile(Profile): def init(self, args: SynchronousProfileArgs): super().init(args)

args = SynchronousProfileArgs(kind="synchronous")
profile = Profile.create(args)
Source code in src/guidellm/benchmark/profiles/profile.py
class Profile(ABC):
    """
    Coordinate multi-strategy benchmark execution with automatic strategy generation.

    Manages sequential execution of scheduling strategies with automatic strategy
    generation, constraint management, and completion tracking. Subclasses define
    specific execution patterns like synchronous, concurrent, throughput-focused,
    rate-based async, or adaptive sweep profiles.

    Example:
    ::
        @Profile.register("synchronous")
        class SynchronousProfile(Profile):
            def __init__(self, args: SynchronousProfileArgs):
                super().__init__(args)

        args = SynchronousProfileArgs(kind="synchronous")
        profile = Profile.create(args)
    """

    def __init__(
        self,
        args: ProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        """
        Initialize a profile instance.

        :param args: Validated profile argument model for this profile type
        :param random_seed: Seed for reproducible random operations in profile
            strategies.
        :param constraints: Constraints for the profile strategies.
        :param kwargs: Additional profile-specific configuration parameters
        """
        _ = kwargs  # unused
        self.kind = args.kind
        self.args = args
        self.random_seed = random_seed
        self.constraints = dict(constraints or {})
        self.completed_strategies: list[SchedulingStrategy] = []

    @property
    def info(self) -> dict[str, Any]:
        """
        Help json serialization by deferring to ProfileArgs.
        """
        return self.args.model_dump()

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Strategy types executed or to be executed in this profile
        """
        return [strat.type_ for strat in self.completed_strategies]

    @property
    def conclusion(self) -> dict[str, Any] | None:
        """
        What the profile concluded, available once its run has finished.

        Profiles that answer a question rather than execute a fixed sequence
        override this. It is read after the final strategy completes, which is
        the only point at which such an answer exists: ``info`` is captured
        into each benchmark's config before that benchmark runs, so it can
        never carry the last strategy's contribution.

        :return: Serializable conclusion mapping, or None for profiles that
            only execute a planned sequence
        """
        return None

    @staticmethod
    def _should_stop_escalating(prev_benchmark: Benchmark) -> bool:
        """
        Check if a benchmark was terminated by a constraint with stopping_scope="all".

        Inspects the scheduler state's end_queuing_constraints for any constraint
        whose stopping_scope is "all", indicating the system could not handle the
        load and escalation to subsequent rates/streams should halt.

        :param prev_benchmark: Benchmark instance
        :return: True if escalation should stop, False otherwise
        """
        scheduler_state = getattr(prev_benchmark, "scheduler_state", None)
        if scheduler_state is None:
            return False

        for name, action in scheduler_state.end_queuing_constraints.items():
            if action.stopping_scope == "all":
                logger.debug(
                    "Stopping rate escalation: constraint '{}' "
                    "triggered (stopping_scope=all)",
                    name,
                )
                return True
        return False

    def strategies_generator(
        self,
    ) -> Generator[
        tuple[SchedulingStrategy, dict[str, Constraint] | None],
        Benchmark | None,
        None,
    ]:
        """
        Generate strategies and constraints for sequential execution.

        :return: Generator yielding (strategy, constraints) tuples and receiving
            benchmark results after each execution
        """
        prev_strategy: SchedulingStrategy | None = None
        prev_benchmark: Benchmark | None = None

        while (
            strategy := self.next_strategy(prev_strategy, prev_benchmark)
        ) is not None:
            constraints = self.next_strategy_constraints(
                strategy, prev_strategy, prev_benchmark
            )
            prev_benchmark = yield (
                strategy,
                constraints,
            )
            prev_strategy = strategy
            self.completed_strategies.append(prev_strategy)

    @abstractmethod
    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> SchedulingStrategy | None:
        """
        Generate next strategy in the profile execution sequence.

        :param prev_strategy: Previously completed strategy instance
        :param prev_benchmark: Benchmark results from previous strategy execution
        :return: Next strategy to execute, or None if profile complete
        """
        ...

    def next_strategy_constraints(
        self,
        next_strategy: SchedulingStrategy | None,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> dict[str, Constraint] | None:
        """
        Generate constraints for next strategy execution.

        :param next_strategy: Strategy to be executed next
        :param prev_strategy: Previously completed strategy instance
        :param prev_benchmark: Benchmark results from previous strategy execution
        :return: Constraints dictionary for next strategy, or None
        """
        _ = (prev_strategy, prev_benchmark)  # unused
        return (
            ConstraintsInitializerFactory.resolve(self.constraints)
            if next_strategy and self.constraints
            else None
        )

conclusion property

What the profile concluded, available once its run has finished.

Profiles that answer a question rather than execute a fixed sequence override this. It is read after the final strategy completes, which is the only point at which such an answer exists: info is captured into each benchmark's config before that benchmark runs, so it can never carry the last strategy's contribution.

Returns:

Type Description
dict[str, Any] | None

Serializable conclusion mapping, or None for profiles that only execute a planned sequence

info property

Help json serialization by deferring to ProfileArgs.

strategy_types property

Returns:

Type Description
list[str]

Strategy types executed or to be executed in this profile

__init__(args, random_seed, constraints, **kwargs)

Initialize a profile instance.

Parameters:

Name Type Description Default
args ProfileArgs

Validated profile argument model for this profile type

required
random_seed int

Seed for reproducible random operations in profile strategies.

required
constraints MutableMapping[str, ConstraintInitializer | Any] | None

Constraints for the profile strategies.

required
kwargs Any

Additional profile-specific configuration parameters

{}
Source code in src/guidellm/benchmark/profiles/profile.py
def __init__(
    self,
    args: ProfileArgs,
    random_seed: int,
    constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
    **kwargs: Any,
):
    """
    Initialize a profile instance.

    :param args: Validated profile argument model for this profile type
    :param random_seed: Seed for reproducible random operations in profile
        strategies.
    :param constraints: Constraints for the profile strategies.
    :param kwargs: Additional profile-specific configuration parameters
    """
    _ = kwargs  # unused
    self.kind = args.kind
    self.args = args
    self.random_seed = random_seed
    self.constraints = dict(constraints or {})
    self.completed_strategies: list[SchedulingStrategy] = []

next_strategy(prev_strategy, prev_benchmark) abstractmethod

Generate next strategy in the profile execution sequence.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy instance

required
prev_benchmark Benchmark | None

Benchmark results from previous strategy execution

required

Returns:

Type Description
SchedulingStrategy | None

Next strategy to execute, or None if profile complete

Source code in src/guidellm/benchmark/profiles/profile.py
@abstractmethod
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> SchedulingStrategy | None:
    """
    Generate next strategy in the profile execution sequence.

    :param prev_strategy: Previously completed strategy instance
    :param prev_benchmark: Benchmark results from previous strategy execution
    :return: Next strategy to execute, or None if profile complete
    """
    ...

next_strategy_constraints(next_strategy, prev_strategy, prev_benchmark)

Generate constraints for next strategy execution.

Parameters:

Name Type Description Default
next_strategy SchedulingStrategy | None

Strategy to be executed next

required
prev_strategy SchedulingStrategy | None

Previously completed strategy instance

required
prev_benchmark Benchmark | None

Benchmark results from previous strategy execution

required

Returns:

Type Description
dict[str, Constraint] | None

Constraints dictionary for next strategy, or None

Source code in src/guidellm/benchmark/profiles/profile.py
def next_strategy_constraints(
    self,
    next_strategy: SchedulingStrategy | None,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> dict[str, Constraint] | None:
    """
    Generate constraints for next strategy execution.

    :param next_strategy: Strategy to be executed next
    :param prev_strategy: Previously completed strategy instance
    :param prev_benchmark: Benchmark results from previous strategy execution
    :return: Constraints dictionary for next strategy, or None
    """
    _ = (prev_strategy, prev_benchmark)  # unused
    return (
        ConstraintsInitializerFactory.resolve(self.constraints)
        if next_strategy and self.constraints
        else None
    )

strategies_generator()

Generate strategies and constraints for sequential execution.

Returns:

Type Description
Generator[tuple[SchedulingStrategy, dict[str, Constraint] | None], Benchmark | None, None]

Generator yielding (strategy, constraints) tuples and receiving benchmark results after each execution

Source code in src/guidellm/benchmark/profiles/profile.py
def strategies_generator(
    self,
) -> Generator[
    tuple[SchedulingStrategy, dict[str, Constraint] | None],
    Benchmark | None,
    None,
]:
    """
    Generate strategies and constraints for sequential execution.

    :return: Generator yielding (strategy, constraints) tuples and receiving
        benchmark results after each execution
    """
    prev_strategy: SchedulingStrategy | None = None
    prev_benchmark: Benchmark | None = None

    while (
        strategy := self.next_strategy(prev_strategy, prev_benchmark)
    ) is not None:
        constraints = self.next_strategy_constraints(
            strategy, prev_strategy, prev_benchmark
        )
        prev_benchmark = yield (
            strategy,
            constraints,
        )
        prev_strategy = strategy
        self.completed_strategies.append(prev_strategy)

ProfileFactory

Bases: RegistryMixin['type[Profile]']

Source code in src/guidellm/benchmark/profiles/profile.py
class ProfileFactory(RegistryMixin["type[Profile]"]):
    @classmethod
    def create(
        cls,
        args: ProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None = None,
        **kwargs: Any,
    ) -> Profile:
        """
        Create profile instances from validated profile arguments.

        :param args: Validated profile argument model for the target profile type
        :param random_seed: Seed for reproducible random operations in profile
            strategies.
        :param constraints: Constraints for the profile strategies.
        :param kwargs: Additional profile-specific configuration parameters
        :return: Configured profile instance for the specified type
        :raises ValueError: If the profile kind is not registered
        """
        kind = args.kind

        profile_class = cls.get_registered_object(kind)

        if profile_class is None:
            raise ValueError(
                f"Profile type '{kind}' is not registered. "
                f"Available types: {list(cls.registry.keys()) if cls.registry else []}"
            )

        return profile_class(args, random_seed, constraints, **kwargs)

    @classmethod
    def registered_names(cls) -> tuple[str, ...]:
        """
        Get all registered names from the registry.
        """
        return tuple(cls.registry.keys() if cls.registry else [])

create(args, random_seed, constraints=None, **kwargs) classmethod

Create profile instances from validated profile arguments.

Parameters:

Name Type Description Default
args ProfileArgs

Validated profile argument model for the target profile type

required
random_seed int

Seed for reproducible random operations in profile strategies.

required
constraints MutableMapping[str, ConstraintInitializer | Any] | None

Constraints for the profile strategies.

None
kwargs Any

Additional profile-specific configuration parameters

{}

Returns:

Type Description
Profile

Configured profile instance for the specified type

Raises:

Type Description
ValueError

If the profile kind is not registered

Source code in src/guidellm/benchmark/profiles/profile.py
@classmethod
def create(
    cls,
    args: ProfileArgs,
    random_seed: int,
    constraints: MutableMapping[str, ConstraintInitializer | Any] | None = None,
    **kwargs: Any,
) -> Profile:
    """
    Create profile instances from validated profile arguments.

    :param args: Validated profile argument model for the target profile type
    :param random_seed: Seed for reproducible random operations in profile
        strategies.
    :param constraints: Constraints for the profile strategies.
    :param kwargs: Additional profile-specific configuration parameters
    :return: Configured profile instance for the specified type
    :raises ValueError: If the profile kind is not registered
    """
    kind = args.kind

    profile_class = cls.get_registered_object(kind)

    if profile_class is None:
        raise ValueError(
            f"Profile type '{kind}' is not registered. "
            f"Available types: {list(cls.registry.keys()) if cls.registry else []}"
        )

    return profile_class(args, random_seed, constraints, **kwargs)

registered_names() classmethod

Get all registered names from the registry.

Source code in src/guidellm/benchmark/profiles/profile.py
@classmethod
def registered_names(cls) -> tuple[str, ...]:
    """
    Get all registered names from the registry.
    """
    return tuple(cls.registry.keys() if cls.registry else [])

ReplayProfile

Bases: Profile

Replay a trace file using per-row relative_timestamp from the dataset.

schedule_turn=idle_gap (the default) keeps the idle gap after each request's recorded duration, so a slow or late predecessor shifts the following request by that overrun. schedule_turn=timestamp schedules each request at start_time + time_scale * relative_timestamp. A later turn waits only while its predecessor is still running.

Dataset-side time_scale and wait caps are applied by the trace dataset before this scheduler scale.

When data_samples is set, the default max_requests constraint matches the truncated dataset size.

Source code in src/guidellm/benchmark/profiles/replay.py
@ProfileFactory.register("replay")
class ReplayProfile(Profile):
    """
    Replay a trace file using per-row ``relative_timestamp`` from the dataset.

    ``schedule_turn=idle_gap`` (the default) keeps the idle gap after each
    request's recorded duration, so a slow or late predecessor shifts the
    following request by that overrun. ``schedule_turn=timestamp`` schedules
    each request at ``start_time + time_scale * relative_timestamp``. A later
    turn waits only while its predecessor is still running.

    Dataset-side ``time_scale`` and wait caps are applied by the trace dataset
    before this scheduler scale.

    When ``data_samples`` is set, the default ``max_requests`` constraint matches
    the truncated dataset size.
    """

    args: ReplayProfileArgs

    def __init__(
        self,
        args: ReplayProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args

    @property
    def strategy_types(self) -> list[str]:
        return ["trace"]

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> TraceReplayStrategy | None:
        _ = prev_benchmark
        # Replay has a single strategy; return it once, then None
        if prev_strategy is not None:
            return None
        return TraceReplayStrategy(
            time_scale=self.args.time_scale,
            schedule_turn=self.args.schedule_turn,
        )

SweepProfile

Bases: Profile

Discover optimal rate range through adaptive multi-strategy execution.

Automatically discovers optimal rate range by executing synchronous and throughput strategies first, then interpolating rates for async strategies to comprehensively sweep the performance space.

Source code in src/guidellm/benchmark/profiles/sweep.py
@ProfileFactory.register("sweep")
class SweepProfile(Profile):
    """
    Discover optimal rate range through adaptive multi-strategy execution.

    Automatically discovers optimal rate range by executing synchronous and
    throughput strategies first, then interpolating rates for async strategies
    to comprehensively sweep the performance space.
    """

    args: SweepProfileArgs

    def __init__(
        self,
        args: SweepProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args
        self.synchronous_rate = -1.0
        self.throughput_rate = -1.0
        self.async_rates: list[float] = []
        self.measured_rates: list[float] = []

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Strategy types for the complete sweep sequence
        """
        types = ["synchronous", "throughput"]
        types += [self.args.strategy_type] * (self.args.sweep_size - len(types))
        return types

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> (
        AsyncConstantStrategy
        | AsyncPoissonStrategy
        | SynchronousStrategy
        | ThroughputStrategy
        | None
    ):
        """
        Generate next strategy in adaptive sweep sequence.

        Executes synchronous and throughput strategies first to measure baseline
        rates, then generates interpolated rates for async strategies. If a
        failure constraint is triggered during the async phase, all remaining
        higher rates are skipped.

        :param prev_strategy: Previously completed strategy instance
        :param prev_benchmark: Benchmark results from previous strategy execution
        :return: Next strategy in sweep sequence, or None if complete
        :raises ValueError: If strategy_type is neither 'constant' nor 'poisson'
        """
        if prev_strategy is None:
            return SynchronousStrategy()

        if prev_strategy.type_ == "synchronous":
            self.synchronous_rate = prev_benchmark.request_throughput.successful.mean

            return ThroughputStrategy(
                max_concurrency=self.args.max_concurrency,
                rampup_duration=self.args.rampup_duration,
            )

        if prev_strategy.type_ == "throughput":
            self.throughput_rate = prev_benchmark.request_throughput.successful.mean
            if self.synchronous_rate <= 0 and self.throughput_rate <= 0:
                raise RuntimeError(
                    "Invalid rates in sweep; aborting. "
                    "Were there any successful requests?"
                )
            self.measured_rates = list(
                np.linspace(
                    self.synchronous_rate,
                    self.throughput_rate,
                    self.args.sweep_size - 1,
                )
            )[1:]  # don't rerun synchronous

        # Stop escalation if a constraint with stopping_scope='all' triggered
        # during the async phase. Throughput is excluded because it intentionally
        # pushes beyond sustainable load. Synchronous never reaches here.
        if (
            prev_strategy.type_ != "throughput"
            and self._should_stop_escalating(prev_benchmark)  # type: ignore[arg-type]
        ):
            return None

        next_index = (
            len(self.completed_strategies) - 1 - 1
        )  # subtract synchronous and throughput
        next_rate = (
            self.measured_rates[next_index]
            if next_index < len(self.measured_rates)
            else None
        )

        if next_rate is None or next_rate <= 0:
            # Stop if we don't have another valid rate to run
            return None

        if self.args.strategy_type == "constant":
            return AsyncConstantStrategy(
                rate=next_rate, max_concurrency=self.args.max_concurrency
            )
        if self.args.strategy_type == "poisson":
            return AsyncPoissonStrategy(
                rate=next_rate,
                max_concurrency=self.args.max_concurrency,
                random_seed=self.random_seed,
            )
        raise ValueError(f"Invalid strategy type: {self.args.strategy_type}")

strategy_types property

Returns:

Type Description
list[str]

Strategy types for the complete sweep sequence

next_strategy(prev_strategy, prev_benchmark)

Generate next strategy in adaptive sweep sequence.

Executes synchronous and throughput strategies first to measure baseline rates, then generates interpolated rates for async strategies. If a failure constraint is triggered during the async phase, all remaining higher rates are skipped.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy instance

required
prev_benchmark Benchmark | None

Benchmark results from previous strategy execution

required

Returns:

Type Description
AsyncConstantStrategy | AsyncPoissonStrategy | SynchronousStrategy | ThroughputStrategy | None

Next strategy in sweep sequence, or None if complete

Raises:

Type Description
ValueError

If strategy_type is neither 'constant' nor 'poisson'

Source code in src/guidellm/benchmark/profiles/sweep.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> (
    AsyncConstantStrategy
    | AsyncPoissonStrategy
    | SynchronousStrategy
    | ThroughputStrategy
    | None
):
    """
    Generate next strategy in adaptive sweep sequence.

    Executes synchronous and throughput strategies first to measure baseline
    rates, then generates interpolated rates for async strategies. If a
    failure constraint is triggered during the async phase, all remaining
    higher rates are skipped.

    :param prev_strategy: Previously completed strategy instance
    :param prev_benchmark: Benchmark results from previous strategy execution
    :return: Next strategy in sweep sequence, or None if complete
    :raises ValueError: If strategy_type is neither 'constant' nor 'poisson'
    """
    if prev_strategy is None:
        return SynchronousStrategy()

    if prev_strategy.type_ == "synchronous":
        self.synchronous_rate = prev_benchmark.request_throughput.successful.mean

        return ThroughputStrategy(
            max_concurrency=self.args.max_concurrency,
            rampup_duration=self.args.rampup_duration,
        )

    if prev_strategy.type_ == "throughput":
        self.throughput_rate = prev_benchmark.request_throughput.successful.mean
        if self.synchronous_rate <= 0 and self.throughput_rate <= 0:
            raise RuntimeError(
                "Invalid rates in sweep; aborting. "
                "Were there any successful requests?"
            )
        self.measured_rates = list(
            np.linspace(
                self.synchronous_rate,
                self.throughput_rate,
                self.args.sweep_size - 1,
            )
        )[1:]  # don't rerun synchronous

    # Stop escalation if a constraint with stopping_scope='all' triggered
    # during the async phase. Throughput is excluded because it intentionally
    # pushes beyond sustainable load. Synchronous never reaches here.
    if (
        prev_strategy.type_ != "throughput"
        and self._should_stop_escalating(prev_benchmark)  # type: ignore[arg-type]
    ):
        return None

    next_index = (
        len(self.completed_strategies) - 1 - 1
    )  # subtract synchronous and throughput
    next_rate = (
        self.measured_rates[next_index]
        if next_index < len(self.measured_rates)
        else None
    )

    if next_rate is None or next_rate <= 0:
        # Stop if we don't have another valid rate to run
        return None

    if self.args.strategy_type == "constant":
        return AsyncConstantStrategy(
            rate=next_rate, max_concurrency=self.args.max_concurrency
        )
    if self.args.strategy_type == "poisson":
        return AsyncPoissonStrategy(
            rate=next_rate,
            max_concurrency=self.args.max_concurrency,
            random_seed=self.random_seed,
        )
    raise ValueError(f"Invalid strategy type: {self.args.strategy_type}")

SynchronousProfile

Bases: Profile

Execute single synchronous strategy for baseline performance metrics.

Executes requests sequentially with one request at a time, establishing baseline performance metrics without concurrent execution overhead.

Source code in src/guidellm/benchmark/profiles/synchronous.py
@ProfileFactory.register("synchronous")
class SynchronousProfile(Profile):
    """
    Execute single synchronous strategy for baseline performance metrics.

    Executes requests sequentially with one request at a time, establishing
    baseline performance metrics without concurrent execution overhead.
    """

    args: SynchronousProfileArgs

    def __init__(
        self,
        args: SynchronousProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Single synchronous strategy type
        """
        return [self.kind]

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> SynchronousStrategy | None:
        """
        Generate synchronous strategy for first execution only.

        :param prev_strategy: Previously completed strategy (unused)
        :param prev_benchmark: Benchmark results from previous execution (unused)
        :return: SynchronousStrategy for first execution, None afterward
        """
        _ = (prev_strategy, prev_benchmark)  # unused
        if len(self.completed_strategies) >= 1:
            return None

        return SynchronousStrategy()

strategy_types property

Returns:

Type Description
list[str]

Single synchronous strategy type

next_strategy(prev_strategy, prev_benchmark)

Generate synchronous strategy for first execution only.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy (unused)

required
prev_benchmark Benchmark | None

Benchmark results from previous execution (unused)

required

Returns:

Type Description
SynchronousStrategy | None

SynchronousStrategy for first execution, None afterward

Source code in src/guidellm/benchmark/profiles/synchronous.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> SynchronousStrategy | None:
    """
    Generate synchronous strategy for first execution only.

    :param prev_strategy: Previously completed strategy (unused)
    :param prev_benchmark: Benchmark results from previous execution (unused)
    :return: SynchronousStrategy for first execution, None afterward
    """
    _ = (prev_strategy, prev_benchmark)  # unused
    if len(self.completed_strategies) >= 1:
        return None

    return SynchronousStrategy()

ThroughputProfile

Bases: Profile

Maximize system throughput with optional concurrency constraints.

Maximizes system throughput by maintaining maximum concurrent requests, optionally constrained by a concurrency limit.

Source code in src/guidellm/benchmark/profiles/throughput.py
@ProfileFactory.register("throughput")
class ThroughputProfile(Profile):
    """
    Maximize system throughput with optional concurrency constraints.

    Maximizes system throughput by maintaining maximum concurrent requests,
    optionally constrained by a concurrency limit.
    """

    args: ThroughputProfileArgs

    def __init__(
        self,
        args: ThroughputProfileArgs,
        random_seed: int,
        constraints: MutableMapping[str, ConstraintInitializer | Any] | None,
        **kwargs: Any,
    ):
        super().__init__(args, random_seed, constraints, **kwargs)
        self.args = args

    @property
    def strategy_types(self) -> list[str]:
        """
        :return: Single throughput strategy type
        """
        return [self.kind]

    def next_strategy(
        self,
        prev_strategy: SchedulingStrategy | None,
        prev_benchmark: Benchmark | None,
    ) -> ThroughputStrategy | None:
        """
        Generate throughput strategy for first execution only.

        :param prev_strategy: Previously completed strategy (unused)
        :param prev_benchmark: Benchmark results from previous execution (unused)
        :return: ThroughputStrategy for first execution, None afterward
        """
        _ = (prev_strategy, prev_benchmark)  # unused
        if len(self.completed_strategies) >= 1:
            return None

        return ThroughputStrategy(
            max_concurrency=self.args.max_concurrency,
            rampup_duration=self.args.rampup_duration,
        )

strategy_types property

Returns:

Type Description
list[str]

Single throughput strategy type

next_strategy(prev_strategy, prev_benchmark)

Generate throughput strategy for first execution only.

Parameters:

Name Type Description Default
prev_strategy SchedulingStrategy | None

Previously completed strategy (unused)

required
prev_benchmark Benchmark | None

Benchmark results from previous execution (unused)

required

Returns:

Type Description
ThroughputStrategy | None

ThroughputStrategy for first execution, None afterward

Source code in src/guidellm/benchmark/profiles/throughput.py
def next_strategy(
    self,
    prev_strategy: SchedulingStrategy | None,
    prev_benchmark: Benchmark | None,
) -> ThroughputStrategy | None:
    """
    Generate throughput strategy for first execution only.

    :param prev_strategy: Previously completed strategy (unused)
    :param prev_benchmark: Benchmark results from previous execution (unused)
    :return: ThroughputStrategy for first execution, None afterward
    """
    _ = (prev_strategy, prev_benchmark)  # unused
    if len(self.completed_strategies) >= 1:
        return None

    return ThroughputStrategy(
        max_concurrency=self.args.max_concurrency,
        rampup_duration=self.args.rampup_duration,
    )