Skip to content

guidellm.schemas.benchmark

Centralized benchmark argument schemas for GuideLLM.

AsyncProfileArgs

Bases: ProfileArgs

Pydantic model for asynchronous profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/asynchronous.py
@ProfileArgs.register(["async", "constant", "poisson"])
class AsyncProfileArgs(ProfileArgs):
    """Pydantic model for asynchronous profile creation arguments."""

    kind: Literal["async", "constant", "poisson"] = Field(
        default="async",
        description="Profile type discriminator for asynchronous scheduling",
    )
    rate: list[PositiveFloat] = Field(
        description="Request scheduling rates in requests per second",
        examples=[1.0, [1.0, 2.0, 3.0]],
    )
    max_concurrency: PositiveInt | None = Field(
        default=None,
        description="Maximum concurrent requests to schedule",
        examples=[10],
    )

    @field_validator("rate", mode="before")
    @classmethod
    def _coerce_rate_to_list(
        cls, value: list[PositiveFloat] | PositiveFloat
    ) -> list[PositiveFloat]:
        """Normalize rate to a list of integers.

        Allow single integer or list of integers.
        """
        if isinstance(value, str):
            with contextlib.suppress(json.JSONDecodeError, ValueError):
                value = json.loads(value)
        if not value:
            raise ValueError("rate requires at least one value")
        if isinstance(value, list | tuple):
            return value
        if isinstance(value, int | float):
            return [value]
        raise ValueError(
            "rate must be a number or a list of numeric values, "
            f"got {type(value).__name__}"
        )

BenchmarkArgs

Bases: ReloadableBaseModel

Common benchmark configuration arguments.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
class BenchmarkArgs(ReloadableBaseModel):
    """Common benchmark configuration arguments."""

    model_config = args_model_config()

    backend: BackendArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("openai_http"),
        description=(
            "Backend configuration to define how to send requests to the model."
        ),
        examples=[
            {
                "kind": "openai_http",
                "target": "http://localhost:8000/v1",
            }
        ],
        json_schema_extra={"argument_alias": "backend"},
    )
    profile: ProfileArgs = Field(  # type: ignore[assignment]
        description="Profile configuration to control benchmark execution.",
        examples=[{"kind": "sweep", "sweep_size": [10.0]}],
        json_schema_extra={"argument_alias": "profile"},
    )
    constraints: list[ConstraintArgs] = Field(  # type: ignore[assignment]
        description="Execution constraints to enforce during benchmark execution",
        examples=[{"kind": "max_requests", "value": 10}],
        default_factory=list,
        json_schema_extra={"argument_alias": "constraint"},
    )
    tokenizer: DataTokenizerArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("huggingface_auto"),
        description="Tokenizer configuration",
        examples=[{"kind": "huggingface_auto"}],
        json_schema_extra={"argument_alias": "tokenizer"},
    )
    data: list[DataArgs] = Field(  # type: ignore[assignment]
        description="List of dataset sources to use in the benchmarks",
        examples=[
            {"kind": "synthetic_text", "prompt_tokens": 100, "output_tokens": 100},
            {
                "kind": "huggingface",
                "source": "my/dataset",
                "load_kwargs": {"split": "test", "name": "my_dataset"},
            },
        ],
        min_length=1,
        json_schema_extra={"argument_alias": "data"},
    )
    data_column_mapper: DataPreprocessorArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("generative_column_mapper"),
        description="Specify how to map dataset columns into prompts and outputs.",
        examples=[{"kind": "generative_column_mapper"}],
        json_schema_extra={"argument_alias": "data_column_mapper"},
    )
    data_preprocessors: list[DataPreprocessorArgs] = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind_list("encode_media"),  # type: ignore[arg-type]
        description="List of dataset preprocessors to apply to the datasets.",
        examples=[{"kind": "encode_media"}],
        json_schema_extra={"argument_alias": "data_preprocessor"},
    )
    data_finalizer: DataFinalizerArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("generative"),
        description="Finalizer for preparing data samples into requests",
        examples=[{"kind": "generative"}],
        json_schema_extra={"argument_alias": "data_finalizer"},
    )
    data_loader: DataLoaderArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("pytorch"),
        description="Specify how to load the datasets into memory.",
        examples=[{"kind": "pytorch"}],
        json_schema_extra={"argument_alias": "data_loader"},
    )
    seed: RandomArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("static"),
        description="Random configuration for reproducibility (e.g., seed value)",
        examples=[{"kind": "static", "value": 42}],
        json_schema_extra={"argument_alias": "seed"},
    )
    outputs: list[BenchmarkOutputArgs] = Field(
        default_factory=lambda: default_kind_list("json", "csv"),  # type: ignore[arg-type]
        description="Benchmark output formats and paths.",
        examples=[
            {"kind": "json", "filename": "benchmarks.json"},
        ],
        json_schema_extra={"argument_alias": "output"},
    )
    metrics: MetricsArgs = Field(  # type: ignore[assignment]
        default_factory=lambda: default_kind("generative"),
        description="Configuration for metrics collection and request sampling.",
        json_schema_extra={"argument_alias": "metrics"},
    )

    @model_validator(mode="after")
    def _check_profile_supports_metrics(self) -> BenchmarkArgs:
        """
        Let the profile reject a metrics configuration it cannot work with.

        :return: The validated instance
        :raises ValueError: If the profile rejects the metrics configuration
        """
        self.profile.validate_metrics(self.metrics)

        return self

BenchmarkMetadata

Bases: StandardBaseModel

Metadata about the benchmark scenario.

Contains information such as name, description, and tags that describe the benchmark scenario. This metadata is used for reporting and organizational purposes but does not affect benchmark execution.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
class BenchmarkMetadata(StandardBaseModel):
    """
    Metadata about the benchmark scenario.

    Contains information such as name, description, and tags that describe the
    benchmark scenario. This metadata is used for reporting and organizational
    purposes but does not affect benchmark execution.
    """

    # Allow arbitrary metadata fields
    model_config = args_model_config() | ConfigDict(extra="allow")

    labels: dict[str, str] = Field(
        default_factory=dict,
        description=(
            "Key-value pairs of metadata labels which will be written to the "
            "output reports."
        ),
        examples=[{"name": "benchmark", "description": "Benchmark description"}],
    )

BenchmarkOutputArgs

Bases: PydanticClassRegistryMixin['BenchmarkOutputArgs'], ABC

Base class for output creation arguments.

This class serves as a base for defining argument models used in the creation of output instances. It inherits from PydanticClassRegistryMixin to enable automatic registration of subclasses, allowing for flexible and extensible output configurations.

Attributes:

Name Type Description
schema_discriminator str

Field name for polymorphic deserialization

Source code in src/guidellm/schemas/benchmark/outputs/output.py
class BenchmarkOutputArgs(PydanticClassRegistryMixin["BenchmarkOutputArgs"], ABC):
    """Base class for output creation arguments.

    This class serves as a base for defining argument models used in the creation
    of output instances. It inherits from PydanticClassRegistryMixin to enable
    automatic registration of subclasses, allowing for flexible and extensible
    output configurations.

    :cvar schema_discriminator: Field name for polymorphic deserialization
    """

    model_config = standard_model_config()

    schema_discriminator: ClassVar[str] = "kind"

    @classmethod
    def __pydantic_schema_base_type__(cls) -> type[BenchmarkOutputArgs]:
        """
        Return base type for polymorphic validation hierarchy.

        :return: Base BenchmarkOutputArgs class for schema validation
        """
        if cls.__name__ == "BenchmarkOutputArgs":
            return cls

        return BenchmarkOutputArgs

    kind: str = Field(
        description="Type identifier for the output configuration.",
    )

__pydantic_schema_base_type__() classmethod

Return base type for polymorphic validation hierarchy.

Returns:

Type Description
type[BenchmarkOutputArgs]

Base BenchmarkOutputArgs class for schema validation

Source code in src/guidellm/schemas/benchmark/outputs/output.py
@classmethod
def __pydantic_schema_base_type__(cls) -> type[BenchmarkOutputArgs]:
    """
    Return base type for polymorphic validation hierarchy.

    :return: Base BenchmarkOutputArgs class for schema validation
    """
    if cls.__name__ == "BenchmarkOutputArgs":
        return cls

    return BenchmarkOutputArgs

BenchmarkScenario

Bases: ReloadableBaseModel, BaseSettings

Configuration arguments for generative text benchmark execution.

Defines all parameters for benchmark setup including target endpoint, data sources, backend configuration, processing pipeline, output formatting, and execution constraints. Supports loading from scenario files and merging with runtime overrides for flexible benchmark construction from multiple sources.

Example::

# Load from built-in scenario with overrides
args = BenchmarkScenario.create(
    scenario="chat",
    spec={"backend": {"kind": "openai_http", "target": "http://localhost:8000/v1"}},
)

# Create from keyword arguments only
args = BenchmarkScenario(
    spec=BenchmarkArgs(
        backend={"kind": "openai_http", "target": "http://localhost:8000/v1"},
        data=[{"kind": "synthetic_text"}],
    ),
)
Source code in src/guidellm/schemas/benchmark/entrypoints.py
class BenchmarkScenario(ReloadableBaseModel, BaseSettings):
    """
    Configuration arguments for generative text benchmark execution.

    Defines all parameters for benchmark setup including target endpoint, data
    sources, backend configuration, processing pipeline, output formatting, and
    execution constraints. Supports loading from scenario files and merging with
    runtime overrides for flexible benchmark construction from multiple sources.

    Example::

        # Load from built-in scenario with overrides
        args = BenchmarkScenario.create(
            scenario="chat",
            spec={"backend": {"kind": "openai_http", "target": "http://localhost:8000/v1"}},
        )

        # Create from keyword arguments only
        args = BenchmarkScenario(
            spec=BenchmarkArgs(
                backend={"kind": "openai_http", "target": "http://localhost:8000/v1"},
                data=[{"kind": "synthetic_text"}],
            ),
        )
    """

    model_config = SettingsConfigDict(
        env_prefix="GUIDELLM__",
        env_nested_delimiter="__",
        validate_default=True,
    )

    @classmethod
    def create(cls, scenario: Path | str | None, **kwargs: Any) -> BenchmarkScenario:
        """
        Create benchmark args from scenario file and keyword arguments.

        Loads base configuration from scenario file (built-in or custom) and merges
        with provided keyword arguments. Arguments explicitly set via kwargs override
        scenario values, while defaulted kwargs are ignored to preserve scenario
        settings.

        :param scenario: Path to scenario file, built-in scenario name, or None
        :param kwargs: Keyword arguments to override scenario values
        :return: Configured benchmark args instance
        :raises ValueError: If scenario is not found or file format is unsupported
        """
        constructor_kwargs = {}

        if scenario is not None:
            if isinstance(scenario, str) and scenario in (
                builtin_scenarios := get_builtin_scenarios()
            ):
                scenario_path = builtin_scenarios[scenario]
            elif Path(scenario).exists() and Path(scenario).is_file():
                scenario_path = Path(scenario)
            else:
                raise ValueError(f"Scenario '{scenario}' not found.")

            with scenario_path.open() as file:
                if scenario_path.suffix == ".json":
                    scenario_data = json.load(file)
                elif scenario_path.suffix in {".yaml", ".yml"}:
                    scenario_data = yaml.safe_load(file)
                else:
                    raise ValueError(
                        f"Unsupported scenario file format: {scenario_path.suffix}"
                    )
            # NOTE: If the scenario file is a report, it contains a "config" key with
            # the benchmark configuration. This is a hack and should be replaced.
            if "config" in scenario_data:
                # loading from a report file
                scenario_data = scenario_data["config"]
            constructor_kwargs.update(scenario_data)

        # NOTE In the future replace deep_update with a more intelligent merging
        #      strategy that accounts for changes to `kind`.
        # Apply overrides from kwargs
        deep_update(constructor_kwargs, kwargs)

        return cls.model_validate(constructor_kwargs)

    def get_benchmarks(self) -> list[BenchmarkArgs]:
        """
        Get list of benchmark argument instances for each individual benchmark.

        Combines global arguments with individual benchmark overrides to produce a
        list of fully configured benchmark argument instances for execution.

        :return: List of benchmark argument instances
        """
        parser = ArgStringParser(allow_overwrite=True)
        benchmarks = []
        for benchmark_override in self.benchmarks:
            if benchmark_override is None:
                benchmarks.append(self.spec.model_copy(deep=True))
            else:
                # Create a copy of the common args to apply overrides to
                benchmark_args = self.spec.model_dump(mode="python")
                for key, value in benchmark_override.items():
                    parser.set(benchmark_args, key, value)
                benchmarks.append(BenchmarkArgs.model_validate(benchmark_args))

        return benchmarks

    metadata: BenchmarkMetadata = Field(
        default_factory=BenchmarkMetadata,
        description=(
            "User metadata to describe the benchmark run. This data is written "
            "to the output file but not otherwise used by GuideLLM)."
        ),
        examples=[
            {"labels": {"name": "benchmark", "description": "Benchmark description"}}
        ],
    )
    spec: BenchmarkArgs = Field(
        default_factory=BenchmarkArgs,  # type: ignore[arg-type]
        description="Global configuration parameters for benchmark execution.",
        examples=[
            {
                "backend": {
                    "kind": "openai_http",
                    "target": "http://localhost:8000/v1",
                },
                "data": [{"kind": "synthetic_text"}],
            }
        ],
    )
    benchmarks: list[dict[str, Any] | None] = Field(
        default_factory=lambda: [None],  # type: ignore[arg-type]
        description=(
            "Individual benchmark parameter overrides. This allows overriding "
            "parameters and constraints for each benchmark run by a profile."
        ),
        min_length=1,
        examples=[
            {"profile.rate": 10.0, "constraints[0].seconds": 10},
            {"profile.rate": 20.0, "constraints[0].seconds": 20},
        ],
    )

    @model_validator(mode="before")
    @classmethod
    def insert_first_benchmark(cls, data: Any) -> Any:
        """
        Inserts the first benchmark into the common args.

        This allows users to ommit fields from the common args if they have overrides
        in the first benchmark.
        """
        if not isinstance(data, dict):
            return data

        if "benchmarks" not in data or not data["benchmarks"]:
            # No benchmarks provided, insert a blank one
            data["benchmarks"] = [None]

        first_benchmark: dict[str, Any] | None = data["benchmarks"][0]
        if isinstance(first_benchmark, dict) and first_benchmark:
            # Ensure "spec" field exists for the parser to insert into
            data["spec"] = data.get("spec", {})
            parser = ArgStringParser(allow_overwrite=True)

            # Insert the first benchmark into the common args
            # Create fields recursively.
            for key, value in first_benchmark.items():
                parser.set(data["spec"], key, value)

        return data

create(scenario, **kwargs) classmethod

Create benchmark args from scenario file and keyword arguments.

Loads base configuration from scenario file (built-in or custom) and merges with provided keyword arguments. Arguments explicitly set via kwargs override scenario values, while defaulted kwargs are ignored to preserve scenario settings.

Parameters:

Name Type Description Default
scenario Path | str | None

Path to scenario file, built-in scenario name, or None

required
kwargs Any

Keyword arguments to override scenario values

{}

Returns:

Type Description
BenchmarkScenario

Configured benchmark args instance

Raises:

Type Description
ValueError

If scenario is not found or file format is unsupported

Source code in src/guidellm/schemas/benchmark/entrypoints.py
@classmethod
def create(cls, scenario: Path | str | None, **kwargs: Any) -> BenchmarkScenario:
    """
    Create benchmark args from scenario file and keyword arguments.

    Loads base configuration from scenario file (built-in or custom) and merges
    with provided keyword arguments. Arguments explicitly set via kwargs override
    scenario values, while defaulted kwargs are ignored to preserve scenario
    settings.

    :param scenario: Path to scenario file, built-in scenario name, or None
    :param kwargs: Keyword arguments to override scenario values
    :return: Configured benchmark args instance
    :raises ValueError: If scenario is not found or file format is unsupported
    """
    constructor_kwargs = {}

    if scenario is not None:
        if isinstance(scenario, str) and scenario in (
            builtin_scenarios := get_builtin_scenarios()
        ):
            scenario_path = builtin_scenarios[scenario]
        elif Path(scenario).exists() and Path(scenario).is_file():
            scenario_path = Path(scenario)
        else:
            raise ValueError(f"Scenario '{scenario}' not found.")

        with scenario_path.open() as file:
            if scenario_path.suffix == ".json":
                scenario_data = json.load(file)
            elif scenario_path.suffix in {".yaml", ".yml"}:
                scenario_data = yaml.safe_load(file)
            else:
                raise ValueError(
                    f"Unsupported scenario file format: {scenario_path.suffix}"
                )
        # NOTE: If the scenario file is a report, it contains a "config" key with
        # the benchmark configuration. This is a hack and should be replaced.
        if "config" in scenario_data:
            # loading from a report file
            scenario_data = scenario_data["config"]
        constructor_kwargs.update(scenario_data)

    # NOTE In the future replace deep_update with a more intelligent merging
    #      strategy that accounts for changes to `kind`.
    # Apply overrides from kwargs
    deep_update(constructor_kwargs, kwargs)

    return cls.model_validate(constructor_kwargs)

get_benchmarks()

Get list of benchmark argument instances for each individual benchmark.

Combines global arguments with individual benchmark overrides to produce a list of fully configured benchmark argument instances for execution.

Returns:

Type Description
list[BenchmarkArgs]

List of benchmark argument instances

Source code in src/guidellm/schemas/benchmark/entrypoints.py
def get_benchmarks(self) -> list[BenchmarkArgs]:
    """
    Get list of benchmark argument instances for each individual benchmark.

    Combines global arguments with individual benchmark overrides to produce a
    list of fully configured benchmark argument instances for execution.

    :return: List of benchmark argument instances
    """
    parser = ArgStringParser(allow_overwrite=True)
    benchmarks = []
    for benchmark_override in self.benchmarks:
        if benchmark_override is None:
            benchmarks.append(self.spec.model_copy(deep=True))
        else:
            # Create a copy of the common args to apply overrides to
            benchmark_args = self.spec.model_dump(mode="python")
            for key, value in benchmark_override.items():
                parser.set(benchmark_args, key, value)
            benchmarks.append(BenchmarkArgs.model_validate(benchmark_args))

    return benchmarks

insert_first_benchmark(data) classmethod

Inserts the first benchmark into the common args.

This allows users to ommit fields from the common args if they have overrides in the first benchmark.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
@model_validator(mode="before")
@classmethod
def insert_first_benchmark(cls, data: Any) -> Any:
    """
    Inserts the first benchmark into the common args.

    This allows users to ommit fields from the common args if they have overrides
    in the first benchmark.
    """
    if not isinstance(data, dict):
        return data

    if "benchmarks" not in data or not data["benchmarks"]:
        # No benchmarks provided, insert a blank one
        data["benchmarks"] = [None]

    first_benchmark: dict[str, Any] | None = data["benchmarks"][0]
    if isinstance(first_benchmark, dict) and first_benchmark:
        # Ensure "spec" field exists for the parser to insert into
        data["spec"] = data.get("spec", {})
        parser = ArgStringParser(allow_overwrite=True)

        # Insert the first benchmark into the common args
        # Create fields recursively.
        for key, value in first_benchmark.items():
            parser.set(data["spec"], key, value)

    return data

CSVBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Model for CSV benchmark output arguments.

Source code in src/guidellm/schemas/benchmark/outputs/csv.py
@BenchmarkOutputArgs.register("csv")
class CSVBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Model for CSV benchmark output arguments."""

    kind: Literal["csv"] = Field(
        default="csv",
        description="The kind of output.",
    )
    path: Path = Field(
        default_factory=lambda: settings.default_results_dir / "benchmarks.csv",
        description="The file to save the output to.",
    )

ConcurrentProfileArgs

Bases: ProfileArgs

Pydantic model for concurrent profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/concurrent.py
@ProfileArgs.register("concurrent")
class ConcurrentProfileArgs(ProfileArgs):
    """Pydantic model for concurrent profile creation arguments."""

    kind: Literal["concurrent"] = Field(
        default="concurrent",
        description="Profile type discriminator for concurrent scheduling",
    )
    streams: list[PositiveInt] = Field(
        description="Concurrent stream counts to execute",
        examples=[[1, 2, 3], 10],
    )

    @field_validator("streams", mode="before")
    @classmethod
    def _coerce_streams_to_list(cls, value: Any) -> Any:
        """Normalize streams to a list of integers.

        Allow single integer or list of integers.
        """
        if isinstance(value, str):
            with contextlib.suppress(json.JSONDecodeError, ValueError):
                value = json.loads(value)
        if not value:
            raise ValueError("streams requires at least one value")
        if isinstance(value, list | tuple):
            return [int(stream) for stream in value]
        if isinstance(value, int | float):
            return [int(value)]
        raise ValueError(
            "streams must be a number or a list of numeric values, "
            f"got {type(value).__name__}"
        )

ConsoleBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Base class for console benchmark output arguments.

Source code in src/guidellm/schemas/benchmark/outputs/console.py
@BenchmarkOutputArgs.register("console")
class ConsoleBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Base class for console benchmark output arguments."""

    kind: Literal["console"] = Field(
        default="console",
        description="The kind of output.",
    )

GenerativeMetricsArgs

Bases: MetricsArgs

Metrics configuration for generative (autoregressive) benchmarks.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
@MetricsArgs.register("generative")
class GenerativeMetricsArgs(MetricsArgs):
    """Metrics configuration for generative (autoregressive) benchmarks."""

    kind: Literal["generative"] = Field(
        default="generative",
        description="The kind of metrics configuration to use.",
    )
    sample_size: int | None = Field(
        default=None,
        description=(
            "Maximum number of requests per status group (completed, errored, "
            "incomplete) to retain full data (prompt, output, tool calls) for in "
            "the final benchmark. Lightweight stats (latency, token counts) are "
            "always kept for every request. None keeps all request data, 0 strips "
            "all request data, N > 0 uses reservoir sampling to retain N per group."
        ),
        examples=[None, 0, 100],
    )
    prefer_response_metrics: bool = Field(
        default=True,
        description=(
            "Prioritize server-reported metrics over client-calculated metrics "
            "when both are available."
        ),
    )
    confidence: float | None = Field(
        default=0.95,
        gt=0.0,
        lt=1.0,
        description=(
            "Two-sided confidence level for the intervals reported alongside "
            "request-level metrics. Set to null to report those metrics without "
            "intervals."
        ),
        examples=[0.95, 0.99, None],
    )
    slo: GoodputSLO | None = Field(
        default=None,
        description=(
            "Per-request latency objectives defining which requests count "
            "toward goodput. None disables goodput measurement."
        ),
        examples=[None, {"ttft_ms": 2000, "tpot_ms": 100}],
    )

GoodputProfileArgs

Bases: ProfileArgs

Pydantic model for goodput search profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/goodput.py
@ProfileArgs.register("goodput")
class GoodputProfileArgs(ProfileArgs):
    """Pydantic model for goodput search profile creation arguments."""

    kind: Literal["goodput"] = Field(
        default="goodput",
        description="Profile type discriminator for goodput search scheduling",
    )
    target_attainment: float = Field(
        default=0.95,
        gt=0.0,
        lt=1.0,
        description=(
            "Fraction of requests that must meet every configured latency "
            "objective for a concurrency level to pass. The default of 0.95 is "
            "equivalent to requiring the p95 of each objective's metric to sit "
            "within its threshold. Must be below 1.0: a finite sample cannot "
            "establish that no request in the population violates an objective"
        ),
        examples=[0.95, 0.99],
    )
    initial_streams: PositiveInt = Field(
        default=4,
        description=(
            "Concurrency level probed first. The search doubles from here until "
            "a level fails, then bisects between the last pass and first failure"
        ),
    )
    max_streams: PositiveInt = Field(
        default=1024,
        description=(
            "Upper limit on concurrency the search will probe. Reaching it "
            "without a failure ends the search and reports the objectives as "
            "met at every level tested"
        ),
    )
    tolerance: float = Field(
        default=0.1,
        gt=0.0,
        le=1.0,
        description=(
            "Relative width of the bracket at which the search stops, as a "
            "fraction of the highest passing concurrency. Bisecting to an exact "
            "integer costs a probe per halving regardless of scale, so the "
            "default stops once the answer is known to within 10 percent"
        ),
        examples=[0.1, 0.05],
    )
    max_probes: PositiveInt = Field(
        default=15,
        description=(
            "Maximum number of benchmark runs the search may execute before "
            "reporting its best result so far. Doubling from initial_streams to "
            "max_streams costs log2(max_streams / initial_streams) probes and "
            "the bisection that follows costs about log2(1 / tolerance) more"
        ),
    )
    confidence: float = Field(
        default=0.95,
        ge=0.5,
        le=0.999,
        description=(
            "Confidence level for the Wilson score interval reported around "
            "each probe's attainment, used to flag results the run was too "
            "short to resolve"
        ),
    )

    def validate_metrics(self, metrics: Any) -> None:
        """
        Require latency objectives for the search to search against.

        Without this the run fails only once the first probe has finished and
        its attainment turns out to be unmeasurable, wasting a full probe
        duration on a configuration error.

        :param metrics: Validated metrics arguments for the run
        :raises ValueError: If no latency objectives are configured
        """
        # Read through model_dump rather than importing GenerativeMetricsArgs,
        # which would import this module back through the args package.
        if metrics.model_dump().get("slo") is None:
            raise ValueError(
                "The goodput profile searches for the highest load meeting "
                "latency objectives, so it requires objectives to be set, for "
                'example --metrics \'{"kind":"generative","slo":'
                '{"ttft_ms":2000}}\''
            )

    @model_validator(mode="after")
    def _check_stream_bounds(self) -> GoodputProfileArgs:
        """
        Validate that the search range is non-empty.

        :return: The validated instance
        :raises ValueError: If initial_streams exceeds max_streams
        """
        if self.initial_streams > self.max_streams:
            raise ValueError(
                f"initial_streams ({self.initial_streams}) must not exceed "
                f"max_streams ({self.max_streams})"
            )

        return self

validate_metrics(metrics)

Require latency objectives for the search to search against.

Without this the run fails only once the first probe has finished and its attainment turns out to be unmeasurable, wasting a full probe duration on a configuration error.

Parameters:

Name Type Description Default
metrics Any

Validated metrics arguments for the run

required

Raises:

Type Description
ValueError

If no latency objectives are configured

Source code in src/guidellm/schemas/benchmark/profiles/goodput.py
def validate_metrics(self, metrics: Any) -> None:
    """
    Require latency objectives for the search to search against.

    Without this the run fails only once the first probe has finished and
    its attainment turns out to be unmeasurable, wasting a full probe
    duration on a configuration error.

    :param metrics: Validated metrics arguments for the run
    :raises ValueError: If no latency objectives are configured
    """
    # Read through model_dump rather than importing GenerativeMetricsArgs,
    # which would import this module back through the args package.
    if metrics.model_dump().get("slo") is None:
        raise ValueError(
            "The goodput profile searches for the highest load meeting "
            "latency objectives, so it requires objectives to be set, for "
            'example --metrics \'{"kind":"generative","slo":'
            '{"ttft_ms":2000}}\''
        )

GoodputSLO

Bases: StandardBaseModel

Per-request latency objectives defining which requests count as conforming.

Every objective left unset is ignored. A request conforms when it satisfies all objectives that are set; a benchmark with no objectives set has no meaningful goodput and reports it as None.

Note the mapping between objective names and GuideLLM metrics. tpot is compared against :attr:GenerativeRequestStats.inter_token_latency_ms, which excludes the first token, and not against GuideLLM's time_per_output_token_ms, which includes it. This is the closest GuideLLM metric to vLLM's tpot but is not identical: vLLM divides by the interval ending at the request's completion, while inter-token latency ends at the last token received.

Example: :: slo = GoodputSLO(ttft_ms=2000, tpot_ms=100) conforming = slo.is_conforming(ttft_ms=150.0, tpot_ms=12.0, e2el_ms=None)

Source code in src/guidellm/schemas/benchmark/goodput.py
class GoodputSLO(StandardBaseModel):
    """
    Per-request latency objectives defining which requests count as conforming.

    Every objective left unset is ignored. A request conforms when it satisfies
    all objectives that are set; a benchmark with no objectives set has no
    meaningful goodput and reports it as None.

    Note the mapping between objective names and GuideLLM metrics. ``tpot`` is
    compared against :attr:`GenerativeRequestStats.inter_token_latency_ms`,
    which excludes the first token, and not against GuideLLM's
    ``time_per_output_token_ms``, which includes it. This is the closest
    GuideLLM metric to vLLM's ``tpot`` but is not identical: vLLM divides by
    the interval ending at the request's completion, while inter-token latency
    ends at the last token received.

    Example:
    ::
        slo = GoodputSLO(ttft_ms=2000, tpot_ms=100)
        conforming = slo.is_conforming(ttft_ms=150.0, tpot_ms=12.0, e2el_ms=None)
    """

    ttft_ms: PositiveFloat | None = Field(
        default=None,
        description=(
            "Maximum time to first token in milliseconds. Compared against "
            "each request's time_to_first_token_ms"
        ),
        examples=[2000.0],
    )
    tpot_ms: PositiveFloat | None = Field(
        default=None,
        description=(
            "Maximum time per output token in milliseconds, excluding the "
            "first token. Compared against each request's "
            "inter_token_latency_ms. Requests producing one token or fewer have "
            "no inter-token latency and are left undetermined"
        ),
        examples=[100.0],
    )
    e2el_ms: PositiveFloat | None = Field(
        default=None,
        description=(
            "Maximum end-to-end request latency in milliseconds. Compared "
            "against each request's request_latency, converted from seconds"
        ),
        examples=[30000.0],
    )

    @model_validator(mode="after")
    def _require_an_objective(self) -> GoodputSLO:
        """
        Validate that at least one objective is set.

        :return: The validated instance
        :raises ValueError: If no objective is set
        """
        if all(value is None for value in (self.ttft_ms, self.tpot_ms, self.e2el_ms)):
            raise ValueError(
                "GoodputSLO requires at least one of ttft_ms, tpot_ms, or e2el_ms"
            )

        return self

    def is_conforming(
        self,
        ttft_ms: float | None,
        tpot_ms: float | None,
        e2el_ms: float | None,
    ) -> bool | None:
        """
        Determine whether one request's measured latencies satisfy the objectives.

        A request is undetermined as soon as any configured objective has no
        corresponding measurement, even if another objective is already
        breached. Deciding such a request on its measurable objectives alone
        would bias the population it is averaged over: on a workload where an
        objective is never measurable, only the requests that happen to breach
        a different objective would remain, driving attainment to zero.

        :param ttft_ms: Measured time to first token in milliseconds
        :param tpot_ms: Measured inter-token latency in milliseconds
        :param e2el_ms: Measured end-to-end latency in milliseconds
        :return: True if conforming, False if violating, None if undetermined
        """
        measured = (ttft_ms, tpot_ms, e2el_ms)
        objectives = (self.ttft_ms, self.tpot_ms, self.e2el_ms)

        conforming = True
        for value, objective in zip(measured, objectives, strict=True):
            if objective is None:
                continue
            if value is None:
                return None
            if value > objective:
                conforming = False

        return conforming

is_conforming(ttft_ms, tpot_ms, e2el_ms)

Determine whether one request's measured latencies satisfy the objectives.

A request is undetermined as soon as any configured objective has no corresponding measurement, even if another objective is already breached. Deciding such a request on its measurable objectives alone would bias the population it is averaged over: on a workload where an objective is never measurable, only the requests that happen to breach a different objective would remain, driving attainment to zero.

Parameters:

Name Type Description Default
ttft_ms float | None

Measured time to first token in milliseconds

required
tpot_ms float | None

Measured inter-token latency in milliseconds

required
e2el_ms float | None

Measured end-to-end latency in milliseconds

required

Returns:

Type Description
bool | None

True if conforming, False if violating, None if undetermined

Source code in src/guidellm/schemas/benchmark/goodput.py
def is_conforming(
    self,
    ttft_ms: float | None,
    tpot_ms: float | None,
    e2el_ms: float | None,
) -> bool | None:
    """
    Determine whether one request's measured latencies satisfy the objectives.

    A request is undetermined as soon as any configured objective has no
    corresponding measurement, even if another objective is already
    breached. Deciding such a request on its measurable objectives alone
    would bias the population it is averaged over: on a workload where an
    objective is never measurable, only the requests that happen to breach
    a different objective would remain, driving attainment to zero.

    :param ttft_ms: Measured time to first token in milliseconds
    :param tpot_ms: Measured inter-token latency in milliseconds
    :param e2el_ms: Measured end-to-end latency in milliseconds
    :return: True if conforming, False if violating, None if undetermined
    """
    measured = (ttft_ms, tpot_ms, e2el_ms)
    objectives = (self.ttft_ms, self.tpot_ms, self.e2el_ms)

    conforming = True
    for value, objective in zip(measured, objectives, strict=True):
        if objective is None:
            continue
        if value is None:
            return None
        if value > objective:
            conforming = False

    return conforming

HTMLBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Model for HTML benchmark output arguments.

Source code in src/guidellm/schemas/benchmark/outputs/html.py
@BenchmarkOutputArgs.register("html")
class HTMLBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Model for HTML benchmark output arguments."""

    kind: Literal["html"] = Field(
        default="html",
        description="The kind of output.",
    )
    path: Path = Field(
        default_factory=lambda: settings.default_results_dir / "benchmarks.html",
        description="The file to save the output to.",
    )

JSONBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Model for JSON benchmark output arguments.

Source code in src/guidellm/schemas/benchmark/outputs/serialized.py
@BenchmarkOutputArgs.register("json")
class JSONBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Model for JSON benchmark output arguments."""

    kind: Literal["json"] = Field(
        default="json",
        description="The kind of output.",
        examples=["json"],
    )
    path: Path = Field(
        default_factory=lambda: settings.default_results_dir / "benchmarks.json",
        description="The file to save the output to.",
        examples=["./benchmarks.json"],
    )

MetricsArgs

Bases: PydanticClassRegistryMixin['MetricsArgs'], ABC

Base class for metrics collection arguments.

Attributes:

Name Type Description
schema_discriminator str

Field name for polymorphic deserialization

Source code in src/guidellm/schemas/benchmark/entrypoints.py
class MetricsArgs(PydanticClassRegistryMixin["MetricsArgs"], ABC):
    """Base class for metrics collection arguments.

    :cvar schema_discriminator: Field name for polymorphic deserialization
    """

    model_config = standard_model_config()

    schema_discriminator: ClassVar[str] = "kind"

    @classmethod
    def __pydantic_schema_base_type__(cls) -> type[MetricsArgs]:
        """
        Return base type for polymorphic validation hierarchy.

        :return: Base MetricsArgs class for schema validation
        """
        if cls.__name__ == "MetricsArgs":
            return cls

        return MetricsArgs

    kind: str = Field(
        description="The kind of metrics configuration to use.",
    )

__pydantic_schema_base_type__() classmethod

Return base type for polymorphic validation hierarchy.

Returns:

Type Description
type[MetricsArgs]

Base MetricsArgs class for schema validation

Source code in src/guidellm/schemas/benchmark/entrypoints.py
@classmethod
def __pydantic_schema_base_type__(cls) -> type[MetricsArgs]:
    """
    Return base type for polymorphic validation hierarchy.

    :return: Base MetricsArgs class for schema validation
    """
    if cls.__name__ == "MetricsArgs":
        return cls

    return MetricsArgs

PlotBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Model for Plot benchmark output arguments.

Defines parameters for generating static image visualizations, enforcing image output suffix.

Source code in src/guidellm/schemas/benchmark/outputs/plot.py
@BenchmarkOutputArgs.register("plot")
class PlotBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Model for Plot benchmark output arguments.

    Defines parameters for generating static image visualizations, enforcing
    image output suffix.
    """

    kind: Literal["plot"] = Field(
        default="plot",
        description="Type identifier for the plot configuration.",
    )
    path: Path = Field(
        default_factory=lambda: settings.default_results_dir / "benchmarks.png",
        description="The file to save the output plot to.",
    )
    dpi: int = Field(
        default=100,
        description="Resolution of the output image in Dots Per Inch.",
    )

    @field_validator("path", mode="after")
    @classmethod
    def validate_plot_suffix(cls, v: Path) -> Path:
        """Ensures the output file path ends with a supported plotting format extension.

        If the suffix is missing, it defaults to .png.
        If an unsupported suffix is provided, it raises a ValueError.
        """
        if not v.suffix:
            return v.with_suffix(".png")
        suffix = v.suffix.lower()
        if suffix in _ALLOWED_PLOT_SUFFIXES:
            return v
        raise ValueError(
            f"Plot output type {suffix} is not supported: valid types are "
            f"{', '.join(sorted(_ALLOWED_PLOT_SUFFIXES))}"
        )

validate_plot_suffix(v) classmethod

Ensures the output file path ends with a supported plotting format extension.

If the suffix is missing, it defaults to .png. If an unsupported suffix is provided, it raises a ValueError.

Source code in src/guidellm/schemas/benchmark/outputs/plot.py
@field_validator("path", mode="after")
@classmethod
def validate_plot_suffix(cls, v: Path) -> Path:
    """Ensures the output file path ends with a supported plotting format extension.

    If the suffix is missing, it defaults to .png.
    If an unsupported suffix is provided, it raises a ValueError.
    """
    if not v.suffix:
        return v.with_suffix(".png")
    suffix = v.suffix.lower()
    if suffix in _ALLOWED_PLOT_SUFFIXES:
        return v
    raise ValueError(
        f"Plot output type {suffix} is not supported: valid types are "
        f"{', '.join(sorted(_ALLOWED_PLOT_SUFFIXES))}"
    )

ProfileArgs

Bases: PydanticClassRegistryMixin['ProfileArgs'], ABC

Base class for profile creation arguments.

This class serves as a base for defining argument models used in the creation of profile instances. It inherits from PydanticClassRegistryMixin to enable automatic registration of subclasses, allowing for flexible and extensible profile configurations.

Attributes:

Name Type Description
schema_discriminator str

Field name for polymorphic deserialization

Source code in src/guidellm/schemas/benchmark/profiles/profile.py
class ProfileArgs(PydanticClassRegistryMixin["ProfileArgs"], ABC):
    """Base class for profile creation arguments.

    This class serves as a base for defining argument models used in the creation
    of profile instances. It inherits from PydanticClassRegistryMixin to enable
    automatic registration of subclasses, allowing for flexible and extensible
    profile configurations.

    :cvar schema_discriminator: Field name for polymorphic deserialization
    """

    model_config = standard_model_config()

    schema_discriminator: ClassVar[str] = "kind"

    @classmethod
    def __pydantic_schema_base_type__(cls) -> type[ProfileArgs]:
        """
        Return base type for polymorphic validation hierarchy.

        :return: Base ProfileArgs class for schema validation
        """
        if cls.__name__ == "ProfileArgs":
            return cls

        return ProfileArgs

    kind: str = Field(
        description="Profile type discriminator",
        examples=["concurrent", "synchronous"],
    )
    rampup_duration: NonNegativeFloat = Field(
        default=0.0,
        description=("Duration in seconds to ramp up the targeted scheduling rate"),
    )
    warmup: TransientPhaseConfig = Field(
        default_factory=TransientPhaseConfig,
        description="Warmup phase to exclude initial transient period",
        examples=[0.0, 1.0, {"mode": "percent", "percent": 2.0}],
    )
    cooldown: TransientPhaseConfig = Field(
        default_factory=TransientPhaseConfig,
        description="Cooldown phase to exclude final transient period",
        examples=[0.0, 1.0, {"mode": "duration", "value": 2.0}],
    )

    def validate_metrics(self, metrics: Any) -> None:
        """
        Check the metrics configuration supports this profile.

        Called once the whole benchmark configuration has validated, so a
        profile that needs a particular metric configured can say so before the
        run starts rather than failing partway through. Defaults to accepting
        any configuration.

        :param metrics: Validated metrics arguments for the run
        :raises ValueError: If the metrics configuration cannot support this
            profile
        """

    @field_validator("warmup", "cooldown", mode="before")
    @classmethod
    def _coerce_transient_phase(cls, v: Any) -> Any:
        if isinstance(v, str):
            with contextlib.suppress(json.JSONDecodeError, ValueError):
                v = json.loads(v)
        if isinstance(v, int | float | None):
            return TransientPhaseConfig.create_from_value(v)
        return v

__pydantic_schema_base_type__() classmethod

Return base type for polymorphic validation hierarchy.

Returns:

Type Description
type[ProfileArgs]

Base ProfileArgs class for schema validation

Source code in src/guidellm/schemas/benchmark/profiles/profile.py
@classmethod
def __pydantic_schema_base_type__(cls) -> type[ProfileArgs]:
    """
    Return base type for polymorphic validation hierarchy.

    :return: Base ProfileArgs class for schema validation
    """
    if cls.__name__ == "ProfileArgs":
        return cls

    return ProfileArgs

validate_metrics(metrics)

Check the metrics configuration supports this profile.

Called once the whole benchmark configuration has validated, so a profile that needs a particular metric configured can say so before the run starts rather than failing partway through. Defaults to accepting any configuration.

Parameters:

Name Type Description Default
metrics Any

Validated metrics arguments for the run

required

Raises:

Type Description
ValueError

If the metrics configuration cannot support this profile

Source code in src/guidellm/schemas/benchmark/profiles/profile.py
def validate_metrics(self, metrics: Any) -> None:
    """
    Check the metrics configuration supports this profile.

    Called once the whole benchmark configuration has validated, so a
    profile that needs a particular metric configured can say so before the
    run starts rather than failing partway through. Defaults to accepting
    any configuration.

    :param metrics: Validated metrics arguments for the run
    :raises ValueError: If the metrics configuration cannot support this
        profile
    """

RandomArgs

Bases: PydanticClassRegistryMixin['RandomArgs'], ABC

Base class for random initialization arguments.

Attributes:

Name Type Description
schema_discriminator str

Field name for polymorphic deserialization

Source code in src/guidellm/schemas/benchmark/random.py
class RandomArgs(PydanticClassRegistryMixin["RandomArgs"], ABC):
    """Base class for random initialization arguments.

    :cvar schema_discriminator: Field name for polymorphic deserialization
    """

    model_config = standard_model_config()

    schema_discriminator: ClassVar[str] = "kind"

    @classmethod
    def __pydantic_schema_base_type__(cls) -> type[RandomArgs]:
        """
        Return base type for polymorphic validation hierarchy.

        :return: Base RandomArgs class for schema validation
        """
        if cls.__name__ == "RandomArgs":
            return cls

        return RandomArgs

    kind: str = Field(
        description="The kind of random configuration to use.",
    )

__pydantic_schema_base_type__() classmethod

Return base type for polymorphic validation hierarchy.

Returns:

Type Description
type[RandomArgs]

Base RandomArgs class for schema validation

Source code in src/guidellm/schemas/benchmark/random.py
@classmethod
def __pydantic_schema_base_type__(cls) -> type[RandomArgs]:
    """
    Return base type for polymorphic validation hierarchy.

    :return: Base RandomArgs class for schema validation
    """
    if cls.__name__ == "RandomArgs":
        return cls

    return RandomArgs

ReplayProfileArgs

Bases: ProfileArgs

Pydantic model for trace replay profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/replay.py
@ProfileArgs.register("replay")
class ReplayProfileArgs(ProfileArgs):
    """Pydantic model for trace replay profile creation arguments."""

    kind: Literal["replay"] = Field(
        default="replay",
        description="Profile type discriminator for trace replay scheduling",
    )
    time_scale: float = Field(
        default=1.0,
        gt=0,
        description="Scheduler scale factor applied to relative timestamps",
    )
    schedule_turn: Literal["timestamp", "idle_gap"] = Field(
        default="idle_gap",
        description=(
            "idle_gap (the default) keeps the idle gap after each request's "
            "recorded duration. timestamp schedules each request at its trace "
            "timestamp and waits only while a prior turn is still running."
        ),
    )

SweepProfileArgs

Bases: ProfileArgs

Pydantic model for sweep profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/sweep.py
@ProfileArgs.register("sweep")
class SweepProfileArgs(ProfileArgs):
    """Pydantic model for sweep profile creation arguments."""

    kind: Literal["sweep"] = Field(
        default="sweep",
        description="Profile type discriminator for sweep scheduling",
    )
    sweep_size: int = Field(
        default=10,
        description="Number of strategies to generate for the sweep",
        ge=2,
    )
    strategy_type: Literal["constant", "poisson"] = Field(
        default="constant",
        description="Type of strategy to use for the asynchronous sweep",
    )
    max_concurrency: PositiveInt | None = Field(
        default=512,
        description="Maximum concurrent requests to schedule",
    )

SynchronousProfileArgs

Bases: ProfileArgs

Pydantic model for synchronous profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/synchronous.py
@ProfileArgs.register("synchronous")
class SynchronousProfileArgs(ProfileArgs):
    """Pydantic model for synchronous profile creation arguments."""

    kind: Literal["synchronous"] = Field(
        default="synchronous",
        description="Profile type discriminator for synchronous scheduling",
    )

ThroughputProfileArgs

Bases: ProfileArgs

Pydantic model for throughput profile creation arguments.

Source code in src/guidellm/schemas/benchmark/profiles/throughput.py
@ProfileArgs.register("throughput")
class ThroughputProfileArgs(ProfileArgs):
    """Pydantic model for throughput profile creation arguments."""

    kind: Literal["throughput"] = Field(
        default="throughput",
        description="Profile type discriminator for throughput scheduling",
    )
    max_concurrency: PositiveInt | None = Field(
        description="Maximum concurrent requests to schedule",
        examples=[10],
    )

TransientPhaseConfig

Bases: StandardBaseModel

Configure warmup and cooldown phases for benchmark execution.

Supports flexible phase definition through percentage or absolute value specifications with multiple interpretation modes. Phases can be bounded by duration, request count, or both, enabling precise control over transient periods that should be excluded from final benchmark metrics.

Source code in src/guidellm/schemas/benchmark/transient.py
class TransientPhaseConfig(StandardBaseModel):
    """Configure warmup and cooldown phases for benchmark execution.

    Supports flexible phase definition through percentage or absolute value
    specifications with multiple interpretation modes. Phases can be bounded
    by duration, request count, or both, enabling precise control over transient
    periods that should be excluded from final benchmark metrics.
    """

    @classmethod
    def create_from_value(
        cls, value: int | float | dict | TransientPhaseConfig | None
    ) -> TransientPhaseConfig:
        """
        Create configuration from flexible input formats.

        :param value: Configuration as int/float (percent if <1.0, absolute
            otherwise), dict (validated to model), TransientPhaseConfig instance,
            or None for defaults
        :return: Configured TransientPhaseConfig instance
        :raises ValueError: If value type is unsupported
        """
        if value is None:
            return TransientPhaseConfig()

        if isinstance(value, TransientPhaseConfig):
            return value

        if isinstance(value, dict):
            return TransientPhaseConfig.model_validate(value)

        if isinstance(value, int | float):
            kwargs: dict[str, Any] = {
                "percent": value if value < 1.0 else None,
                "value": value if value >= 1.0 else None,
            }
            return TransientPhaseConfig.model_validate(kwargs)

        raise ValueError(f"Unsupported type for TransientPhaseConfig: {type(value)}")

    percent: NonNegativeFloat | None = Field(
        default=None,
        description=(
            "Phase size as percentage (0.0-1.0) of total duration/requests; "
            "interpretation depends on mode. Takes precedence over value when target "
            "mode is available, otherwise falls back to value"
        ),
        examples=[0.0, 0.5],
        lt=1.0,
    )
    value: NonNegativeInt | NonNegativeFloat | None = Field(
        default=None,
        description=(
            "Phase size as absolute duration (seconds) or request count; "
            "interpretation depends on mode. Used when percent is unset or "
            "target mode unavailable"
        ),
        examples=[1.0, 2.0],
    )
    mode: Literal[
        "duration", "requests", "prefer_duration", "prefer_requests", "both"
    ] = Field(
        default="prefer_duration",
        description=(
            "Interpretation mode: 'duration' for time-based phases, 'requests' for "
            "count-based phases, 'prefer_duration'/'prefer_requests' for fallback "
            "behavior, 'both' requires satisfying both conditions"
        ),
    )

    def compute_limits(
        self,
        max_requests: int | float | None,
        max_seconds: float | None,
        enforce_preference: bool = True,
    ) -> tuple[float | None, int | None]:
        """
        Calculate phase boundaries from benchmark constraints.

        :param max_requests: Total request budget for benchmark execution
        :param max_seconds: Total duration budget for benchmark execution
        :param enforce_preference: Whether to enforce preferred mode when both
            duration and request constraints are available
        :return: Tuple of (phase duration in seconds, phase request count)
        """
        duration: float | None = None
        requests: int | None = None

        if self.mode != "requests" and max_seconds is not None:
            if self.percent is not None:
                duration = self.percent * max_seconds
            elif self.value is not None:
                duration = float(self.value)

        if self.mode != "duration" and max_requests is not None:
            if self.percent is not None:
                requests = int(self.percent * max_requests)
            elif self.value is not None:
                requests = int(self.value)

        if enforce_preference:
            if self.mode == "prefer_duration" and duration is not None:
                requests = None
            elif self.mode == "prefer_requests" and requests is not None:
                duration = None

        return duration, requests

    def compute_transition_time(
        self,
        info: RequestInfo,
        state: SchedulerState,
        period: Literal["start", "end"],
    ) -> tuple[bool, float | None]:
        """
        Determine transition timestamp for entering or exiting phase.

        :param info: RequestInfo for current request to calculate against
        :param state: SchedulerState with current progress metrics and scheduler info
        :param period: Phase period, either "start" for warmup or "end" for cooldown
        :return: Tuple of (phase active flag, transition timestamp if applicable)
        """
        phase_duration, phase_requests = self.compute_limits(
            max_requests=state.progress.total_requests,
            max_seconds=state.progress.total_duration,
        )
        duration_transition_time: float | None = None
        request_transition_time: float | None = None

        # Calculate transition times for the phase based on phase limits and period
        # Potential phases: start (warmup) -> active -> end (cooldown)
        #   Warmup transition times: (start, start + duration)
        #   Active transition times: (start + duration, end - duration)
        #   Cooldown transition times: (end - duration, end)
        if period == "start":
            if phase_duration is not None:
                # Duration was set and caculating for "warmup" / start phase
                # Phase is active for [start, start + duration]
                duration_transition_time = state.start_time + phase_duration
            if phase_requests is not None:
                # Requests was set and calculating for "warmup" / start phase
                # Phase is active for requests [0, phase_requests]
                # Grab start time of the next request as transition time
                # (all requests up to and including phase_requests are in warmup)
                request_transition_time = (
                    info.started_at
                    if info.started_at is not None
                    and state.processed_requests == phase_requests + 1
                    else -1.0
                )
        elif period == "end":
            if phase_duration is not None:
                # Duration was set and calculating for "cooldown" / end phase
                # Phase is active for [end - duration, end]
                duration_transition_time = (
                    state.start_time + state.progress.total_duration - phase_duration
                    if state.progress.total_duration is not None
                    else -1.0
                )
            if phase_requests is not None:
                # Requests was set and calculating for "cooldown" / end phase
                # Phase is active for requests [total - phase_requests, total]
                # Grab completion time of the request right before cooldown starts
                # (all requests from that point onward are in cooldown)
                request_transition_time = (
                    info.completed_at
                    if info.completed_at is not None
                    and state.progress.remaining_requests is not None
                    and state.progress.remaining_requests == phase_requests + 1
                    else -1.0
                )

        transition_active: bool = False
        transition_time: float | None = None

        if request_transition_time == -1.0 or duration_transition_time == -1.0:
            # Transition defined but not yet reached or passed
            transition_active = True
            request_transition_time = None
        elif (
            request_transition_time is not None and duration_transition_time is not None
        ):
            # Both limits defined; need to satisfy both (min for end, max for start)
            transition_active = True
            transition_time = (
                min(request_transition_time, duration_transition_time)
                if period == "end"
                else max(request_transition_time, duration_transition_time)
            )
        elif (
            request_transition_time is not None or duration_transition_time is not None
        ):
            # One limit defined; satisfy that one
            transition_active = True
            transition_time = request_transition_time or duration_transition_time

        return transition_active, transition_time

compute_limits(max_requests, max_seconds, enforce_preference=True)

Calculate phase boundaries from benchmark constraints.

Parameters:

Name Type Description Default
max_requests int | float | None

Total request budget for benchmark execution

required
max_seconds float | None

Total duration budget for benchmark execution

required
enforce_preference bool

Whether to enforce preferred mode when both duration and request constraints are available

True

Returns:

Type Description
tuple[float | None, int | None]

Tuple of (phase duration in seconds, phase request count)

Source code in src/guidellm/schemas/benchmark/transient.py
def compute_limits(
    self,
    max_requests: int | float | None,
    max_seconds: float | None,
    enforce_preference: bool = True,
) -> tuple[float | None, int | None]:
    """
    Calculate phase boundaries from benchmark constraints.

    :param max_requests: Total request budget for benchmark execution
    :param max_seconds: Total duration budget for benchmark execution
    :param enforce_preference: Whether to enforce preferred mode when both
        duration and request constraints are available
    :return: Tuple of (phase duration in seconds, phase request count)
    """
    duration: float | None = None
    requests: int | None = None

    if self.mode != "requests" and max_seconds is not None:
        if self.percent is not None:
            duration = self.percent * max_seconds
        elif self.value is not None:
            duration = float(self.value)

    if self.mode != "duration" and max_requests is not None:
        if self.percent is not None:
            requests = int(self.percent * max_requests)
        elif self.value is not None:
            requests = int(self.value)

    if enforce_preference:
        if self.mode == "prefer_duration" and duration is not None:
            requests = None
        elif self.mode == "prefer_requests" and requests is not None:
            duration = None

    return duration, requests

compute_transition_time(info, state, period)

Determine transition timestamp for entering or exiting phase.

Parameters:

Name Type Description Default
info RequestInfo

RequestInfo for current request to calculate against

required
state SchedulerState

SchedulerState with current progress metrics and scheduler info

required
period Literal['start', 'end']

Phase period, either "start" for warmup or "end" for cooldown

required

Returns:

Type Description
tuple[bool, float | None]

Tuple of (phase active flag, transition timestamp if applicable)

Source code in src/guidellm/schemas/benchmark/transient.py
def compute_transition_time(
    self,
    info: RequestInfo,
    state: SchedulerState,
    period: Literal["start", "end"],
) -> tuple[bool, float | None]:
    """
    Determine transition timestamp for entering or exiting phase.

    :param info: RequestInfo for current request to calculate against
    :param state: SchedulerState with current progress metrics and scheduler info
    :param period: Phase period, either "start" for warmup or "end" for cooldown
    :return: Tuple of (phase active flag, transition timestamp if applicable)
    """
    phase_duration, phase_requests = self.compute_limits(
        max_requests=state.progress.total_requests,
        max_seconds=state.progress.total_duration,
    )
    duration_transition_time: float | None = None
    request_transition_time: float | None = None

    # Calculate transition times for the phase based on phase limits and period
    # Potential phases: start (warmup) -> active -> end (cooldown)
    #   Warmup transition times: (start, start + duration)
    #   Active transition times: (start + duration, end - duration)
    #   Cooldown transition times: (end - duration, end)
    if period == "start":
        if phase_duration is not None:
            # Duration was set and caculating for "warmup" / start phase
            # Phase is active for [start, start + duration]
            duration_transition_time = state.start_time + phase_duration
        if phase_requests is not None:
            # Requests was set and calculating for "warmup" / start phase
            # Phase is active for requests [0, phase_requests]
            # Grab start time of the next request as transition time
            # (all requests up to and including phase_requests are in warmup)
            request_transition_time = (
                info.started_at
                if info.started_at is not None
                and state.processed_requests == phase_requests + 1
                else -1.0
            )
    elif period == "end":
        if phase_duration is not None:
            # Duration was set and calculating for "cooldown" / end phase
            # Phase is active for [end - duration, end]
            duration_transition_time = (
                state.start_time + state.progress.total_duration - phase_duration
                if state.progress.total_duration is not None
                else -1.0
            )
        if phase_requests is not None:
            # Requests was set and calculating for "cooldown" / end phase
            # Phase is active for requests [total - phase_requests, total]
            # Grab completion time of the request right before cooldown starts
            # (all requests from that point onward are in cooldown)
            request_transition_time = (
                info.completed_at
                if info.completed_at is not None
                and state.progress.remaining_requests is not None
                and state.progress.remaining_requests == phase_requests + 1
                else -1.0
            )

    transition_active: bool = False
    transition_time: float | None = None

    if request_transition_time == -1.0 or duration_transition_time == -1.0:
        # Transition defined but not yet reached or passed
        transition_active = True
        request_transition_time = None
    elif (
        request_transition_time is not None and duration_transition_time is not None
    ):
        # Both limits defined; need to satisfy both (min for end, max for start)
        transition_active = True
        transition_time = (
            min(request_transition_time, duration_transition_time)
            if period == "end"
            else max(request_transition_time, duration_transition_time)
        )
    elif (
        request_transition_time is not None or duration_transition_time is not None
    ):
        # One limit defined; satisfy that one
        transition_active = True
        transition_time = request_transition_time or duration_transition_time

    return transition_active, transition_time

create_from_value(value) classmethod

Create configuration from flexible input formats.

Parameters:

Name Type Description Default
value int | float | dict | TransientPhaseConfig | None

Configuration as int/float (percent if <1.0, absolute otherwise), dict (validated to model), TransientPhaseConfig instance, or None for defaults

required

Returns:

Type Description
TransientPhaseConfig

Configured TransientPhaseConfig instance

Raises:

Type Description
ValueError

If value type is unsupported

Source code in src/guidellm/schemas/benchmark/transient.py
@classmethod
def create_from_value(
    cls, value: int | float | dict | TransientPhaseConfig | None
) -> TransientPhaseConfig:
    """
    Create configuration from flexible input formats.

    :param value: Configuration as int/float (percent if <1.0, absolute
        otherwise), dict (validated to model), TransientPhaseConfig instance,
        or None for defaults
    :return: Configured TransientPhaseConfig instance
    :raises ValueError: If value type is unsupported
    """
    if value is None:
        return TransientPhaseConfig()

    if isinstance(value, TransientPhaseConfig):
        return value

    if isinstance(value, dict):
        return TransientPhaseConfig.model_validate(value)

    if isinstance(value, int | float):
        kwargs: dict[str, Any] = {
            "percent": value if value < 1.0 else None,
            "value": value if value >= 1.0 else None,
        }
        return TransientPhaseConfig.model_validate(kwargs)

    raise ValueError(f"Unsupported type for TransientPhaseConfig: {type(value)}")

YAMLBenchmarkOutputArgs

Bases: BenchmarkOutputArgs

Model for YAML benchmark output arguments.

Source code in src/guidellm/schemas/benchmark/outputs/serialized.py
@BenchmarkOutputArgs.register("yaml")
class YAMLBenchmarkOutputArgs(BenchmarkOutputArgs):
    """Model for YAML benchmark output arguments."""

    kind: Literal["yaml"] = Field(
        default="yaml",
        description="The kind of output.",
        examples=["yaml"],
    )
    path: Path = Field(
        default_factory=lambda: settings.default_results_dir / "benchmarks.yaml",
        description="The file to save the output to.",
        examples=["./benchmarks.yaml"],
    )

default_kind(kind)

Default factory for argument models to set the 'kind' field.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
def default_kind(kind: str) -> dict[str, Any]:
    """Default factory for argument models to set the 'kind' field."""
    return {"kind": kind}

default_kind_list(*kinds)

Default factory for lists of argument models to set the 'kind' field.

Source code in src/guidellm/schemas/benchmark/entrypoints.py
def default_kind_list(*kinds: str) -> list[dict[str, Any]]:
    """Default factory for lists of argument models to set the 'kind' field."""
    return [default_kind(kind) for kind in kinds]

get_builtin_scenarios() cached

Retrieve all builtin scenario definitions from the scenarios directory.

Scans the scenarios directory for JSON files and returns a mapping of scenario names to their file paths. Each scenario is indexed by both its stem name (filename without extension) for convenient lookup.

Returns:

Type Description
dict[str, Path]

Dictionary mapping scenario names and filenames to their Path objects

Source code in src/guidellm/schemas/benchmark/scenarios/__init__.py
@cache
def get_builtin_scenarios() -> dict[str, Path]:
    """
    Retrieve all builtin scenario definitions from the scenarios directory.

    Scans the scenarios directory for JSON files and returns a mapping of scenario
    names to their file paths. Each scenario is indexed by both its stem name
    (filename without extension) for convenient lookup.

    :return: Dictionary mapping scenario names and filenames to their Path objects
    """
    builtin = {}
    for path in SCENARIO_DIR.rglob("*.json"):
        shorthand = str(path.relative_to(SCENARIO_DIR).with_suffix(""))
        builtin[shorthand] = path

    return builtin