Skip to content

guidellm.entrypoints

Contains entrypoints for GuideLLM submodules.

Each entrypoint is lazy loaded to avoid unnecessary imports and dependencies. This is important to ensure a resposive CLI and lightweight worker spawning.

MockServer

High-performance mock server implementing OpenAI and vLLM API endpoints.

Provides a Sanic-based web server that simulates API responses with configurable timing characteristics for testing and benchmarking purposes. Supports chat completions, text completions, tokenization endpoints, and model listing with realistic latency patterns to enable comprehensive performance validation.

Example: :: config = ServerConfig(model="test-model", port=8080) server = MockServer(config) server.run()

Source code in src/guidellm/mock_server/server.py
class MockServer:
    """
    High-performance mock server implementing OpenAI and vLLM API endpoints.

    Provides a Sanic-based web server that simulates API responses with configurable
    timing characteristics for testing and benchmarking purposes. Supports chat
    completions, text completions, tokenization endpoints, and model listing with
    realistic latency patterns to enable comprehensive performance validation.

    Example:
    ::
        config = ServerConfig(model="test-model", port=8080)
        server = MockServer(config)
        server.run()
    """

    def __init__(self, config: MockServerConfig) -> None:
        """
        Initialize the mock server with configuration.

        :param config: Server configuration containing network settings and response
            timing parameters
        """
        self.config = config
        self.app = Sanic("guidellm-mock-server")
        self.chat_handler = ChatCompletionsHandler(config)
        self.completions_handler = CompletionsHandler(config)
        self.responses_handler = ResponsesHandler(config)
        self.tokenizer_handler = TokenizerHandler(config)
        # Deterministic test controls: count accepted generation requests and
        # optionally serialize them with a semaphore (lazy-created in the loop).
        self._accepted_generation_requests = 0
        self._concurrency_semaphore: asyncio.Semaphore | None = None

        self._setup_middleware()
        self._setup_routes()
        self._setup_error_handlers()

    def _get_concurrency_semaphore(self) -> asyncio.Semaphore | None:
        """Return the concurrency semaphore, creating it in the running loop."""
        if self.config.max_concurrent_requests is None:
            return None
        if self._concurrency_semaphore is None:
            self._concurrency_semaphore = asyncio.Semaphore(
                self.config.max_concurrent_requests
            )
        return self._concurrency_semaphore

    async def _run_generation(
        self,
        handler: Callable[[Request], Awaitable[HTTPResponse]],
        request: Request,
    ) -> HTTPResponse:
        """
        Run a generation handler with optional fail-after and concurrency limits.

        :param handler: Async generation endpoint handler
        :param request: Incoming Sanic request
        :return: Handler response, or HTTP 500 when fail_after_requests is exceeded
        """
        fail_after = self.config.fail_after_requests
        if fail_after is not None and self._accepted_generation_requests >= fail_after:
            return response.json(
                {
                    "error": {
                        "message": (
                            f"Mock server fail_after_requests={fail_after} exceeded"
                        ),
                        "type": "server_error",
                        "code": "fail_after_requests",
                    }
                },
                status=500,
            )

        self._accepted_generation_requests += 1
        semaphore = self._get_concurrency_semaphore()
        if semaphore is None:
            return await handler(request)
        async with semaphore:
            return await handler(request)

    def _audio_usage(self, file: File, text: str) -> dict[str, int | float]:
        """
        Build usage statistics for an audio endpoint response.

        Charges prompt tokens for the uploaded audio using the configured
        per-second rate applied to the duration estimated from the payload
        size, and counts completion tokens from the generated text.

        :param file: Uploaded audio file from the multipart form
        :param text: Text returned in the response body
        :return: Usage dict with prompt, completion, and total token counts
        """
        audio_seconds = estimate_audio_seconds(len(file.body), file.type or file.name)
        prompt_tokens = math.ceil(audio_seconds * self.config.audio_tokens_per_second)
        completion_tokens = len(self.tokenizer_handler.tokenizer.tokenize(text))
        return {
            "prompt_tokens": prompt_tokens,
            "completion_tokens": completion_tokens,
            "total_tokens": prompt_tokens + completion_tokens,
            "seconds": round(audio_seconds, 3),
        }

    def _setup_middleware(self):
        """Setup middleware for CORS, logging, etc."""

        @self.app.middleware("request")
        async def log_request_received(request: Request) -> None:
            """Log request arrival when log_request_received is enabled."""
            if self.config.log_request_received:
                logger.info(
                    "Request received: %s %s from %s",
                    request.method,
                    request.path,
                    request.ip,
                )

        @self.app.middleware("request")
        async def add_cors_headers(_request: Request) -> None:
            """Add CORS headers to all requests."""
            return None  # noqa: RET501

        @self.app.middleware("response")
        async def add_response_headers(
            _request: Any, resp: BaseHTTPResponse
        ) -> HTTPResponse:
            """Add standard response headers."""
            resp.headers["Access-Control-Allow-Origin"] = "*"
            resp.headers["Access-Control-Allow-Methods"] = "GET, POST, OPTIONS"
            resp.headers["Access-Control-Allow-Headers"] = "Content-Type, Authorization"
            resp.headers["Server"] = "guidellm-mock-server"
            return resp  # type: ignore[return-value]

    def _setup_routes(self):  # noqa: C901
        @self.app.get("/health")
        async def health_check(_request: Request):
            return response.json({"status": "healthy", "timestamp": time.time()})

        @self.app.get("/v1/models")
        async def list_models(_request: Request):
            return response.json(
                {
                    "object": "list",
                    "data": [
                        {
                            "id": self.config.model,
                            "object": "model",
                            "created": int(time.time()),
                            "owned_by": "guidellm-mock",
                        }
                    ],
                }
            )

        @self.app.route("/v1/chat/completions", methods=["POST", "OPTIONS"])
        async def chat_completions(request: Request):
            if request.method == "OPTIONS":
                return response.text("", status=204)
            return await self._run_generation(self.chat_handler.handle, request)

        @self.app.route("/v1/completions", methods=["POST", "OPTIONS"])
        async def completions(request: Request):
            if request.method == "OPTIONS":
                return response.text("", status=204)
            return await self._run_generation(self.completions_handler.handle, request)

        @self.app.route("/v1/responses", methods=["POST", "OPTIONS"])
        async def responses(request: Request):
            if request.method == "OPTIONS":
                return response.text("", status=204)
            return await self._run_generation(self.responses_handler.handle, request)

        @self.app.route("/tokenize", methods=["POST", "OPTIONS"])
        async def tokenize(request: Request):
            if request.method == "OPTIONS":
                return response.text("", status=204)
            return await self.tokenizer_handler.tokenize(request)

        @self.app.route("/detokenize", methods=["POST", "OPTIONS"])
        async def detokenize(request: Request):
            if request.method == "OPTIONS":
                return response.text("", status=204)
            return await self.tokenizer_handler.detokenize(request)

        @self.app.route("/v1/audio/transcriptions", methods=["POST", "OPTIONS"])
        async def audio_transcriptions(request: Request) -> HTTPResponse:
            """
            Mock OpenAI audio transcription endpoint:
            - receives multipart/form-data
            - file field contains audio file
            - model field is optional, default to "mock-model"
            - returns "transcribed text"
            """
            if request.method == "OPTIONS":
                return response.text("", status=204)
            if request.files is None or request.form is None:
                return response.json({"error": "No form data provided"}, status=400)
            file: File | None = request.files.get("file")
            if "file" not in request.files or "model" not in request.form:
                return response.json(
                    {"error": "Missing 'file' in form-data"}, status=400
                )

            file = cast("File", file)
            model = request.form.get("model", "mock-model")
            text = f"Mock transcription for {file.name}"

            return response.json(
                {
                    "text": text,
                    "file_size": len(file.body),
                    "model_used": model,
                    "transcription": f"Transcribed({file.name}) using {model}",
                    "usage": self._audio_usage(file, text),
                }
            )

        @self.app.route("/v1/audio/translations", methods=["POST", "OPTIONS"])
        async def audio_translations(request: Request) -> HTTPResponse:
            """
            Mock OpenAI audio translation endpoint:
            - receives multipart/form-data
            - file field contains audio file
            - model field is optional, default to "mock-model"
            - returns translated text
            """
            if request.method == "OPTIONS":
                return response.text("", status=204)
            if request.files is None or request.form is None:
                return response.json({"error": "No form data provided"}, status=400)
            file: File | None = request.files.get("file")
            if "file" not in request.files or "model" not in request.form:
                return response.json(
                    {"error": "Missing 'file' in form-data"}, status=400
                )

            file = cast("File", file)
            decoded_text = (
                "This is a mock translation result."  # mock output tranlated text
            )

            return response.json(
                {
                    "text": decoded_text,
                    "file_size": len(file.body),
                    "filename": file.name,
                    "model_used": request.form.get("model", "mock-model"),
                    "mimetype": file.type,
                    "usage": self._audio_usage(file, decoded_text),
                }
            )

    def _setup_error_handlers(self):
        """Setup error handlers."""

        @self.app.exception(Exception)
        async def generic_error_handler(_request: Request, exception: Exception):
            logger.error("Unhandled exception: %s", exception)
            return response.json(
                {
                    "error": {
                        "message": "Internal server error",
                        "type": type(exception).__name__,
                        "error": str(exception),
                    }
                },
                status=500,
            )

        @self.app.exception(NotFound)
        async def not_found_handler(_request: Request, _exception):
            return response.json(
                {
                    "error": {
                        "message": "Not Found",
                        "type": "not_found_error",
                        "code": "not_found",
                    }
                },
                status=404,
            )

    def run(self, *, access_log: bool = True) -> None:
        """
        Start the mock server with configured settings.

        Runs the Sanic application in single-process mode with access logging and
        the Sanic startup MOTD enabled by default. Pass ``access_log=False`` in
        shared-TTY test runners to disable access logs and the MOTD, and to use
        plain (non-ANSI) Sanic formatters so start/stop logs do not overwrite
        the terminal.

        :param access_log: Whether to enable Sanic per-request access logging and
            the startup MOTD banner. When false, start/stop logs still print but
            without ANSI cursor controls.
        """
        # Reconfigure after Sanic.__init__ so quiet mode applies before serve.
        _configure_sanic_logging(access_log=access_log)
        self.app.run(
            host=self.config.host,
            port=self.config.port,
            debug=False,
            single_process=True,
            access_log=access_log,
            motd=access_log,
            register_sys_signals=True,
        )

__init__(config)

Initialize the mock server with configuration.

Parameters:

Name Type Description Default
config MockServerConfig

Server configuration containing network settings and response timing parameters

required
Source code in src/guidellm/mock_server/server.py
def __init__(self, config: MockServerConfig) -> None:
    """
    Initialize the mock server with configuration.

    :param config: Server configuration containing network settings and response
        timing parameters
    """
    self.config = config
    self.app = Sanic("guidellm-mock-server")
    self.chat_handler = ChatCompletionsHandler(config)
    self.completions_handler = CompletionsHandler(config)
    self.responses_handler = ResponsesHandler(config)
    self.tokenizer_handler = TokenizerHandler(config)
    # Deterministic test controls: count accepted generation requests and
    # optionally serialize them with a semaphore (lazy-created in the loop).
    self._accepted_generation_requests = 0
    self._concurrency_semaphore: asyncio.Semaphore | None = None

    self._setup_middleware()
    self._setup_routes()
    self._setup_error_handlers()

run(*, access_log=True)

Start the mock server with configured settings.

Runs the Sanic application in single-process mode with access logging and the Sanic startup MOTD enabled by default. Pass access_log=False in shared-TTY test runners to disable access logs and the MOTD, and to use plain (non-ANSI) Sanic formatters so start/stop logs do not overwrite the terminal.

Parameters:

Name Type Description Default
access_log bool

Whether to enable Sanic per-request access logging and the startup MOTD banner. When false, start/stop logs still print but without ANSI cursor controls.

True
Source code in src/guidellm/mock_server/server.py
def run(self, *, access_log: bool = True) -> None:
    """
    Start the mock server with configured settings.

    Runs the Sanic application in single-process mode with access logging and
    the Sanic startup MOTD enabled by default. Pass ``access_log=False`` in
    shared-TTY test runners to disable access logs and the MOTD, and to use
    plain (non-ANSI) Sanic formatters so start/stop logs do not overwrite
    the terminal.

    :param access_log: Whether to enable Sanic per-request access logging and
        the startup MOTD banner. When false, start/stop logs still print but
        without ANSI cursor controls.
    """
    # Reconfigure after Sanic.__init__ so quiet mode applies before serve.
    _configure_sanic_logging(access_log=access_log)
    self.app.run(
        host=self.config.host,
        port=self.config.port,
        debug=False,
        single_process=True,
        access_log=access_log,
        motd=access_log,
        register_sys_signals=True,
    )

benchmark_generative_text(args, progress=True, console=None, **constraints) async

Execute a comprehensive generative text benchmarking workflow.

Orchestrates the full benchmarking pipeline by resolving all components from provided arguments, executing benchmark runs across configured profiles, and finalizing results in specified output formats. Components include backend initialization, data loading, profile configuration, and output generation.

Parameters:

Name Type Description Default
args BenchmarkScenario

Scenario configuration for the benchmark execution

required
progress bool

Progress tracker for benchmark execution, or None for no tracking

True
console Console | None

Console instance for status reporting, or None for silent operation

None
constraints str | ConstraintInitializer | Any

Additional constraint initializers for benchmark limits

{}

Returns:

Type Description
tuple[GenerativeBenchmarksReport, list[tuple[str, Any]]]

Tuple of GenerativeBenchmarksReport and dictionary of output format results

Source code in src/guidellm/benchmark/entrypoints.py
async def benchmark_generative_text(
    args: BenchmarkScenario,
    progress: bool = True,
    console: Console | None = None,
    **constraints: str | ConstraintInitializer | Any,
) -> tuple[GenerativeBenchmarksReport, list[tuple[str, Any]]]:
    """
    Execute a comprehensive generative text benchmarking workflow.

    Orchestrates the full benchmarking pipeline by resolving all components from
    provided arguments, executing benchmark runs across configured profiles, and
    finalizing results in specified output formats. Components include backend
    initialization, data loading, profile configuration, and output generation.

    :param args: Scenario configuration for the benchmark execution
    :param progress: Progress tracker for benchmark execution, or None for no tracking
    :param console: Console instance for status reporting, or None for silent operation
    :param constraints: Additional constraint initializers for benchmark limits
    :return: Tuple of GenerativeBenchmarksReport and dictionary of output format
        results
    """
    trackers: list[
        BenchmarkerProgress[GenerativeBenchmarkAccumulator, GenerativeBenchmark]
    ] = [GenerativeLoggingBenchmarkerProgress()]
    if progress:
        trackers.append(GenerativeConsoleBenchmarkerProgress())

    benchmark_args = resolve_to_single_benchmark(args.get_benchmarks())

    metrics_args = benchmark_args.metrics
    if not isinstance(metrics_args, GenerativeMetricsArgs):
        raise TypeError(
            f"Expected GenerativeMetricsArgs for generative text benchmark, "
            f"got {type(metrics_args).__name__}"
        )

    backend, model = await resolve_backend(
        backend_args=benchmark_args.backend,
        console=console,
    )
    await resolve_tokenizer(args=benchmark_args, model=model, console=console)
    request_loader: DataLoader[GenerationRequest] = await create_data_loader(
        loader_config=benchmark_args.data_loader,
        data_config=benchmark_args.data,
        tokenizer_config=benchmark_args.tokenizer,
        column_mapper_config=benchmark_args.data_column_mapper,
        preprocessors_config=benchmark_args.data_preprocessors,
        finalizer_config=benchmark_args.data_finalizer,
        random_seed=benchmark_args.seed.value,  # type: ignore[attr-defined]
        console=console,
    )

    warmup = benchmark_args.profile.warmup
    cooldown = benchmark_args.profile.cooldown

    constraints = resolve_constraints(benchmark_args, **constraints)
    profile = await resolve_profile(
        profile=benchmark_args.profile,
        constraints=constraints,
        console=console,
        random_seed=benchmark_args.seed.value,  # type: ignore[attr-defined]
    )
    output_formats = await resolve_output_formats(
        outputs=benchmark_args.outputs, console=console
    )

    report = GenerativeBenchmarksReport(config=args)
    if console:
        console.print_update(
            title="Setup complete, starting benchmarks...", status="success"
        )
        console.print("\n\n")

    benchmarker: Benchmarker[
        GenerativeBenchmark, GenerationRequest, GenerationResponse
    ] = Benchmarker()
    async for benchmark in benchmarker.run(
        accumulator_class=GenerativeBenchmarkAccumulator,
        benchmark_class=GenerativeBenchmark,
        requests=request_loader,  # type: ignore[arg-type]
        backend=backend,
        profile=profile,
        environment=NonDistributedEnvironment(),
        progress=CompositeBenchmarkerProgress(trackers),
        sample_size=metrics_args.sample_size,
        warmup=warmup,
        cooldown=cooldown,
        prefer_response_metrics=metrics_args.prefer_response_metrics,
        slo=metrics_args.slo,
        confidence=metrics_args.confidence,
    ):
        if benchmark:
            report.benchmarks.append(benchmark)

    # Read after the final strategy so the conclusion reflects every benchmark,
    # including the last, whose config was captured before it ran.
    if (conclusion := profile.conclusion) is not None:
        report.conclusions.append(conclusion)

    output_format_results: list[tuple[str, Any]] = []
    for output_arg, output in zip(benchmark_args.outputs, output_formats, strict=True):
        output_format_results.append((output_arg.kind, await output.finalize(report)))

    if console:
        await GenerativeBenchmarkerConsole(console=console).finalize(report)
        console.print("\n\n")
        console.print_update(
            title=(
                "Benchmarking complete, generated "
                f"{len(report.benchmarks)} benchmark(s)"
            ),
            status="success",
        )
        for kind, value in output_format_results:
            console.print_update(title=f"  {kind:<8}: {value}", status="debug")

    return report, output_format_results

process_dataset(data, output_path, tokenizer, strategy, data_column_mapper=None, data_loader=None, push_to_hub=False, hub_dataset_id=None, random_seed=42)

Main method to process and save a dataset with sampled prompt/output token counts.

Parameters:

Name Type Description Default
data DataArgs | dict[str, Any]

Dataset source configuration (DataArgs or equivalent dict).

required
output_path str | Path

File path to save the processed dataset.

required
tokenizer DataTokenizerArgs | dict[str, Any]

Tokenizer configuration (DataTokenizerArgs or dict).

required
strategy PreprocessStrategyArgs | dict[str, Any]

Preprocess strategy configuration including token targets and short-prompt handling (PreprocessStrategyArgs or dict).

required
data_column_mapper DataPreprocessorArgs | dict[str, Any] | None

Optional column mapping configuration.

None
data_loader DataLoaderArgs | dict[str, Any] | None

Optional data loader configuration. samples limits how many processed rows are written; shuffle and num_workers are ignored.

None
push_to_hub bool

Whether to push to Hugging Face Hub.

False
hub_dataset_id str | None

Dataset ID on Hugging Face Hub.

None
random_seed int

Seed for random sampling.

42

Raises:

Type Description
ValueError

If the output path is invalid or pushing conditions unmet.

Source code in src/guidellm/data/entrypoints.py
def process_dataset(
    data: DataArgs | dict[str, Any],
    output_path: str | Path,
    tokenizer: DataTokenizerArgs | dict[str, Any],
    strategy: PreprocessStrategyArgs | dict[str, Any],
    data_column_mapper: DataPreprocessorArgs | dict[str, Any] | None = None,
    data_loader: DataLoaderArgs | dict[str, Any] | None = None,
    push_to_hub: bool = False,
    hub_dataset_id: str | None = None,
    random_seed: int = 42,
) -> None:
    """
    Main method to process and save a dataset with sampled prompt/output token counts.

    :param data: Dataset source configuration (``DataArgs`` or equivalent dict).
    :param output_path: File path to save the processed dataset.
    :param tokenizer: Tokenizer configuration (``DataTokenizerArgs`` or dict).
    :param strategy: Preprocess strategy configuration including token targets and
        short-prompt handling (``PreprocessStrategyArgs`` or dict).
    :param data_column_mapper: Optional column mapping configuration.
    :param data_loader: Optional data loader configuration. ``samples`` limits how
        many processed rows are written; ``shuffle`` and ``num_workers`` are ignored.
    :param push_to_hub: Whether to push to Hugging Face Hub.
    :param hub_dataset_id: Dataset ID on Hugging Face Hub.
    :param random_seed: Seed for random sampling.
    :raises ValueError: If the output path is invalid or pushing conditions unmet.
    """
    data_config = DataArgs.model_validate(data)
    tokenizer_config = DataTokenizerArgs.model_validate(tokenizer)
    strategy_config = PreprocessStrategyArgs.model_validate(strategy)
    column_mapper_config = DataPreprocessorArgs.model_validate(
        data_column_mapper
        if data_column_mapper is not None
        else {"kind": "generative_column_mapper"}
    )
    loader_config = DataLoaderArgs.model_validate(
        data_loader if data_loader is not None else {"kind": "pytorch"}
    )
    builders.process_dataset(
        data_config,
        output_path,
        tokenizer_config,
        strategy_config,
        column_mapper_config,
        push_to_hub,
        hub_dataset_id,
        random_seed,
        loader_config,
    )

reimport_benchmarks_report(file, outputs) async

Load and re-export an existing benchmarks report in specified output formats.

Parameters:

Name Type Description Default
file Path

Path to the existing benchmark report file to load

required
outputs tuple[BenchmarkOutputArgs, ...] | list[dict[str, Any]]

Output format kind strings to resolve and finalize

required

Returns:

Type Description
tuple[GenerativeBenchmarksReport, list[tuple[str, Any]]]

Tuple of loaded GenerativeBenchmarksReport and dictionary of output results

Source code in src/guidellm/benchmark/entrypoints.py
async def reimport_benchmarks_report(
    file: Path,
    outputs: tuple[BenchmarkOutputArgs, ...] | list[dict[str, Any]],
) -> tuple[GenerativeBenchmarksReport, list[tuple[str, Any]]]:
    """
    Load and re-export an existing benchmarks report in specified output formats.

    :param file: Path to the existing benchmark report file to load
    :param outputs: Output format kind strings to resolve and finalize
    :return: Tuple of loaded GenerativeBenchmarksReport and dictionary of output
        results
    """
    console = Console()

    with console.print_update_step(
        title=f"Loading benchmarks from {file}..."
    ) as console_step:
        report = GenerativeBenchmarksReport.load_file(file)
        console_step.finish(
            "Import of old benchmarks complete;"
            f" loaded {len(report.benchmarks)} benchmark(s)"
        )

    output_args: list[BenchmarkOutputArgs] = []
    for fmt in outputs:
        output_args.append(BenchmarkOutputArgs.model_validate(fmt))

    output_results: list[tuple[str, Any]] = []
    for args in output_args:
        output = GenerativeBenchmarkerOutput.resolve(args)
        output_results.append((args.kind, await output.finalize(report)))

    for kind, value in output_results:
        console.print_update(title=f"  {kind:<8}: {value}", status="debug")

    return report, output_results