Skip to content

guidellm.schemas.base.info

Core data structures and interfaces for the GuideLLM scheduler system.

Provides type-safe abstractions for distributed request processing, timing measurements, and backend interfaces for benchmarking operations. Central to the scheduler architecture, enabling request lifecycle tracking, backend coordination, and state management across distributed worker processes.

RequestInfo

Bases: StandardBaseModel

Complete information about a request in the scheduler system.

Encapsulates all metadata, status tracking, and timing information for requests processed through the distributed scheduler. Provides comprehensive lifecycle tracking from initial queuing through final completion, including error handling and node identification for debugging and performance analysis.

Example: :: request = RequestInfo() request.status = "in_progress" start_time = request.started_at completion_time = request.completed_at

Source code in src/guidellm/schemas/base/info.py
class RequestInfo(StandardBaseModel):
    """
    Complete information about a request in the scheduler system.

    Encapsulates all metadata, status tracking, and timing information for requests
    processed through the distributed scheduler. Provides comprehensive lifecycle
    tracking from initial queuing through final completion, including error handling
    and node identification for debugging and performance analysis.

    Example:
    ::
        request = RequestInfo()
        request.status = "in_progress"
        start_time = request.started_at
        completion_time = request.completed_at
    """

    request_id: str = Field(
        description="Unique identifier for the request",
        default_factory=lambda: str(uuid.uuid4()),
    )
    conversation_id: str | None = Field(
        default=None,
        description=(
            "Identifier for the conversation this request is part of, if applicable."
        ),
    )
    history_len: int = Field(
        default=0,
        description=(
            "Number of prior messages in the assembled history sent to the "
            "server with this request. May include messages from diverging "
            "or merged paths (``last`` edges). Branches that start with a "
            "``new`` edge restart at 0."
        ),
    )
    turn_index: int = Field(
        default=0,
        description=(
            "Path-depth turn index for this request: longest path that "
            "resets on ``new`` edges, increments through ``full`` edges, "
            "and treats each ``last`` edge as adding up to 1 without "
            "recursion. Multiple parents take the maximum."
        ),
    )
    preceding_nodes: int = Field(
        default=0,
        description=(
            "Count of graph nodes that precede this node in topological "
            "execution order (0-based). Independent of history_context."
        ),
    )
    node_id: str | None = Field(
        default=None,
        description="Node ID within a conversation graph, if applicable.",
    )
    agent_id: str | None = Field(
        default=None,
        description="Identifier for the simulated agent that owns this request.",
    )
    parent_node_ids: list[str] = Field(
        default_factory=list,
        description="Node IDs of direct predecessors in the DAG.",
    )
    status: Literal[
        "queued",
        "pending",
        "in_progress",
        "first_token",
        "completed",
        "errored",
        "cancelled",
    ] = Field(description="Current processing status of the request", default="queued")
    scheduler_node_id: int = Field(
        description="ID/rank of the scheduler node handling the request",
        default=-1,
    )
    scheduler_process_id: int = Field(
        description="ID/rank of the node's scheduler process handling the request",
        default=-1,
    )
    scheduler_start_time: float = Field(
        description="Unix timestamp when scheduler processing began",
        default=-1,
    )
    timings: RequestTimings = Field(
        default_factory=RequestTimings,
        description="Timing measurements for the request lifecycle",
    )
    settings: RequestSettings = Field(
        default_factory=RequestSettings,
        description="Per-request scheduling metadata for strategy interpretation",
    )

    error: str | None = Field(
        default=None, description="Error message if the request status is 'errored'"
    )
    traceback: str | None = Field(
        default=None,
        description="Full traceback of the error if the request status is 'errored'",
    )

    @computed_field  # type: ignore[misc]
    @property
    def started_at(self) -> float | None:
        """
        Get the effective request processing start time.

        :return: Unix timestamp when processing began, or None if not started
        """
        return self.timings.request_start or self.timings.resolve_start

    @computed_field  # type: ignore[misc]
    @property
    def completed_at(self) -> float | None:
        """
        Get the effective request processing completion time.

        :return: Unix timestamp when processing completed, or None if not completed
        """
        return self.timings.request_end or self.timings.resolve_end

    def model_copy(self, **_kwargs) -> RequestInfo:  # type: ignore[override]  # noqa: ARG002
        """
        Create a deep copy of the request info with copied timing objects.

        :param kwargs: Additional keyword arguments for model copying
        :return: New RequestInfo instance with independent timing objects
        """
        return super().model_copy(
            update={
                "timings": self.timings.model_copy(),
                "settings": self.settings.model_copy(),
            },
            deep=False,
        )

completed_at property

Get the effective request processing completion time.

Returns:

Type Description
float | None

Unix timestamp when processing completed, or None if not completed

started_at property

Get the effective request processing start time.

Returns:

Type Description
float | None

Unix timestamp when processing began, or None if not started

model_copy(**_kwargs)

Create a deep copy of the request info with copied timing objects.

Parameters:

Name Type Description Default
kwargs

Additional keyword arguments for model copying

required

Returns:

Type Description
RequestInfo

New RequestInfo instance with independent timing objects

Source code in src/guidellm/schemas/base/info.py
def model_copy(self, **_kwargs) -> RequestInfo:  # type: ignore[override]  # noqa: ARG002
    """
    Create a deep copy of the request info with copied timing objects.

    :param kwargs: Additional keyword arguments for model copying
    :return: New RequestInfo instance with independent timing objects
    """
    return super().model_copy(
        update={
            "timings": self.timings.model_copy(),
            "settings": self.settings.model_copy(),
        },
        deep=False,
    )

RequestSettings

Bases: StandardBaseDict

Per-request scheduling metadata attached at enqueue, before worker dequeue.

Populated by dataset finalizers (for example from trace relative_timestamp columns). Scheduling strategies read these fields at dequeue. For trace replay, a non-null relative_timestamp becomes an absolute start time at dequeue: start_time + relative_timestamp. When relative_timestamp is null, trace replay schedules the request at benchmark start (no trace offset).

Source code in src/guidellm/schemas/base/info.py
class RequestSettings(StandardBaseDict):
    """
    Per-request scheduling metadata attached at enqueue, before worker dequeue.

    Populated by dataset finalizers (for example from trace ``relative_timestamp``
    columns). Scheduling strategies read these fields at dequeue. For trace replay,
    a non-null ``relative_timestamp`` becomes an absolute start time at dequeue:
    ``start_time + relative_timestamp``. When ``relative_timestamp``
    is null, trace replay schedules the request at benchmark start (no trace offset).
    """

    relative_timestamp: float | None = Field(
        default=None,
        ge=0,
        description=(
            "Trace offset in seconds from the first event after sorting (0 for the "
            "earliest event). Trace replay converts this to an absolute start time "
            "at dequeue: start_time + relative_timestamp. When null, "
            "trace replay uses benchmark start time only."
        ),
    )
    requeue_delay: float | None = Field(
        default=None,
        gt=0,
        description=(
            "Delay in seconds before requeueing the conversation. This number is a "
            "lower bound on the delay, subject to scheduling."
        ),
    )
    trace_duration: float | None = Field(
        default=None,
        ge=0,
        description=(
            "Recorded request duration in seconds from the trace, in the same "
            "units as relative_timestamp after dataset time scaling. None when "
            "the trace has no duration column. schedule_turn=idle_gap uses this "
            "to keep the idle gap before the next request; a missing value is "
            "treated as instantaneous."
        ),
    )

RequestTimings

Bases: StandardBaseDict

Timing measurements for tracking request lifecycle events.

Provides comprehensive timing data for distributed request processing, capturing key timestamps from initial targeting through final completion. Essential for performance analysis, SLA monitoring, and debugging request processing bottlenecks across scheduler workers and backend systems.

Source code in src/guidellm/schemas/base/info.py
class RequestTimings(StandardBaseDict):
    """
    Timing measurements for tracking request lifecycle events.

    Provides comprehensive timing data for distributed request processing, capturing
    key timestamps from initial targeting through final completion. Essential for
    performance analysis, SLA monitoring, and debugging request processing bottlenecks
    across scheduler workers and backend systems.
    """

    targeted_start: float | None = Field(
        default=None,
        description="Unix timestamp when request was initially targeted for execution",
    )
    predecessor_completed: float | None = Field(
        default=None,
        description=(
            "Unix timestamp when the last predecessor finished, excluding think "
            "time. None when the request has no predecessor."
        ),
    )
    queued: float | None = Field(
        default=None,
        description="Unix timestamp when request was placed into processing queue",
    )
    dequeued: float | None = Field(
        default=None,
        description="Unix timestamp when request was removed from queue for processing",
    )
    scheduled_at: float | None = Field(
        default=None,
        description="Unix timestamp when the request was scheduled for processing",
    )
    resolve_start: float | None = Field(
        default=None,
        description="Unix timestamp when backend resolution of the request began",
    )
    request_start: float | None = Field(
        default=None,
        description="Unix timestamp when the backend began processing the request",
    )
    first_request_iteration: float | None = Field(
        default=None,
    )
    first_token_iteration: float | None = Field(
        default=None,
    )
    first_output_token_iteration: float | None = Field(
        default=None,
        description=(
            "Unix timestamp of the first non-reasoning content token. "
            "Equals first_token_iteration when no reasoning tokens are emitted."
        ),
    )
    last_token_iteration: float | None = Field(
        default=None,
    )
    last_request_iteration: float | None = Field(
        default=None,
    )
    request_iterations: int = Field(
        default=0,
    )
    token_iterations: int = Field(
        default=0,
    )
    last_request_sent: float | None = Field(
        default=None,
        description=(
            "Unix timestamp of the last packet sent to the server, used for "
            "round-trip metrics (openai_websocket backend)"
        ),
    )
    request_sent_sum: float = Field(
        default=0.0,
        description=(
            "Sum of sent-packet timestamps for mean round-trip estimation "
            "(openai_websocket backend)"
        ),
    )
    request_sent_count: int = Field(
        default=0,
        description="Number of packets sent to the server (openai_websocket backend)",
    )
    token_received_sum: float = Field(
        default=0.0,
        description=(
            "Sum of received content-token timestamps for mean round-trip "
            "estimation (openai_websocket backend)"
        ),
    )
    token_received_count: int = Field(
        default=0,
        description="Number of content tokens received (openai_websocket backend)",
    )
    request_end: float | None = Field(
        default=None,
        description="Unix timestamp when the backend completed processing the request",
    )
    resolve_end: float | None = Field(
        default=None,
        description="Unix timestamp when backend resolution of the request completed",
    )
    finalized: float | None = Field(
        default=None,
        description="Unix timestamp when request was processed by the scheduler",
    )

    @property
    def last_reported(self) -> float | None:
        """
        Get the most recent timing measurement available.

        :return: The latest Unix timestamp from the timing fields, or None if none
        """
        timing_fields = [
            self.queued,
            self.dequeued,
            self.scheduled_at,
            self.resolve_start,
            self.request_start,
            self.request_end,
            self.resolve_end,
        ]
        valid_timings = [field for field in timing_fields if field is not None]
        return max(valid_timings) if valid_timings else None

last_reported property

Get the most recent timing measurement available.

Returns:

Type Description
float | None

The latest Unix timestamp from the timing fields, or None if none