Skip to content

guidellm.scheduler.constraints.error

Error-based constraint implementations.

Provides constraint types for limiting benchmark execution based on error rates and error counts. These constraints monitor request error status to determine when to stop benchmark execution due to excessive errors.

MaxErrorRateConstraint

Bases: PydanticConstraintInitializer

Constraint that limits execution based on sliding window error rate.

Tracks error status of recent requests in a sliding window and stops all processing when the error rate exceeds the threshold. Only applies the constraint after processing enough requests to fill the minimum window size for statistical significance.

Source code in src/guidellm/scheduler/constraints/error.py
@ConstraintsInitializerFactory.register("max_error_rate")
class MaxErrorRateConstraint(PydanticConstraintInitializer):
    """
    Constraint that limits execution based on sliding window error rate.

    Tracks error status of recent requests in a sliding window and stops all
    processing when the error rate exceeds the threshold. Only applies the
    constraint after processing enough requests to fill the minimum window size
    for statistical significance.
    """

    type_: Literal["max_error_rate"] = "max_error_rate"  # type: ignore[assignment]
    args: MaxErrorRateConstraintArgs = Field(
        description="Configuration arguments for max error rate constraint",
    )
    error_window: list[bool] = Field(
        default_factory=list,
        description="Sliding window tracking error status of recent requests",
    )
    current_index: int = Field(
        default=-1, description="Current index in the error window"
    )

    def create_constraint(self, **_kwargs) -> Constraint:
        """
        Create a new instance of MaxErrorRateConstraint (due to stateful window).

        :param kwargs: Additional keyword arguments (unused)
        :return: New instance of the constraint
        """
        self.current_index += 1

        return cast("Constraint", self.model_copy())

    def __call__(
        self, state: SchedulerState, request_info: RequestInfo | None
    ) -> SchedulerUpdateAction:
        """
        Evaluate constraint against sliding window error rate.

        :param state: Current scheduler state with request counts
        :param request_info: Individual request with completion status, or ``None``
            on poll (does not record a window sample)
        :return: Action indicating whether to continue or stop operations
        """
        current_index = max(0, self.current_index)
        max_error_rate = (
            self.args.rate
            if isinstance(self.args.rate, int | float)
            else self.args.rate[min(current_index, len(self.args.rate) - 1)]
        )

        if request_info is not None and request_info.status in [
            "completed",
            "errored",
            "cancelled",
        ]:
            self.error_window.append(request_info.status == "errored")
            if len(self.error_window) > self.args.window:
                self.error_window.pop(0)

        error_count = sum(self.error_window)
        window_requests = len(self.error_window)
        error_rate = (
            error_count / float(window_requests) if window_requests > 0 else 0.0
        )
        exceeded_min_processed = state.processed_requests >= self.args.window
        exceeded_error_rate = error_rate >= max_error_rate
        exceeded = exceeded_min_processed and exceeded_error_rate
        stop_time = constraint_stop_time(request_info, stopped=exceeded)

        return SchedulerUpdateAction(
            request_queuing="stop" if exceeded else "continue",
            request_processing="stop_all" if exceeded else "continue",
            stopping_scope=self.args.stopping_scope,
            metadata={
                "max_error_rate": max_error_rate,
                "window_size": self.args.window,
                "error_count": error_count,
                "processed_count": state.processed_requests,
                "current_window_size": len(self.error_window),
                "current_error_rate": error_rate,
                "exceeded_min_processed": exceeded_min_processed,
                "exceeded_error_rate": exceeded_error_rate,
                "exceeded": exceeded,
                "stop_time": stop_time,
            },
        )

__call__(state, request_info)

Evaluate constraint against sliding window error rate.

Parameters:

Name Type Description Default
state SchedulerState

Current scheduler state with request counts

required
request_info RequestInfo | None

Individual request with completion status, or None on poll (does not record a window sample)

required

Returns:

Type Description
SchedulerUpdateAction

Action indicating whether to continue or stop operations

Source code in src/guidellm/scheduler/constraints/error.py
def __call__(
    self, state: SchedulerState, request_info: RequestInfo | None
) -> SchedulerUpdateAction:
    """
    Evaluate constraint against sliding window error rate.

    :param state: Current scheduler state with request counts
    :param request_info: Individual request with completion status, or ``None``
        on poll (does not record a window sample)
    :return: Action indicating whether to continue or stop operations
    """
    current_index = max(0, self.current_index)
    max_error_rate = (
        self.args.rate
        if isinstance(self.args.rate, int | float)
        else self.args.rate[min(current_index, len(self.args.rate) - 1)]
    )

    if request_info is not None and request_info.status in [
        "completed",
        "errored",
        "cancelled",
    ]:
        self.error_window.append(request_info.status == "errored")
        if len(self.error_window) > self.args.window:
            self.error_window.pop(0)

    error_count = sum(self.error_window)
    window_requests = len(self.error_window)
    error_rate = (
        error_count / float(window_requests) if window_requests > 0 else 0.0
    )
    exceeded_min_processed = state.processed_requests >= self.args.window
    exceeded_error_rate = error_rate >= max_error_rate
    exceeded = exceeded_min_processed and exceeded_error_rate
    stop_time = constraint_stop_time(request_info, stopped=exceeded)

    return SchedulerUpdateAction(
        request_queuing="stop" if exceeded else "continue",
        request_processing="stop_all" if exceeded else "continue",
        stopping_scope=self.args.stopping_scope,
        metadata={
            "max_error_rate": max_error_rate,
            "window_size": self.args.window,
            "error_count": error_count,
            "processed_count": state.processed_requests,
            "current_window_size": len(self.error_window),
            "current_error_rate": error_rate,
            "exceeded_min_processed": exceeded_min_processed,
            "exceeded_error_rate": exceeded_error_rate,
            "exceeded": exceeded,
            "stop_time": stop_time,
        },
    )

create_constraint(**_kwargs)

Create a new instance of MaxErrorRateConstraint (due to stateful window).

Parameters:

Name Type Description Default
kwargs

Additional keyword arguments (unused)

required

Returns:

Type Description
Constraint

New instance of the constraint

Source code in src/guidellm/scheduler/constraints/error.py
def create_constraint(self, **_kwargs) -> Constraint:
    """
    Create a new instance of MaxErrorRateConstraint (due to stateful window).

    :param kwargs: Additional keyword arguments (unused)
    :return: New instance of the constraint
    """
    self.current_index += 1

    return cast("Constraint", self.model_copy())

MaxErrorsConstraint

Bases: PydanticConstraintInitializer

Constraint that limits execution based on absolute error count.

Stops both request queuing and all request processing when the total number of errored requests reaches the maximum threshold. Uses global error tracking across all requests for immediate constraint evaluation.

Source code in src/guidellm/scheduler/constraints/error.py
@ConstraintsInitializerFactory.register("max_errors")
class MaxErrorsConstraint(PydanticConstraintInitializer):
    """
    Constraint that limits execution based on absolute error count.

    Stops both request queuing and all request processing when the total number
    of errored requests reaches the maximum threshold. Uses global error tracking
    across all requests for immediate constraint evaluation.
    """

    type_: Literal["max_errors"] = "max_errors"  # type: ignore[assignment]
    args: MaxErrorsConstraintArgs = Field(
        description="Configuration arguments for max errors constraint",
    )
    current_index: int = Field(default=-1, description="Current index in error list")

    def create_constraint(self, **_kwargs) -> Constraint:
        """
        Return self as the constraint instance.

        :param kwargs: Additional keyword arguments (unused)
        :return: Self instance as the constraint
        """
        self.current_index += 1

        return cast("Constraint", self.model_copy())

    def __call__(
        self, state: SchedulerState, request_info: RequestInfo | None
    ) -> SchedulerUpdateAction:
        """
        Evaluate constraint against current error count.

        :param state: Current scheduler state with error counts
        :param request_info: Individual request information, or ``None`` on poll
        :return: Action indicating whether to continue or stop operations
        """
        current_index = max(0, self.current_index)
        max_errors = (
            self.args.count
            if isinstance(self.args.count, int | float)
            else self.args.count[min(current_index, len(self.args.count) - 1)]
        )
        errors_exceeded = state.errored_requests >= max_errors
        stop_time = constraint_stop_time(request_info, stopped=errors_exceeded)

        return SchedulerUpdateAction(
            request_queuing="stop" if errors_exceeded else "continue",
            request_processing="stop_all" if errors_exceeded else "continue",
            stopping_scope=self.args.stopping_scope,
            metadata={
                "max_errors": max_errors,
                "errors_exceeded": errors_exceeded,
                "current_errors": state.errored_requests,
                "stop_time": stop_time,
            },
            progress=SchedulerProgress(stop_time=stop_time),
        )

__call__(state, request_info)

Evaluate constraint against current error count.

Parameters:

Name Type Description Default
state SchedulerState

Current scheduler state with error counts

required
request_info RequestInfo | None

Individual request information, or None on poll

required

Returns:

Type Description
SchedulerUpdateAction

Action indicating whether to continue or stop operations

Source code in src/guidellm/scheduler/constraints/error.py
def __call__(
    self, state: SchedulerState, request_info: RequestInfo | None
) -> SchedulerUpdateAction:
    """
    Evaluate constraint against current error count.

    :param state: Current scheduler state with error counts
    :param request_info: Individual request information, or ``None`` on poll
    :return: Action indicating whether to continue or stop operations
    """
    current_index = max(0, self.current_index)
    max_errors = (
        self.args.count
        if isinstance(self.args.count, int | float)
        else self.args.count[min(current_index, len(self.args.count) - 1)]
    )
    errors_exceeded = state.errored_requests >= max_errors
    stop_time = constraint_stop_time(request_info, stopped=errors_exceeded)

    return SchedulerUpdateAction(
        request_queuing="stop" if errors_exceeded else "continue",
        request_processing="stop_all" if errors_exceeded else "continue",
        stopping_scope=self.args.stopping_scope,
        metadata={
            "max_errors": max_errors,
            "errors_exceeded": errors_exceeded,
            "current_errors": state.errored_requests,
            "stop_time": stop_time,
        },
        progress=SchedulerProgress(stop_time=stop_time),
    )

create_constraint(**_kwargs)

Return self as the constraint instance.

Parameters:

Name Type Description Default
kwargs

Additional keyword arguments (unused)

required

Returns:

Type Description
Constraint

Self instance as the constraint

Source code in src/guidellm/scheduler/constraints/error.py
def create_constraint(self, **_kwargs) -> Constraint:
    """
    Return self as the constraint instance.

    :param kwargs: Additional keyword arguments (unused)
    :return: Self instance as the constraint
    """
    self.current_index += 1

    return cast("Constraint", self.model_copy())

MaxGlobalErrorRateConstraint

Bases: PydanticConstraintInitializer

Constraint that limits execution based on global error rate.

Calculates error rate across all processed requests and stops all processing when the rate exceeds the threshold. Only applies the constraint after processing the minimum number of requests to ensure statistical significance for global error rate calculations.

Source code in src/guidellm/scheduler/constraints/error.py
@ConstraintsInitializerFactory.register("max_global_error_rate")
class MaxGlobalErrorRateConstraint(PydanticConstraintInitializer):
    """
    Constraint that limits execution based on global error rate.

    Calculates error rate across all processed requests and stops all processing
    when the rate exceeds the threshold. Only applies the constraint after
    processing the minimum number of requests to ensure statistical significance
    for global error rate calculations.
    """

    type_: Literal["max_global_error_rate"] = "max_global_error_rate"  # type: ignore[assignment]
    args: MaxGlobalErrorRateConstraintArgs = Field(
        description="Configuration arguments for max global error rate constraint",
    )
    current_index: int = Field(
        default=-1, description="Current index for list-based max_error_rate values"
    )

    def create_constraint(self, **_kwargs) -> Constraint:
        """
        Return self as the constraint instance.

        :param kwargs: Additional keyword arguments (unused)
        :return: Self instance as the constraint
        """
        self.current_index += 1

        return cast("Constraint", self.model_copy())

    def __call__(
        self, state: SchedulerState, request_info: RequestInfo | None
    ) -> SchedulerUpdateAction:
        """
        Evaluate constraint against global error rate.

        :param state: Current scheduler state with global request and error counts
        :param request_info: Individual request information, or ``None`` on poll
        :return: Action indicating whether to continue or stop operations
        """
        current_index = max(0, self.current_index)
        max_error_rate = (
            self.args.rate
            if isinstance(self.args.rate, int | float)
            else self.args.rate[min(current_index, len(self.args.rate) - 1)]
        )

        exceeded_min_processed = (
            self.args.minimum is None or state.processed_requests >= self.args.minimum
        )
        error_rate = (
            state.errored_requests / float(state.processed_requests)
            if state.processed_requests > 0
            else 0.0
        )
        exceeded_error_rate = error_rate >= max_error_rate
        exceeded = exceeded_min_processed and exceeded_error_rate
        stop_time = constraint_stop_time(request_info, stopped=exceeded)

        return SchedulerUpdateAction(
            request_queuing="stop" if exceeded else "continue",
            request_processing="stop_all" if exceeded else "continue",
            stopping_scope=self.args.stopping_scope,
            metadata={
                "max_error_rate": max_error_rate,
                "min_processed": self.args.minimum,
                "processed_requests": state.processed_requests,
                "errored_requests": state.errored_requests,
                "error_rate": error_rate,
                "exceeded_min_processed": exceeded_min_processed,
                "exceeded_error_rate": exceeded_error_rate,
                "exceeded": exceeded,
                "stop_time": stop_time,
            },
            progress=SchedulerProgress(stop_time=stop_time),
        )

__call__(state, request_info)

Evaluate constraint against global error rate.

Parameters:

Name Type Description Default
state SchedulerState

Current scheduler state with global request and error counts

required
request_info RequestInfo | None

Individual request information, or None on poll

required

Returns:

Type Description
SchedulerUpdateAction

Action indicating whether to continue or stop operations

Source code in src/guidellm/scheduler/constraints/error.py
def __call__(
    self, state: SchedulerState, request_info: RequestInfo | None
) -> SchedulerUpdateAction:
    """
    Evaluate constraint against global error rate.

    :param state: Current scheduler state with global request and error counts
    :param request_info: Individual request information, or ``None`` on poll
    :return: Action indicating whether to continue or stop operations
    """
    current_index = max(0, self.current_index)
    max_error_rate = (
        self.args.rate
        if isinstance(self.args.rate, int | float)
        else self.args.rate[min(current_index, len(self.args.rate) - 1)]
    )

    exceeded_min_processed = (
        self.args.minimum is None or state.processed_requests >= self.args.minimum
    )
    error_rate = (
        state.errored_requests / float(state.processed_requests)
        if state.processed_requests > 0
        else 0.0
    )
    exceeded_error_rate = error_rate >= max_error_rate
    exceeded = exceeded_min_processed and exceeded_error_rate
    stop_time = constraint_stop_time(request_info, stopped=exceeded)

    return SchedulerUpdateAction(
        request_queuing="stop" if exceeded else "continue",
        request_processing="stop_all" if exceeded else "continue",
        stopping_scope=self.args.stopping_scope,
        metadata={
            "max_error_rate": max_error_rate,
            "min_processed": self.args.minimum,
            "processed_requests": state.processed_requests,
            "errored_requests": state.errored_requests,
            "error_rate": error_rate,
            "exceeded_min_processed": exceeded_min_processed,
            "exceeded_error_rate": exceeded_error_rate,
            "exceeded": exceeded,
            "stop_time": stop_time,
        },
        progress=SchedulerProgress(stop_time=stop_time),
    )

create_constraint(**_kwargs)

Return self as the constraint instance.

Parameters:

Name Type Description Default
kwargs

Additional keyword arguments (unused)

required

Returns:

Type Description
Constraint

Self instance as the constraint

Source code in src/guidellm/scheduler/constraints/error.py
def create_constraint(self, **_kwargs) -> Constraint:
    """
    Return self as the constraint instance.

    :param kwargs: Additional keyword arguments (unused)
    :return: Self instance as the constraint
    """
    self.current_index += 1

    return cast("Constraint", self.model_copy())