Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions contextual_orchestrator/orchestrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -11016,6 +11016,11 @@ def attempt(agent: ModelAgent) -> EndpointAttempt[Any]:
# incidental wording in an upstream error body (e.g. a
# 400 that happens to mention "invalid arguments").
decision = classify_provider_transport_failure(exc.retryable)
if exc.provider_status == 429:
# Quota refusal already selected a cooldown above;
# another immediate call to this agent only spends
# the same quota and must not open its health circuit.
decision = replace(downgrade_to_failover(decision), circuit_failure=False)
elif isinstance(exc, ProviderResponseError):
if allowed_agent_ids is None:
raise
Expand Down
4 changes: 3 additions & 1 deletion tests/test_rate_limit_aware_admission.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@
import urllib.error
from contextlib import contextmanager
from copy import deepcopy
from dataclasses import replace
from email.message import Message
from typing import Any

Expand Down Expand Up @@ -506,8 +507,9 @@ def _post_chat_completion(port: int, payload: dict[str, Any], token: str):
def test_http_route_once_waits_out_storm_and_serves_the_request() -> None:
"""orchestrator/free over /v1/chat/completions (route_once) waits out a 429 storm."""
orchestrator = TaskOrchestrator(
_free_route_agents(), tool_retry_attempts=0, rate_limit_wait_seconds=5.0
_free_route_agents(), tool_retry_attempts=1, rate_limit_wait_seconds=5.0
)
orchestrator.policy = replace(orchestrator.policy, realtime_judge=False)
slept: list[float] = []
orchestrator._rate_limit_sleep = slept.append
chat_outcomes = QueuedChatOutcomes(
Expand Down
Loading