mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
* feat(conventions): standard schema models + effective-map merge * feat(orchestrator): park provider on persistent server overload (529/500) A 429 rate limit already parks a provider — queue its spawns, probe until it recovers — but a persistent 529/500/503 overload had no such break: the run died and the orchestrator crash-retried straight back into the overload, burning tokens in a respawn loop. Generalize the park to provider-unavailability. On a non-graceful Anthropic agent exit, match the API's overload markers (overloaded_error / internal_server_error / "API Error: 5xx") against the dead container's own output and park the provider with kind="overloaded"; the existing spawn gate already queues any parked provider, and the probe-resume loop revives the task when it recovers. Grok keeps its exit-75 path; both now route through one _park_provider_unavailable helper. Markers are kept specific so an agent that merely writes about HTTP 500/529 can't trip the break. Fix the recovery probe to require a 2xx: it treated any non-429 as recovered, so a probe that itself got a 529 would have resumed agents straight back into the overload — wrong for the new path and for a 429 that lifts into a 5xx. Gated by ROBOCO_OVERLOAD_BREAK_ENABLED (default on; off => crash-retry). --------- Co-authored-by: Renn F <rennf93@users.noreply.github.com>
151 lines
5.2 KiB
Python
151 lines
5.2 KiB
Python
"""Rate-limit recovery probe — real per-provider liveness check.
|
|
|
|
``_do_probe`` replaced a time-based stub that always returned True. It now
|
|
makes a free, unmetered call (Anthropic ``GET /v1/models`` / Ollama
|
|
``GET /api/tags``) and treats only a 2xx response as the provider having
|
|
recovered — a 429 (rate limit) and a 5xx (overload) both keep it parked.
|
|
These tests pin that contract: target resolution per provider, the
|
|
2xx-vs-error decision, network-error → stay-parked, and the un-probeable
|
|
fallback to time-expiry optimism.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import TYPE_CHECKING
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import httpx
|
|
import pytest
|
|
from roboco.config import settings
|
|
from roboco.runtime.orchestrator import (
|
|
_HTTP_TOO_MANY_REQUESTS,
|
|
AgentOrchestrator,
|
|
)
|
|
|
|
if TYPE_CHECKING:
|
|
from collections.abc import Iterator
|
|
|
|
_HTTP_OK = 200
|
|
|
|
|
|
@pytest.fixture
|
|
def orch() -> AgentOrchestrator:
|
|
return AgentOrchestrator.__new__(AgentOrchestrator)
|
|
|
|
|
|
@pytest.fixture
|
|
def with_anthropic_key() -> Iterator[None]:
|
|
original = settings.anthropic_api_key
|
|
settings.anthropic_api_key = "sk-test-key"
|
|
yield
|
|
settings.anthropic_api_key = original
|
|
|
|
|
|
def _fake_async_client(
|
|
*, status_code: int | None = None, raise_exc: Exception | None = None
|
|
) -> MagicMock:
|
|
"""Patch target for ``httpx.AsyncClient`` — async ctx mgr whose get() responds."""
|
|
response = MagicMock()
|
|
response.status_code = status_code
|
|
client = MagicMock()
|
|
client.__aenter__ = AsyncMock(return_value=client)
|
|
client.__aexit__ = AsyncMock(return_value=False)
|
|
client.get = (
|
|
AsyncMock(side_effect=raise_exc)
|
|
if raise_exc is not None
|
|
else AsyncMock(return_value=response)
|
|
)
|
|
return MagicMock(return_value=client)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# _probe_target — provider → (url, headers)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.usefixtures("with_anthropic_key")
|
|
def test_target_anthropic_with_key() -> None:
|
|
url, headers = AgentOrchestrator._probe_target("anthropic")
|
|
assert url is not None
|
|
assert url.endswith("/v1/models")
|
|
assert headers["x-api-key"] == "sk-test-key"
|
|
assert "anthropic-version" in headers
|
|
|
|
|
|
def test_target_anthropic_without_key() -> None:
|
|
original = settings.anthropic_api_key
|
|
settings.anthropic_api_key = None
|
|
try:
|
|
url, headers = AgentOrchestrator._probe_target("anthropic")
|
|
finally:
|
|
settings.anthropic_api_key = original
|
|
assert url is None
|
|
assert headers == {}
|
|
|
|
|
|
def test_target_ollama_uses_tags_endpoint() -> None:
|
|
url, _headers = AgentOrchestrator._probe_target("ollama_cloud")
|
|
assert url is not None
|
|
assert url.endswith("/api/tags")
|
|
|
|
|
|
def test_target_unknown_provider_is_unprobeable() -> None:
|
|
assert AgentOrchestrator._probe_target("mystery") == (None, {})
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# _do_probe — the liveness decision
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.usefixtures("with_anthropic_key")
|
|
async def test_probe_anthropic_ok_is_lifted(orch: AgentOrchestrator) -> None:
|
|
fake = _fake_async_client(status_code=_HTTP_OK)
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient", fake):
|
|
assert await orch._do_probe("anthropic") is True
|
|
|
|
|
|
@pytest.mark.usefixtures("with_anthropic_key")
|
|
async def test_probe_anthropic_429_stays_limited(orch: AgentOrchestrator) -> None:
|
|
fake = _fake_async_client(status_code=_HTTP_TOO_MANY_REQUESTS)
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient", fake):
|
|
assert await orch._do_probe("anthropic") is False
|
|
|
|
|
|
@pytest.mark.usefixtures("with_anthropic_key")
|
|
@pytest.mark.parametrize("status", [500, 503, 529])
|
|
async def test_probe_anthropic_5xx_stays_parked(
|
|
orch: AgentOrchestrator, status: int
|
|
) -> None:
|
|
"""A 5xx (overload) keeps the provider parked — resuming would re-overload it."""
|
|
fake = _fake_async_client(status_code=status)
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient", fake):
|
|
assert await orch._do_probe("anthropic") is False
|
|
|
|
|
|
@pytest.mark.usefixtures("with_anthropic_key")
|
|
async def test_probe_network_error_stays_parked(orch: AgentOrchestrator) -> None:
|
|
fake = _fake_async_client(raise_exc=httpx.ConnectError("boom"))
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient", fake):
|
|
assert await orch._do_probe("anthropic") is False
|
|
|
|
|
|
async def test_probe_ollama_ok_is_lifted(orch: AgentOrchestrator) -> None:
|
|
fake = _fake_async_client(status_code=_HTTP_OK)
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient", fake):
|
|
assert await orch._do_probe("ollama_cloud") is True
|
|
|
|
|
|
async def test_probe_unprobeable_falls_back_to_optimism(
|
|
orch: AgentOrchestrator,
|
|
) -> None:
|
|
"""No key / unknown provider → trust the elapsed retry_after window, no HTTP."""
|
|
original = settings.anthropic_api_key
|
|
settings.anthropic_api_key = None
|
|
try:
|
|
with patch("roboco.runtime.orchestrator.httpx.AsyncClient") as client_cls:
|
|
assert await orch._do_probe("anthropic") is True
|
|
client_cls.assert_not_called()
|
|
finally:
|
|
settings.anthropic_api_key = original
|