mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
[aaac85d2] Rate limit guardrails for Anthropic and Ollama providers (#104)
* [25aa5b24] Implement rate-limit Zustand store, Axios interceptor, WebSocket hook, banner component, and page-load sync (#99) (#101) * [25aa5b24] feat(rate-limits): add types, Zustand store, Axios 429 interceptor, WS hook, sync hook, and banner component - panel/src/types/rate-limits.ts: RateLimitEntry, RateLimitHitEvent, RateLimitLiftedEvent, RateLimitApiResponse - panel/src/store/rate-limit-store.ts: useRateLimitStore with Map state, hitRateLimit/liftRateLimit/syncFromApi - panel/src/lib/api/rate-limits.ts: GET /api/system/rate-limits with isMockMode guard - panel/src/lib/api/client.ts: 429 interceptor dispatches to store first, Sonner toast on retry exhaustion - panel/src/hooks/use-rate-limit-websocket.ts: RATE_LIMIT_HIT/LIFTED events + onReconnect callback - panel/src/hooks/use-rate-limit-sync.ts: mount sync + no-op with console.warn when endpoint unavailable - panel/src/components/rate-limit/rate-limit-banner.tsx: amber rows with countdown, no dismiss button - panel/src/app/(dashboard)/layout.tsx: RateLimitBanner mounted below Header - store/index.ts, hooks/index.ts: export new store and hooks * [25aa5b24] fix(rate-limit-banner): use lint-clean countdown pattern (computeSecondsLeft outside render) * [25aa5b24] fix(client): add real retry loop to 429 interceptor so Sonner toast fires on exhaustion - Increment error.config._retryCount and return api(error.config) when retryCount < RATE_LIMIT_MAX_RETRIES, actually retrying the request. - Toast fires only when retryCount >= RATE_LIMIT_MAX_RETRIES (3 attempts). - Fixes AC4: toast was dead code because without return api(error.config) every 429 saw retryCount=1, permanently below the threshold of 3. --------- Co-authored-by: Frontend Developer 1 <fe-dev-1@agents.roboco.dev> * [4112cd34] feat(rate-limit): add RateLimitError with 5-retry exponential backoff at all LLM call sites (#102) (#103) - Create roboco/services/exceptions.py with RateLimitError(provider, retry_after), HTTP_TOO_MANY_REQUESTS, MAX_RATE_LIMIT_RETRIES constants, and parse_retry_after_header() helper - extraction.py: extract _call_anthropic_with_retry() helper; retry Anthropic call 5x on 429 with exponential backoff; re-raise RateLimitError from outer except instead of swallowing it - ollama_embedder.py: 5-retry outer loop (429) wrapping existing 3-retry inner loop (ConnectError/Timeout) for all 4 call sites; two concerns kept isolated - indexes/base.py, mentor.py, validator.py: replace magic 429 literals with HTTP_TOO_MANY_REQUESTS; 5-retry loop on 429 for LLM calls - middleware.py: add rate_limit_exception_handler returning HTTP 429 with Retry-After response header - tests/unit/services/test_rate_limit_retry.py: 28 tests covering exhaustion, Retry-After header sleep, partial retries then success, ConnectError isolation Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> * [18107054] feat(rate-limit): Redis rate-limit state tracker + i_am_blocked rate_limited path (#105) (#106) - Add RateLimitStateTracker in roboco/services/gateway/rate_limit_tracker.py with activate(), clear(), is_rate_limited(), get_state(), increment_probe_failures(), reset_probe_failures() backed by redis.asyncio - Add RATE_LIMIT_HIT = "rate_limit.hit" to EventType StrEnum in events.py - Add _handle_rate_limited_parking() to Choreographer: intercepts i_am_blocked(reason='rate_limited') before block state transition, parks all active agents sharing affected provider via mark_waiting_long, publishes RATE_LIMIT_HIT event to StreamEventBus, task stays in_progress - Add get_provider_for_agent() and get_active_agent_slugs_for_provider() helper methods to AgentOrchestrator - Wire orchestrator and stream_bus into ChoreographerDeps via deps.py - Add test_rate_limit_tracker.py (basic ops, probe failures, cross-reconnection persistence, provider isolation) and test_i_am_blocked_rate_limited.py (AC3/AC4/AC5 coverage: task stays in_progress, mark_waiting_long call count equals active agent count, RATE_LIMIT_HIT event payload structure) Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> * [5501e4b4] Wire RateLimitStateTracker into live orchestrator paths — 4 CEO-identified integration gaps (#109) * [8451ca50] feat(gateway): wire RateLimitStateTracker.activate() into i_am_blocked rate-limited path and add provider-rate-limit gate to decide_spawn() (#107) - Add provider/provider_rate_limited optional fields to TriggerContext (backward-compatible defaults) - Insert rule 2 in decide_spawn(): QUEUE when trigger.provider_rate_limited is True with reason 'provider X rate-limited' - Call RateLimitStateTracker(provider).activate() in _handle_rate_limited_parking() after mark_waiting_long loop (wrapped in contextlib.suppress for Redis fault tolerance) - Extend gateway_pre_spawn_check() with optional provider param; check RateLimitStateTracker.is_rate_limited() when provider is known - Pass provider=self.get_provider_for_agent(agent_id) from orchestrator call site - Add TestProviderRateLimitGate (6 tests) to test_trigger_filter.py - Add TestRateLimitTrackerActivateOnParking (6 tests) to test_i_am_blocked_rate_limited.py - All 38 unit tests pass; ruff and mypy clean on changed files Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> * [e9cef0f0] feat(rate-limits): sweeper probe loop, CEO notification, and GET /api/system/rate-limits endpoint (AC4, AC8, AC9) (#108) - Add RATE_LIMIT_LIFTED event type to EventType enum in models/events.py - Add RateLimitStateTracker.list_rate_limited_providers() classmethod to scan Redis for all currently rate-limited providers (used by the new endpoint) - Add orchestrator._rate_limit_probe_loop(): background task started/stopped in start()/stop(), runs _sweep_rate_limit_probes() every 30s - Add orchestrator._probe_one_provider(): checks estimated_lift_at gate, calls _do_probe(); on success: tracker.clear(), resolve_wait() for all parked agents with waiting_for='rate_limit_lifted' matching the provider, publishes RATE_LIMIT_LIFTED event; on failure: increments probe_failures counter, sends CEO notification at threshold 10 (once per episode via _rate_limit_ceo_notified) - Add orchestrator._make_tracker(): injectable factory for RateLimitStateTracker - Add orchestrator._do_probe(): overridable async bool probe (default: True) - Add orchestrator._notify_rate_limit_ceo(): high-priority notification to CEO containing provider name, duration since activation, and paused agent count - Add roboco/api/routes/system.py with GET /rate-limits endpoint (AC9) - Register system_router in app.py under /api/system prefix - Add 17 unit tests in tests/unit/runtime/test_rate_limit_sweep.py covering all AC4/AC8/AC9 paths: probe success/failure, CEO threshold, endpoint schema Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> --------- Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> --------- Co-authored-by: Frontend Developer 1 <fe-dev-1@agents.roboco.dev> Co-authored-by: Backend Developer 1 <be-dev-1@agents.roboco.dev> Co-authored-by: Renn F <rennf93@users.noreply.github.com>
This commit is contained in:
co-authored by
Backend Developer 1
Frontend Developer 1
Renn F
parent
cc4ccb7ea3
commit
98e618c243
@@ -1,5 +1,17 @@
|
||||
import axios, { AxiosInstance, AxiosError } from "axios";
|
||||
import { toast } from "sonner";
|
||||
import { API_URL, CEO_AGENT_ID, CEO_ROLE } from "@/lib/constants";
|
||||
import { useRateLimitStore } from "@/store/rate-limit-store";
|
||||
import type { RateLimitHitEvent } from "@/types/rate-limits";
|
||||
|
||||
// Custom Axios config extension for retry tracking
|
||||
declare module "axios" {
|
||||
interface InternalAxiosRequestConfig {
|
||||
_retryCount?: number;
|
||||
}
|
||||
}
|
||||
|
||||
const RATE_LIMIT_MAX_RETRIES = 3;
|
||||
|
||||
// Create axios instance with default config
|
||||
const api: AxiosInstance = axios.create({
|
||||
@@ -47,6 +59,46 @@ api.interceptors.response.use(
|
||||
const errorData = error.response?.data as Record<string, unknown> | undefined;
|
||||
const errorDetail = errorData?.detail || error.message;
|
||||
|
||||
// -------------------------------------------------------------------------
|
||||
// 429 Rate-limit handling — FIRST side-effect, before any other logic
|
||||
// -------------------------------------------------------------------------
|
||||
if (status === 429) {
|
||||
const retryAfterHeader = error.response?.headers?.["retry-after"];
|
||||
const retryAfterSeconds = retryAfterHeader ? parseInt(String(retryAfterHeader), 10) : 60;
|
||||
const safeRetryAfter = isNaN(retryAfterSeconds) ? 60 : retryAfterSeconds;
|
||||
|
||||
// Extract provider from custom header or fall back to URL path heuristics
|
||||
const providerHeader = error.response?.headers?.["x-provider"];
|
||||
const urlProvider = url
|
||||
? (["anthropic", "openai", "ollama"].find((p) => url.includes(p)) ?? "unknown")
|
||||
: "unknown";
|
||||
const provider = (providerHeader as string | undefined) ?? urlProvider;
|
||||
|
||||
// Dispatch to store as first side-effect
|
||||
const hitEvent: RateLimitHitEvent = {
|
||||
type: "RATE_LIMIT_HIT",
|
||||
provider,
|
||||
affectedAgents: [],
|
||||
retryAfterSeconds: safeRetryAfter,
|
||||
timestamp: new Date().toISOString(),
|
||||
};
|
||||
useRateLimitStore.getState().hitRateLimit(hitEvent);
|
||||
|
||||
// Track retry count; retry the request until exhausted, then toast
|
||||
const retryCount = (error.config?._retryCount ?? 0) + 1;
|
||||
if (error.config) {
|
||||
error.config._retryCount = retryCount;
|
||||
if (retryCount < RATE_LIMIT_MAX_RETRIES) {
|
||||
// Retry the request — interceptor re-runs on each subsequent 429
|
||||
return api(error.config);
|
||||
}
|
||||
}
|
||||
// Retries exhausted — notify the user via Sonner toast
|
||||
toast.warning(
|
||||
`Rate limited by ${provider}. The system has paused operations and will resume automatically in ~${safeRetryAfter}s.`
|
||||
);
|
||||
}
|
||||
|
||||
// Log comprehensive error info
|
||||
console.error(`[API] ✗ ${method} ${url}`, {
|
||||
status,
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
import api from "./client";
|
||||
import type { RateLimitApiResponse } from "@/types/rate-limits";
|
||||
import { isMockMode } from "@/lib/mock-data";
|
||||
|
||||
export const rateLimitsApi = {
|
||||
/**
|
||||
* GET /api/system/rate-limits — fetch active rate limits on page load or WS reconnect.
|
||||
* Returns empty list in mock mode (rate limits are a real-backend-only concern).
|
||||
*/
|
||||
getRateLimits: async (): Promise<RateLimitApiResponse> => {
|
||||
if (isMockMode()) {
|
||||
console.warn("[rate-limits] isMockMode: skipping GET /api/system/rate-limits");
|
||||
return { entries: [] };
|
||||
}
|
||||
const { data } = await api.get<RateLimitApiResponse>("/system/rate-limits");
|
||||
return data;
|
||||
},
|
||||
};
|
||||
Reference in New Issue
Block a user