fix(orchestrator): break the notification-driven respawn loop (#643)

The escalation/approval dispatchers spawn a notification's recipient every
cooldown window for as long as it stays pending. These spawns carry no
task_id, so the PM respawn breaker never sees them — a single wedged
alert/escalation whose recipient never resolves it respawns that recipient
forever. Observed live: fe-pm's unacked alerts kept main-pm/fe-pm spawning
every ~2-3 min for 6+ hours.

Two guards, both gating the spawn after the existing cooldown:
- A hard per-(agent, notification) attempt cap (notification_spawn_max_attempts,
  default 5): once a notification has respawned its target that many times
  without being acknowledged, stop and log once. The count is id-scoped and
  survives map pruning (re-stamp), so a fresh escalation is unaffected.
- A live-work check before spawning: skip when the notification has expired,
  is stale past notification_spawn_max_age_seconds (default 6h — wedged or
  reloaded from before a restart), or its related task is already terminal.
  Fail-open — a failed task fetch or unparseable field never suppresses a
  real escalation.

Co-authored-by: Renn F <rennf93@users.noreply.github.com>
This commit is contained in:
Renzo F
2026-07-22 17:10:55 +02:00
committed by GitHub
co-authored by Renn F
parent f74131a122
commit b91229f487
4 changed files with 304 additions and 7 deletions
@@ -0,0 +1,98 @@
"""The 'is there actually live work' gate for notification-triggered spawns.
Escalation/approval dispatchers must not revive an agent for a notification
that has expired, is stale past the spawn-age window, or whose related task is
already terminal — otherwise a wedged/old notification loops the fleet.
"""
from __future__ import annotations
from datetime import UTC, datetime, timedelta
from typing import Any
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from roboco.config import settings
from roboco.runtime.orchestrator import AgentOrchestrator
def _orch() -> AgentOrchestrator:
# _api_url is a read-only property (settings.internal_api_url); the mock
# client below ignores the URL, so no wiring is needed.
return AgentOrchestrator.__new__(AgentOrchestrator)
def _client(task_status: str | None = None, *, fail: bool = False) -> Any:
client = MagicMock()
if fail:
client.get = AsyncMock(side_effect=RuntimeError("boom"))
return client
resp = MagicMock()
resp.status_code = 200
resp.json = MagicMock(return_value={"status": task_status})
client.get = AsyncMock(return_value=resp)
return client
def _iso(dt: datetime) -> str:
return dt.isoformat()
@pytest.mark.asyncio
async def test_expired_notification_has_no_work() -> None:
orch = _orch()
notif = {"expires_at": _iso(datetime.now(UTC) - timedelta(minutes=1))}
assert await orch._notification_has_live_work(_client(), notif) is False
@pytest.mark.asyncio
async def test_stale_notification_has_no_work() -> None:
orch = _orch()
old = datetime.now(UTC) - timedelta(
seconds=settings.notification_spawn_max_age_seconds + 60
)
notif = {"timestamp": _iso(old)}
assert await orch._notification_has_live_work(_client(), notif) is False
@pytest.mark.asyncio
async def test_terminal_related_task_has_no_work() -> None:
orch = _orch()
notif = {"timestamp": _iso(datetime.now(UTC)), "related_task_id": "t1"}
assert await orch._notification_has_live_work(_client("completed"), notif) is False
assert await orch._notification_has_live_work(_client("cancelled"), notif) is False
@pytest.mark.asyncio
async def test_fresh_notification_with_live_task_has_work() -> None:
orch = _orch()
notif = {"timestamp": _iso(datetime.now(UTC)), "related_task_id": "t1"}
assert await orch._notification_has_live_work(_client("in_progress"), notif) is True
@pytest.mark.asyncio
async def test_fresh_notification_no_task_has_work() -> None:
orch = _orch()
notif = {"timestamp": _iso(datetime.now(UTC))}
assert await orch._notification_has_live_work(_client(), notif) is True
@pytest.mark.asyncio
async def test_fail_open_on_fetch_error_and_bad_timestamp() -> None:
orch = _orch()
# A failed task fetch must not suppress a real escalation.
notif = {"timestamp": _iso(datetime.now(UTC)), "related_task_id": "t1"}
assert await orch._notification_has_live_work(_client(fail=True), notif) is True
# An unparseable timestamp is ignored (no false-stale), not treated as old.
assert (
await orch._notification_has_live_work(_client(), {"timestamp": "nope"}) is True
)
@pytest.mark.asyncio
async def test_staleness_gate_disabled_when_zero() -> None:
orch = _orch()
ancient = datetime.now(UTC) - timedelta(days=30)
notif = {"timestamp": _iso(ancient)}
with patch.object(settings, "notification_spawn_max_age_seconds", 0):
assert await orch._notification_has_live_work(_client(), notif) is True
@@ -59,6 +59,67 @@ def test_missing_notification_id_never_damped() -> None:
assert orch._notification_spawn_at == {}
def test_hard_cap_breaks_respawn_loop() -> None:
"""Past max_attempts spawns for one unacked notification, the target is
suppressed forever — the loop breaker for no-task_id escalations."""
orch = _orch()
with (
patch.object(settings, "notification_spawn_cooldown_seconds", 600),
patch.object(settings, "notification_spawn_max_attempts", 3),
patch("roboco.runtime.orchestrator.time.monotonic") as clock,
):
# Each retry lands in a fresh cooldown window (advance past 600s).
for i in range(3):
clock.return_value = 1_000.0 + i * 700
assert orch._notification_spawn_cooled("fe-pm", "stuck") is False
# 4th+ window: cap tripped — suppressed despite the cooldown elapsing.
for i in range(3, 8):
clock.return_value = 1_000.0 + i * 700
assert orch._notification_spawn_cooled("fe-pm", "stuck") is True
# A different notification is unaffected by another's cap.
clock.return_value = 9_000.0
assert orch._notification_spawn_cooled("fe-pm", "other") is False
def test_zero_max_attempts_disables_cap() -> None:
orch = _orch()
with (
patch.object(settings, "notification_spawn_cooldown_seconds", 600),
patch.object(settings, "notification_spawn_max_attempts", 0),
patch("roboco.runtime.orchestrator.time.monotonic") as clock,
):
# Cap off: only the cooldown gates, every elapsed window respawns.
for i in range(20):
clock.return_value = 1_000.0 + i * 700
assert orch._notification_spawn_cooled("fe-pm", "stuck") is False
def test_cap_survives_prune() -> None:
"""A capped entry stays suppressed even after a prune sweep fires (the
prune must not drop the count and reset the cap)."""
orch = _orch()
prune_at = AgentOrchestrator._NOTIFICATION_COOLDOWN_PRUNE_AT
with (
patch.object(settings, "notification_spawn_cooldown_seconds", 600),
patch.object(settings, "notification_spawn_max_attempts", 2),
patch("roboco.runtime.orchestrator.time.monotonic") as clock,
):
clock.return_value = 10_000.0
assert orch._notification_spawn_cooled("fe-pm", "stuck") is False
clock.return_value = 10_700.0
assert orch._notification_spawn_cooled("fe-pm", "stuck") is False
clock.return_value = 11_400.0
assert orch._notification_spawn_cooled("fe-pm", "stuck") is True # capped
# Force a prune sweep with many fresh keys.
for i in range(prune_at + 1):
orch._notification_spawn_cooled("be-pm", f"n{i}")
# Advance well past the cooldown so only the surviving cap — not the
# cooldown — can suppress the next "stuck" check. A prune that dropped
# the count would reset the cap and allow a spawn (return False) here.
clock.return_value = 20_000.0
assert orch._notification_spawn_cooled("fe-pm", "stuck") is True
def test_map_prunes_expired_entries() -> None:
orch = _orch()
prune_at = AgentOrchestrator._NOTIFICATION_COOLDOWN_PRUNE_AT