mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
Claim-shaped verbs were failing 7/7 (claim_review) and 6/6 (claim_doc_task) as silent 120s FlowVerbTimeout 504s on the NAS: the per-claim ownership repair walked the whole clone issuing two stat syscalls per entry (chown_ms 39502 vs git_ms 8 in the live log), several passes stacked per claim, and the claim transaction held the task row the whole time — so concurrent writers queued behind it into the 60s lock_timeout. The walk now does one stat per entry shared by the chown-skip and chmod-skip checks, and a .git/roboco-owned sentinel (worktree-aware via _resolve_clone_root, written only after a zero-failure pass) skips the walk entirely when the tree is already agent-owned. Every root-side git write invalidates the sentinel BEFORE its subprocess runs — GitService._run_git for scope != none, plus the three raw-subprocess paths inside WorkspaceService the adversarial pass proved bypass it deterministically on the common respawn shape (_worktree_git for mutating verbs, _fetch_branch_ref, _fetch_origin_best_effort) — so a live marker can never vouch for files a root write is about to create. One of those queued writers was the PM journal-decision auto-record: its INSERT hit the lock timeout, _ensure_pm_decision's catch-all swallowed it without rollback, and the poisoned session blew up escalate_up with PendingRollbackError (live incident). The helper's try body now runs in a savepoint — one fix covering all seven PM verbs that route through it — verified empirically against real Postgres in both directions: the failure path leaves the session healthy and the task object readable, and create_entry's internal commit inside the savepoint drains the transactional outbox exactly once.
161 lines
5.2 KiB
Python
161 lines
5.2 KiB
Python
"""The block/unblock flip-flop breaker.
|
|
|
|
Live wedge: fe-pm's escalate_up auto-blocks a task and main_pm's unblock
|
|
resolves it, repeat — 10 flips, 43 spawns, no forward progress, no cycle
|
|
breaker. ``unblock`` now stamps a per-task flip counter
|
|
(``markers.block_flip_count``) and, at exactly the 3rd flip, best-effort
|
|
alerts the CEO once — the unblock itself always still succeeds.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from datetime import UTC, datetime
|
|
from typing import Any
|
|
from unittest.mock import AsyncMock, MagicMock
|
|
from uuid import uuid4
|
|
|
|
import pytest
|
|
from roboco.foundation.policy.content import markers
|
|
from roboco.services.gateway.choreographer import Choreographer, ChoreographerDeps
|
|
from roboco.services.notification import NotificationService
|
|
|
|
# Named constants — ruff PLR2004 forbids magic-value comparisons.
|
|
_TWO_FLIPS = 2
|
|
_THREE_FLIPS = 3
|
|
_FOUR_FLIPS = 4
|
|
|
|
|
|
def _make_deps(**overrides: Any) -> ChoreographerDeps:
|
|
base: dict[str, Any] = {
|
|
"task": AsyncMock(),
|
|
"work_session": AsyncMock(),
|
|
"git": AsyncMock(),
|
|
"a2a": AsyncMock(),
|
|
"journal": AsyncMock(),
|
|
"audit": AsyncMock(),
|
|
"evidence_repo": AsyncMock(),
|
|
}
|
|
base.update(overrides)
|
|
base["journal"].has_decision_for_task.return_value = True
|
|
base["journal"].latest_decision_at.return_value = datetime.now(UTC)
|
|
# _ensure_pm_decision's journal write is savepoint-guarded — an
|
|
# unconfigured AsyncMock's begin_nested() call returns a raw unawaited
|
|
# coroutine, which `async with` cannot use.
|
|
base["task"].session.begin_nested = MagicMock(
|
|
return_value=MagicMock(
|
|
__aenter__=AsyncMock(return_value=None),
|
|
__aexit__=AsyncMock(return_value=False),
|
|
)
|
|
)
|
|
return ChoreographerDeps(**base)
|
|
|
|
|
|
def _flip_setup() -> tuple[Choreographer, Any, Any, Any]:
|
|
"""A blocked task whose ``unblock_with_restore`` returns the SAME mock
|
|
object each call, so the flip-counter marker persists across repeated
|
|
unblock() calls the way it would on one real ORM row across requests.
|
|
"""
|
|
pm_id = uuid4()
|
|
task_id = uuid4()
|
|
t = MagicMock(
|
|
id=task_id,
|
|
status="blocked",
|
|
pre_block_state="in_progress",
|
|
pre_block_assignee=uuid4(),
|
|
pre_block_metadata={},
|
|
dependency_ids=[],
|
|
orchestration_markers=None,
|
|
)
|
|
task_svc = AsyncMock()
|
|
task_svc.get.return_value = t
|
|
task_svc.unblock_with_restore.return_value = t
|
|
task_svc.unmet_dependency_ids.return_value = []
|
|
c = Choreographer(_make_deps(task=task_svc))
|
|
return c, pm_id, task_id, t
|
|
|
|
|
|
async def _unblock_once(c: Choreographer, pm_id: Any, task_id: Any, t: Any) -> Any:
|
|
"""Re-block before each call — a fresh flip in the cycle."""
|
|
t.status = "blocked"
|
|
return await c.unblock(pm_id, task_id, "resolved upstream; restoring")
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_first_and_second_unblock_do_not_notify() -> None:
|
|
c, pm_id, task_id, t = _flip_setup()
|
|
cc: Any = c
|
|
notify = AsyncMock()
|
|
cc._notify_ceo_block_flip = notify
|
|
|
|
for _ in range(2):
|
|
env = await _unblock_once(c, pm_id, task_id, t)
|
|
assert env.error is None, env.as_dict()
|
|
|
|
notify.assert_not_awaited()
|
|
assert markers.get_block_flip_count(t) == _TWO_FLIPS
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_third_unblock_notifies_ceo_once() -> None:
|
|
c, pm_id, task_id, t = _flip_setup()
|
|
cc: Any = c
|
|
notify = AsyncMock()
|
|
cc._notify_ceo_block_flip = notify
|
|
|
|
for _ in range(3):
|
|
env = await _unblock_once(c, pm_id, task_id, t)
|
|
assert env.error is None, env.as_dict()
|
|
|
|
notify.assert_awaited_once_with(task_id, _THREE_FLIPS, t.title)
|
|
assert markers.is_block_flip_notified(t) is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fourth_unblock_does_not_renotify() -> None:
|
|
c, pm_id, task_id, t = _flip_setup()
|
|
cc: Any = c
|
|
notify = AsyncMock()
|
|
cc._notify_ceo_block_flip = notify
|
|
|
|
for _ in range(4):
|
|
env = await _unblock_once(c, pm_id, task_id, t)
|
|
assert env.error is None, env.as_dict()
|
|
|
|
notify.assert_awaited_once()
|
|
assert markers.get_block_flip_count(t) == _FOUR_FLIPS
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_notification_failure_does_not_fail_unblock(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""The real ``_notify_ceo_block_flip`` swallows a notify-service failure —
|
|
unblock still succeeds on the 3rd flip."""
|
|
c, pm_id, task_id, t = _flip_setup()
|
|
monkeypatch.setattr(
|
|
NotificationService,
|
|
"send_block_flip_notification",
|
|
AsyncMock(side_effect=RuntimeError("notification service down")),
|
|
)
|
|
|
|
env = None
|
|
for _ in range(3):
|
|
env = await _unblock_once(c, pm_id, task_id, t)
|
|
assert env.error is None, env.as_dict()
|
|
|
|
assert env is not None
|
|
assert env.error is None
|
|
assert markers.is_block_flip_notified(t) is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_counter_persists_via_marker_across_calls() -> None:
|
|
c, pm_id, task_id, t = _flip_setup()
|
|
cc: Any = c
|
|
cc._notify_ceo_block_flip = AsyncMock()
|
|
|
|
await _unblock_once(c, pm_id, task_id, t)
|
|
assert markers.get_block_flip_count(t) == 1
|
|
await _unblock_once(c, pm_id, task_id, t)
|
|
assert markers.get_block_flip_count(t) == _TWO_FLIPS
|