mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
* fix(board): LEARN decisions name the item, not its per-cycle index A cycle's reject reasons are rendered into the NEXT cycle's exploration prompt, but the ref recorded alongside each reason was the item's stored id (item-0/item-1) — a per-cycle index that means something different every cycle and appears nowhere the explorer can resolve. The reason survived the loop; what it was about did not. Record the item's title instead, via a shared learn_ref() helper (falls back to the id when title-less, and reads target_task_title for Scales, whose items name the live task they mutate). * chore(lint): satisfy ruff 0.16 — keyword-only signatures and markdown formatting The dev toolchain resolved ruff 0.16.0, which stabilises PLR0917 (too many positional arguments) and formats python code blocks inside markdown. Both fired repo-wide and neither had anything to do with the code they flagged. - 36 signatures gain a `*` so their tail arguments are keyword-only, and the 104 call sites that passed them positionally are converted. mypy was the safety net for the static ones; the full suite caught nine more that only bind at runtime (the MCP tool functions, whose real callers already pass named JSON arguments). - 28 markdown files reformatted by 0.16's code-block formatter. - One RUF036 (`None` mid-union) autofixed in the GitLab provider. * fix(gateway): log the reason when a verb rejects A rejected envelope rides an HTTP 200, its body is never logged, and there is no trace table — so in the access log a verb an agent could not satisfy looks identical to one that worked. On 2026-07-25 four Board Programs (Periscope, Sentinel, Scales, Barfly) each POSTed their propose verb three or four times, persisted nothing, and left their exploration tasks PENDING; the reason was unrecoverable afterwards, from the logs or from the agents' own transcripts. Log error/message/remediate/missing plus the calling agent at envelope_to_response — the one chokepoint every v1 flow and do route returns through. Success envelopes stay silent. --------- Co-authored-by: Renn F <rennf93@users.noreply.github.com>
360 lines
13 KiB
Python
360 lines
13 KiB
Python
"""The one-task-per-agent invariant is enforced at the DB level by a
|
|
PostgreSQL transaction-scoped advisory lock keyed by agent_id, acquired in
|
|
``_claim_plan_start_gate`` BEFORE the guard reads (non-coordinator roles
|
|
only) and held until the request transaction commits, so a second concurrent
|
|
claim's guard read sees the first's committed in_progress task and is
|
|
rejected.
|
|
|
|
CRITICAL regression guard: the lock is acquired ONLY for non-coordinator
|
|
roles — acquiring it for a coordinator would serialize a cell_pm / main_pm's
|
|
parallel root planning and regress coordinator concurrency (matches the
|
|
``_COORDINATOR_ROLES`` guard exemption).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from datetime import UTC, datetime
|
|
from typing import Any
|
|
from unittest.mock import AsyncMock, MagicMock
|
|
from uuid import uuid4
|
|
|
|
import pytest
|
|
from roboco.services.gateway.choreographer import Choreographer, ChoreographerDeps
|
|
|
|
# A developer fresh claim must carry a substantive step checklist + rich plan
|
|
# (parity with a PM) so the tracing-gate passes and the flow reaches claim.
|
|
_STEPS = [
|
|
{
|
|
"title": "Implement the change",
|
|
"description": (
|
|
"edit the target file, add tests, run them, and stage the "
|
|
"change for commit on the task branch"
|
|
),
|
|
}
|
|
]
|
|
_GOOD_PLAN = (
|
|
"Append the timestamp HTML comment to the very bottom of README.md without "
|
|
"touching any other line, then commit it on the task branch and open a PR. "
|
|
"Verify the diff is a single-line addition before submitting for QA."
|
|
)
|
|
_GOOD_TC = ["Use a trailing newline so the comment sits on its own line."]
|
|
_GOOD_RISKS = [
|
|
{
|
|
"risk": "An accidental reformat of README.md balloons the diff.",
|
|
"mitigation": "Append only; assert the diff touches one line pre-commit.",
|
|
}
|
|
]
|
|
|
|
|
|
def _make_deps(task: AsyncMock) -> ChoreographerDeps:
|
|
task.session = MagicMock()
|
|
task.session.begin_nested = MagicMock(
|
|
return_value=MagicMock(
|
|
__aenter__=AsyncMock(return_value=None),
|
|
__aexit__=AsyncMock(return_value=False),
|
|
)
|
|
)
|
|
evidence_repo = AsyncMock()
|
|
for method in (
|
|
"list_unread_a2a",
|
|
"list_unread_mentions",
|
|
"list_pending_notifications",
|
|
"task_metadata_gaps",
|
|
"recent_team_activity",
|
|
"blockers_in_lane",
|
|
"journal_highlights_for_task",
|
|
):
|
|
getattr(evidence_repo, method).return_value = []
|
|
journal = AsyncMock()
|
|
journal.latest_decision_at.return_value = datetime.now(UTC)
|
|
return ChoreographerDeps(
|
|
task=task,
|
|
work_session=AsyncMock(),
|
|
git=AsyncMock(),
|
|
a2a=AsyncMock(),
|
|
journal=journal,
|
|
audit=AsyncMock(),
|
|
evidence_repo=evidence_repo,
|
|
)
|
|
|
|
|
|
def _dev_task_svc(agent_id: object, task_id: object) -> AsyncMock:
|
|
"""TaskService mock that completes the dev claim→plan→start flow."""
|
|
pending = MagicMock(
|
|
id=task_id,
|
|
status="pending",
|
|
plan=None,
|
|
assigned_to=None,
|
|
team="backend",
|
|
parent_task_id=None,
|
|
sequence=0,
|
|
task_type="code",
|
|
commits=[],
|
|
pr_number=None,
|
|
branch_name="feature/backend/abc",
|
|
quick_context=None,
|
|
)
|
|
in_progress = MagicMock(
|
|
id=task_id, status="in_progress", plan={"text": "do x"}, assigned_to=agent_id
|
|
)
|
|
task_svc = AsyncMock()
|
|
task_svc.get.return_value = pending
|
|
task_svc.agent_for.return_value = MagicMock(
|
|
id=agent_id, role="developer", team="backend", slug=None
|
|
)
|
|
task_svc.list_in_progress_for_agent.return_value = []
|
|
task_svc.list_paused_for_agent.return_value = []
|
|
task_svc.get_subtasks.return_value = []
|
|
task_svc.claim.return_value = MagicMock(
|
|
id=task_id, status="claimed", plan=None, assigned_to=agent_id
|
|
)
|
|
task_svc.set_plan.return_value = MagicMock(
|
|
id=task_id, status="claimed", plan={"text": "do x"}, assigned_to=agent_id
|
|
)
|
|
task_svc.start.return_value = in_progress
|
|
return task_svc
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_dev_claim_acquires_lock_before_guard_read() -> None:
|
|
"""A developer claim acquires the per-agent advisory lock BEFORE the
|
|
already_active guard reads the agent's other tasks — the ordering that
|
|
closes the TOCTOU (the second concurrent claim's read must see the first's
|
|
committed in_progress task, which requires the lock to be held first)."""
|
|
agent_id = uuid4()
|
|
task_id = uuid4()
|
|
task_svc = _dev_task_svc(agent_id, task_id)
|
|
|
|
# Shared call-order recorder: the lock MUST be acquired before the guard
|
|
# read, so a shared list distinguishes the correct ordering from the bug
|
|
# (read-then-claim, where the lock is too late).
|
|
calls: list[str] = []
|
|
|
|
async def _lock(_aid: object) -> None:
|
|
calls.append("lock")
|
|
|
|
async def _read_in_progress(_aid: object) -> list[Any]:
|
|
calls.append("read_in_progress")
|
|
return []
|
|
|
|
async def _read_paused(_aid: object) -> list[Any]:
|
|
calls.append("read_paused")
|
|
return []
|
|
|
|
task_svc.acquire_claim_lock = _lock
|
|
task_svc.list_in_progress_for_agent.side_effect = _read_in_progress
|
|
task_svc.list_paused_for_agent.side_effect = _read_paused
|
|
|
|
deps = _make_deps(task_svc)
|
|
c = Choreographer(deps)
|
|
|
|
env = await c.i_will_work_on(
|
|
agent_id=agent_id,
|
|
task_id=task_id,
|
|
plan=_GOOD_PLAN,
|
|
steps=_STEPS,
|
|
technical_considerations=_GOOD_TC,
|
|
risks=_GOOD_RISKS,
|
|
)
|
|
assert env.error is None, env.as_dict()
|
|
|
|
# The lock was acquired once for this agent, and BEFORE the guard reads.
|
|
assert calls[:1] == ["lock"], (
|
|
f"advisory lock must be acquired before the guard reads; order was {calls}"
|
|
)
|
|
assert "read_in_progress" in calls and calls.index("lock") < calls.index(
|
|
"read_in_progress"
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_coordinator_claim_does_not_acquire_lock() -> None:
|
|
"""A coordinator PM (cell_pm / main_pm) must NOT acquire the per-agent
|
|
advisory lock — the PM coordinator concurrency feature (CLAUDE.md) lets a
|
|
PM plan + delegate many roots in parallel. Acquiring the lock would
|
|
serialize those claims and REGRESS that feature. The lock is only for the
|
|
one-task-per-agent invariant, which coordinators are exempt from (matching
|
|
the existing ``_COORDINATOR_ROLES`` already_active/paused guard exemption)."""
|
|
pm_id = uuid4()
|
|
task_id = uuid4()
|
|
task_svc = AsyncMock()
|
|
pending = MagicMock(
|
|
id=task_id,
|
|
status="pending",
|
|
plan=None,
|
|
assigned_to=None,
|
|
task_type="planning",
|
|
parent_task_id=None,
|
|
sequence=0,
|
|
team="backend",
|
|
commits=[],
|
|
pr_number=None,
|
|
branch_name=None,
|
|
quick_context=None,
|
|
work_session_id=None,
|
|
)
|
|
task_svc.get.return_value = pending
|
|
task_svc.agent_for.return_value = MagicMock(
|
|
id=pm_id, role="cell_pm", team="backend", slug=None
|
|
)
|
|
task_svc.list_in_progress_for_agent.return_value = []
|
|
task_svc.list_paused_for_agent.return_value = []
|
|
task_svc.get_subtasks.return_value = []
|
|
task_svc.claim.return_value = MagicMock(
|
|
id=task_id, status="claimed", plan=None, assigned_to=pm_id
|
|
)
|
|
task_svc.set_plan.return_value = MagicMock(
|
|
id=task_id, status="claimed", plan={"text": "x"}, assigned_to=pm_id
|
|
)
|
|
task_svc.start.return_value = MagicMock(
|
|
id=task_id, status="in_progress", plan={"text": "x"}, assigned_to=pm_id
|
|
)
|
|
task_svc.ensure_work_session.return_value = None
|
|
|
|
lock_calls: list[object] = []
|
|
|
|
async def _lock(aid: object) -> None:
|
|
lock_calls.append(aid)
|
|
|
|
task_svc.acquire_claim_lock = _lock
|
|
|
|
deps = _make_deps(task_svc)
|
|
c = Choreographer(deps)
|
|
|
|
env = await c.i_will_plan(
|
|
pm_id,
|
|
task_id,
|
|
plan="decompose the feature into cell tasks with acceptance criteria",
|
|
rich_plan={
|
|
"approach": (
|
|
"Two-slice decomposition: backend builds the API + DB migration "
|
|
"first, UX consumes it after the endpoint merges. Strict "
|
|
"sequencing; the UX slice depends on the backend slice landing."
|
|
),
|
|
"sub_tasks": [
|
|
{
|
|
"title": "Backend slice",
|
|
"description": (
|
|
"be-dev-1 implements the API endpoint and DB migration, "
|
|
"with tests, behind the existing service layer."
|
|
),
|
|
},
|
|
{
|
|
"title": "UX slice",
|
|
"description": (
|
|
"fe-dev-1 wires the panel view to the new endpoint and "
|
|
"renders the result list with loading + error states."
|
|
),
|
|
},
|
|
],
|
|
"risks": [
|
|
{
|
|
"risk": "schema migration may block",
|
|
"mitigation": "rehearse the migration on a throwaway DB first",
|
|
}
|
|
],
|
|
},
|
|
)
|
|
body = env.as_dict()
|
|
# The flow MUST reach the claim gate (not short-circuit at input validation)
|
|
# — otherwise ``lock_calls == []`` would pass for the wrong reason.
|
|
assert body.get("error") != "incomplete_input", body
|
|
|
|
# The coordinator did NOT acquire the lock — parallel planning preserved.
|
|
assert lock_calls == [], (
|
|
"coordinator PM must not acquire the per-agent claim lock (would regress "
|
|
"the PM coordinator concurrency feature)"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# the unmet_dependency guard reads dependency state via an unlocked SELECT,
|
|
# then fires release_dependency_blocked_claim (claimed/in_progress -> pending,
|
|
# clears branch_name, abandons WorkSession) BEFORE returning the rejection. If
|
|
# the upstream completes in the microseconds between the read and the release,
|
|
# the task is NEEDLESSLY released. Dependencies are monotonic (unmet -> met,
|
|
# terminal never reopen), so a fresh re-read that now finds them met stays met:
|
|
# safe to proceed without releasing. The fix re-checks unmet_dependency_ids
|
|
# immediately before the release and skips it when the upstream just completed.
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Initial dependency read + the re-check before release.
|
|
_DEP_READ_INITIAL_PLUS_RECHECK = 2
|
|
|
|
|
|
def _dep_task_svc(agent_id: object, task_id: object, dep_id: object) -> AsyncMock:
|
|
"""TaskService mock for the dependency-guard race test. The task carries one
|
|
dependency; the guard reads its state via ``unmet_dependency_ids``."""
|
|
task = MagicMock(
|
|
id=task_id,
|
|
status="claimed",
|
|
assigned_to=agent_id,
|
|
parent_task_id=None,
|
|
dependency_ids=[dep_id],
|
|
team="backend",
|
|
)
|
|
task_svc = AsyncMock()
|
|
task_svc.get.return_value = task
|
|
task_svc.list_in_progress_for_agent.return_value = []
|
|
task_svc.list_paused_for_agent.return_value = []
|
|
return task_svc
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_dependency_guard_skips_release_when_upstream_just_completed() -> None:
|
|
"""The first dependency read sees the upstream still unmet, but the re-check
|
|
sees it completed; the guard must NOT release the task (the dependency is
|
|
now met). Returns None (proceed), no release."""
|
|
agent_id = uuid4()
|
|
task_id = uuid4()
|
|
dep_id = uuid4()
|
|
task_svc = _dep_task_svc(agent_id, task_id, dep_id)
|
|
# First read: upstream still unmet. Re-check: upstream just completed -> met.
|
|
task_svc.unmet_dependency_ids.side_effect = [[dep_id], []]
|
|
|
|
deps = _make_deps(task_svc)
|
|
c = Choreographer(deps)
|
|
|
|
guard = await c._run_claim_guards(
|
|
agent_id=agent_id,
|
|
task=task_svc.get.return_value,
|
|
role_str="developer",
|
|
)
|
|
|
|
# The dependency just completed — the task can proceed, so the guard
|
|
# returns None (no rejection) and does NOT release the task.
|
|
assert guard is None, guard
|
|
task_svc.release_dependency_blocked_claim.assert_not_awaited()
|
|
# The re-check happened (initial read + re-check): the first saw unmet,
|
|
# the second saw met.
|
|
assert task_svc.unmet_dependency_ids.await_count == _DEP_READ_INITIAL_PLUS_RECHECK
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_dependency_guard_releases_when_still_unmet_no_regression() -> None:
|
|
"""No-regression: both the first read AND the re-check see the upstream
|
|
still unmet; the guard releases the task to pending and returns the
|
|
rejection, so the re-check must not weaken the genuine-blocked release
|
|
path."""
|
|
agent_id = uuid4()
|
|
task_id = uuid4()
|
|
dep_id = uuid4()
|
|
task_svc = _dep_task_svc(agent_id, task_id, dep_id)
|
|
# Both reads: upstream still unmet.
|
|
task_svc.unmet_dependency_ids.side_effect = [[dep_id], [dep_id]]
|
|
|
|
deps = _make_deps(task_svc)
|
|
c = Choreographer(deps)
|
|
|
|
guard = await c._run_claim_guards(
|
|
agent_id=agent_id,
|
|
task=task_svc.get.return_value,
|
|
role_str="developer",
|
|
)
|
|
|
|
# Still unmet -> release + reject (the unchanged genuine-blocked path).
|
|
assert guard is not None, guard
|
|
assert guard.error == "invalid_state", guard.as_dict()
|
|
task_svc.release_dependency_blocked_claim.assert_awaited_once_with(task_id)
|
|
assert task_svc.unmet_dependency_ids.await_count == _DEP_READ_INITIAL_PLUS_RECHECK
|