Files
roboco/tests/unit/services/test_findings_consumer_helpers.py
T
cea3e56628 feat(lifecycle): revision findings ledger — structured failure feedback, persisted and delivered down the chain (#486)
* feat(lifecycle): revision findings ledger — structured QA/PR/PM/CEO failure feedback, persisted and delivered down the chain

Every bounce used to survive only as flattened prose: rounds overwrote each
other in notes_structured, request_changes persisted nothing, two raw
dev_notes appends were silently destroyed by the next handoff note, and the
dev prompt pointed at fields (qa_notes via evidence(), pm_notes) the API
never delivered. Agents re-interpreted and re-discovered every failure
before they could start fixing it.

- task_review_findings (migration 071, append-only): file/line/severity/
  criterion(AC-id-validated)/expected/actual/fix/evidence per finding, with
  origin (qa|pr_gate|pm|ceo), round, and an open->addressed->verified
  lifecycle (waived reserved); new tasks.pm_notes + PmReviewContent give
  request_changes a structured home
- producers: fail_review/pr_fail/request_changes take findings=[...] (prose
  issues shimmed+merged for one release, deprecation-logged); ceo_reject
  validates its reason (no 500), lands an origin=ceo finding, and bumps
  round+audit on branchless coordination roots; guardrails at the verb
  chokepoint (nudge >5, hard reject >10, field caps, traversal-safe file);
  the dev_notes data-loss appends are removed; new task.request_changes +
  task.ceo_reject audit events close rework attribution
- delivery: qa_notes/pr_reviewer_notes/pm_notes carry the deterministic
  [F-id8] rendering; claim briefings, evidence(), the REVISION_REQUIRED
  spawn prompt, PM triage bounced-blocks, and A2A bodies deliver open
  findings; round-N+1 QA and gate reviewers get the full prior ledger;
  panel Findings tab + bounced-xN chip; metrics pm_rejects/ceo_rejects +
  findings counts; vault task notes render a Findings section (fail-open)
- resolution closes for every origin: i_am_done and submit_up/submit_root
  take resolved_findings gated by FINDINGS_ADDRESSED (owner-gated so a
  stale non-owner PM can never mutate the ledger); pass_review/pr_pass/
  complete verify-stamp same-transaction; ceo_approve stamps best-effort
- 24 real-DB integration tests drive the full loop through the real
  choreographer; full suite 12856 green

* docs: revision findings ledger sweep — CLAUDE.md, map, RAG corpus

- CLAUDE.md: new ledger section + corrected request_changes row
- docs/map/review-findings.md (new subsystem map) + surgical updates to
  task-service/pr-gate-review/metrics-observability/vault/panel maps
- docs/rag: producers' findings contract across qa/pr-reviewer/developer/
  cell-pm/main-pm/ceo role docs (the PM docs were missing request_changes
  entirely), verb references, and a new architecture/review-findings.md
  disambiguating ledger findings from convention findings

* test(e2e): resubmit resolves the pr_fail finding per the ledger contract

The scripted pr_fail revision loop resubmitted submit_up without
resolved_findings — correctly rejected now that FINDINGS_ADDRESSED gates
the PM resubmit verbs (green locally, red only in CI since the e2e suite
skips without ROBOCO_E2E_SMOKE=1). The scripted PM now reads the open
ledger row pr_fail persisted (new open_finding_ids arc helper) and
resolves it on resubmit, asserting the open set drains — exercising the
coordinator half of the new contract end to end.

---------

Co-authored-by: Renn F <rennf93@users.noreply.github.com>
2026-07-11 22:54:42 +02:00

281 lines
9.0 KiB
Python

"""Real-DB tests for the consumer-side findings helpers in
``choreographer/findings.py`` (``open_findings_for_task``,
``full_ledger_for_task``, ``stamp_addressed_verified``) — the fetch/stamp
layer every evidence/handoff/claim surface and the pass_review/pr_pass
verified-stamp thread through.
Follows ``test_review_findings_repository.py``'s pattern: real Postgres via
the session-scoped test DB (local: ROBOCO_TEST_DB_PORT=55432
ROBOCO_TEST_DB_USER=renzof).
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from uuid import UUID, uuid4
import pytest
from roboco.db.tables import AgentTable, TaskTable
from roboco.foundation.policy.content import Finding, Severity
from roboco.models.base import AgentRole, AgentStatus, TaskStatus, TaskType, Team
from roboco.services.gateway.choreographer import findings as findings_lib
from roboco.services.repositories.review_findings import (
STATUS_ADDRESSED,
STATUS_OPEN,
STATUS_VERIFIED,
STATUS_WAIVED,
ReviewFindingsRepository,
)
if TYPE_CHECKING:
from sqlalchemy.ext.asyncio import AsyncSession
_EXPECTED_TWO = 2
async def _seed_agent(session: AsyncSession) -> UUID:
agent = AgentTable(
id=uuid4(),
name="Consumer Helper Test Agent",
slug=f"consumer-helper-test-{uuid4().hex[:8]}",
role=AgentRole.DEVELOPER,
team=None,
status=AgentStatus.ACTIVE,
model_config={},
system_prompt="consumer helper test",
capabilities=[],
permissions={},
metrics={},
)
session.add(agent)
await session.flush()
return UUID(str(agent.id))
async def _seed_task(session: AsyncSession, created_by: UUID) -> UUID:
task = TaskTable(
id=uuid4(),
title="consumer helper seed task",
description="seed",
acceptance_criteria=["seeded"],
status=TaskStatus.NEEDS_REVISION,
priority=2,
task_type=TaskType.CODE,
team=Team.BACKEND,
created_by=created_by,
)
session.add(task)
await session.flush()
return UUID(str(task.id))
def _finding(**overrides: object) -> Finding:
base: dict[str, object] = {
"file": "roboco/services/task.py",
"line": 10,
"severity": Severity.MAJOR,
"expected": "raises on bad input",
"actual": "swallows the error",
}
base.update(overrides)
return Finding.model_validate(base)
# ---------------------------------------------------------------------------
# open_findings_for_task / full_ledger_for_task
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_open_findings_for_task_excludes_addressed(
db_session: AsyncSession,
) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
rows = await repo.insert_many(
task_id=task_id,
origin="qa",
round=1,
author_slug="be-qa",
findings=[_finding(), _finding(actual="second")],
)
await repo.mark_addressed(task_id, str(rows[0].id), commit="abc", note="fixed")
open_rows = await findings_lib.open_findings_for_task(db_session, task_id)
assert len(open_rows) == 1
assert open_rows[0].id == rows[1].id
@pytest.mark.asyncio
async def test_open_findings_for_task_caps(db_session: AsyncSession) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
findings = [_finding(actual=f"issue {i}") for i in range(3)]
await repo.insert_many(
task_id=task_id, origin="qa", round=1, author_slug="be-qa", findings=findings
)
cap = _EXPECTED_TWO
rows = await findings_lib.open_findings_for_task(db_session, task_id, limit=cap)
assert len(rows) == cap
@pytest.mark.asyncio
async def test_full_ledger_for_task_includes_every_status(
db_session: AsyncSession,
) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
rows = await repo.insert_many(
task_id=task_id,
origin="qa",
round=1,
author_slug="be-qa",
findings=[_finding(), _finding(actual="second")],
)
await repo.mark_addressed(task_id, str(rows[0].id), commit="abc", note="fixed")
full = await findings_lib.full_ledger_for_task(db_session, task_id)
assert len(full) == _EXPECTED_TWO
statuses = {r.status for r in full}
assert statuses == {STATUS_ADDRESSED, STATUS_OPEN}
@pytest.mark.asyncio
async def test_open_and_full_ledger_empty_for_unknown_task(
db_session: AsyncSession,
) -> None:
assert await findings_lib.open_findings_for_task(db_session, uuid4()) == []
assert await findings_lib.full_ledger_for_task(db_session, uuid4()) == []
class _BoomSession:
"""A session stand-in whose ``execute`` always raises — simulates a
ledger-read failure so the fetch helpers' fail-open posture is provable
without a real outage."""
async def execute(self, *_args: object, **_kwargs: object) -> None:
raise RuntimeError("db unavailable")
@pytest.mark.asyncio
async def test_open_findings_for_task_fails_open_on_db_error() -> None:
assert await findings_lib.open_findings_for_task(_BoomSession(), uuid4()) == []
@pytest.mark.asyncio
async def test_full_ledger_for_task_fails_open_on_db_error() -> None:
assert await findings_lib.full_ledger_for_task(_BoomSession(), uuid4()) == []
# ---------------------------------------------------------------------------
# stamp_addressed_verified — origin-scoped, status-scoped verification
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_stamp_verifies_only_addressed_rows_of_the_given_origin(
db_session: AsyncSession,
) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
qa_rows = await repo.insert_many(
task_id=task_id,
origin="qa",
round=1,
author_slug="be-qa",
findings=[_finding(actual="qa addressed"), _finding(actual="qa still open")],
)
gate_rows = await repo.insert_many(
task_id=task_id,
origin="pr_gate",
round=1,
author_slug="be-pr-reviewer",
findings=[_finding(actual="gate addressed")],
)
# Mark one qa finding + the one pr_gate finding addressed; leave the
# second qa finding open.
await repo.mark_addressed(task_id, str(qa_rows[0].id), commit="c1", note="fixed")
await repo.mark_addressed(task_id, str(gate_rows[0].id), commit="c2", note="fixed")
count = await findings_lib.stamp_addressed_verified(
db_session, task_id, origin="qa"
)
assert count == 1
all_rows = await repo.list_for_task(task_id)
by_id = {r.id: r for r in all_rows}
# The addressed qa finding is now verified.
assert by_id[qa_rows[0].id].status == STATUS_VERIFIED
# The still-open qa finding is untouched.
assert by_id[qa_rows[1].id].status == STATUS_OPEN
# The addressed pr_gate finding is untouched — different origin.
assert by_id[gate_rows[0].id].status == STATUS_ADDRESSED
@pytest.mark.asyncio
async def test_stamp_does_not_touch_waived_rows(db_session: AsyncSession) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
rows = await repo.insert_many(
task_id=task_id,
origin="qa",
round=1,
author_slug="be-qa",
findings=[_finding()],
)
await repo.mark_waived(UUID(str(rows[0].id)), "not a real defect")
count = await findings_lib.stamp_addressed_verified(
db_session, task_id, origin="qa"
)
assert count == 0
waived = await repo.list_for_task(task_id, status=STATUS_WAIVED)
assert len(waived) == 1
@pytest.mark.asyncio
async def test_stamp_is_a_noop_when_nothing_addressed(
db_session: AsyncSession,
) -> None:
agent_id = await _seed_agent(db_session)
task_id = await _seed_task(db_session, agent_id)
repo = ReviewFindingsRepository(db_session)
await repo.insert_many(
task_id=task_id,
origin="qa",
round=1,
author_slug="be-qa",
findings=[_finding()],
)
count = await findings_lib.stamp_addressed_verified(
db_session, task_id, origin="qa"
)
assert count == 0
open_rows = await repo.list_for_task(task_id, status=STATUS_OPEN)
assert len(open_rows) == 1
@pytest.mark.asyncio
async def test_stamp_propagates_on_repo_error() -> None:
"""Not best-effort — a repo error must propagate so the caller (pass_review /
pr_pass) fails the whole verb cleanly instead of silently landing a
passed/gated task against a stale ledger."""
with pytest.raises(RuntimeError):
await findings_lib.stamp_addressed_verified(
_BoomSession(), uuid4(), origin="qa"
)
if __name__ == "__main__":
pytest.main([__file__, "-q"])