mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
feat(eval): golden-task eval harness + doctrine cohort stamp (#655)
* fix(notifications): exponential backoff + CAS claim for expired-unacked re-escalation The sweep re-escalated every expired unacked ack-required notification on every ~60s tick, forever — the live incident: 3 fresh blocker escalations + Telegram DMs per minute from a static stale pile. Now each notification carries reescalation_count / last_reescalated_at / reescalation_delivered_count (migration 079): first fire at expiry, then doubling intervals from 1h capped at 24h, hard stop after ROBOCO_NOTIFICATION_MAX_REESCALATIONS (default 5) with one permanent log carrying attempts-vs-delivered so 'seen and ignored' is distinguishable from 'route never worked'. The due/wait/capped decision is a pure function in foundation/policy/communications.py. Per adversarial review, the attempt slot is claimed by compare-and-set (UPDATE ... WHERE reescalation_count = :n) BEFORE delivery — the previous draft leaned on the 60s dedup window, which never engages for BLOCKER_ESCALATION (_LOOP_PRONE_TYPES excludes it), so concurrent sweeps would have double-delivered. A lost claim skips delivery outright. Legacy rows read as count=0 and keep today's first-fire semantics. 61 tests incl. a two-session CAS race and a real alembic upgrade/downgrade round trip. * feat(budgets): per-task and per-project cost budgets (flag-gated) tasks.budget_usd + projects.monthly_budget_usd (migration 080, chained on 079; adds ix_agent_spawn_sessions_task_id since both enforcement seams filter on bare task_id). Behind ROBOCO_TASK_BUDGETS_ENABLED (default off, feature-flags card) — verifiably inert when off. Claim-time: a project-month-spend guard applies to WORK-STARTING claims only (i_will_work_on / i_will_plan) — per adversarial review, review/ doc/gate/inbound-PR claims are exempt so in-flight work can always finish reviewing and merging at cap. Spend counts closed sessions' estimated_cost_usd PLUS open sessions priced live from token snapshots (the original closed-only sum read parallel long sessions as $0). Sweep-side: the existing budget sweep also prices the active task's spend vs budget_usd (TaskType defaults when null); on breach the task is BLOCKED (HUMAN resolver, budget marker) BEFORE the graceful stop so the unclaim no-ops and the dispatcher never respawns onto it, and the CEO notification names both recovery steps. unblock on a budget-blocked task re-checks live spend and refuses while still over — no silent re-breach loop. Panel: budget inputs in both dialogs (0 rejected — a zero budget silently blocks everything), spend logic consolidated in TaskService.task_spend_usd. 42 new tests incl. a real-DB spend-query suite and a two-tick non-refire sweep test. * feat(eval): golden-task eval harness + doctrine cohort stamp roboco/eval: 6 BenchTaskSpec fixtures run through the real lifecycle in a disposable environment (the e2e_smoke harness's fake GitHub + local git origin + throwaway DB catalog — real isolation, not convention), scored deterministically (terminal status, revision_count, cycle time, tokens/cost via the agent_spawn_sessions task_id join) plus a local- model judge whose output is nested under a non_deterministic-marked object so cohort diffs don't read judge noise as regression. CLI: python -m roboco.eval run --role <slug> --cohort <name>. Source- checkout-only by declared posture (deptry-scoped ignore + a hard ImportError guard naming why; tests/ never ships in images or wheels). agent_spawn_sessions.doctrine_version (migration 081, chained on 080) is stamped at spawn-session finalize from the composed prompt layers — with the session's model column it identifies a cohort durably. Per adversarial review: bench runs patch the vault flags off (they were writing real markdown into the operator's vault), and the real-spawn OrchestratorStageSpawner is deliberately cut to NotImplementedError — spawned containers' MCP wiring resolves to the production orchestrator under real agent UUIDs, so real spawns wait for a dedicated follow-up; the injectable scripted spawner is the working path. Full suite 13852 passed / 94% coverage in the source worktree; deptry/mypy/xenon clean. --------- Co-authored-by: Renn F <rennf93@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,221 @@
|
||||
"""Unit tests for the eval bench's scorer math (roboco/eval/runner.py).
|
||||
|
||||
Pure dataclass/aggregate-property tests — no DB, no network, no asyncio.
|
||||
`_build_judge_prompt` and `BenchJudge`'s score-parsing regex are covered too
|
||||
since both are pure string logic with no I/O.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from roboco.eval.fixtures import FIXTURES
|
||||
from roboco.eval.runner import (
|
||||
_JUDGE_SCORE_RE,
|
||||
CohortResult,
|
||||
DeterministicMetrics,
|
||||
FixtureResult,
|
||||
JudgeVerdict,
|
||||
OrchestratorStageSpawner,
|
||||
_build_judge_prompt,
|
||||
)
|
||||
|
||||
_EXPECTED_TOTAL_TOKENS = 180
|
||||
_HALF_PASS_RATE = 0.5
|
||||
_COHORT_TOTAL_TOKENS = 600
|
||||
_COHORT_MEAN_CYCLE_SECONDS = 20.0
|
||||
_COHORT_MEAN_JUDGE_SCORE = 5.0
|
||||
_PASSING_JUDGE_SCORE = 4
|
||||
|
||||
|
||||
def _metrics(
|
||||
*,
|
||||
final_status: str = "completed",
|
||||
stalled: bool = False,
|
||||
cycle_time_seconds: float = 10.0,
|
||||
tokens_input: int = 100,
|
||||
tokens_output: int = 50,
|
||||
tokens_cache_read: int = 0,
|
||||
tokens_cache_write: int = 0,
|
||||
estimated_cost_usd: float = 0.01,
|
||||
) -> DeterministicMetrics:
|
||||
return DeterministicMetrics(
|
||||
final_status=final_status,
|
||||
stalled=stalled,
|
||||
revision_count=0,
|
||||
cycle_time_seconds=cycle_time_seconds,
|
||||
tokens_input=tokens_input,
|
||||
tokens_output=tokens_output,
|
||||
tokens_cache_read=tokens_cache_read,
|
||||
tokens_cache_write=tokens_cache_write,
|
||||
estimated_cost_usd=estimated_cost_usd,
|
||||
)
|
||||
|
||||
|
||||
def test_deterministic_metrics_total_tokens_sums_all_four_buckets() -> None:
|
||||
m = _metrics(
|
||||
tokens_input=100, tokens_output=50, tokens_cache_read=25, tokens_cache_write=5
|
||||
)
|
||||
assert m.total_tokens == _EXPECTED_TOTAL_TOKENS
|
||||
|
||||
|
||||
def test_fixture_result_passed_requires_completed_and_not_stalled() -> None:
|
||||
passed = FixtureResult(
|
||||
fixture_key="a",
|
||||
metrics=_metrics(final_status="completed", stalled=False),
|
||||
judge=JudgeVerdict(score=None, rationale=None),
|
||||
)
|
||||
assert passed.passed is True
|
||||
|
||||
cancelled = FixtureResult(
|
||||
fixture_key="b",
|
||||
metrics=_metrics(final_status="cancelled", stalled=False),
|
||||
judge=JudgeVerdict(score=None, rationale=None),
|
||||
)
|
||||
assert cancelled.passed is False
|
||||
|
||||
# A stall that happens to leave the row at "completed" is still not a
|
||||
# pass — `stalled` overrides the status.
|
||||
stalled_completed = FixtureResult(
|
||||
fixture_key="c",
|
||||
metrics=_metrics(final_status="completed", stalled=True),
|
||||
judge=JudgeVerdict(score=None, rationale=None),
|
||||
)
|
||||
assert stalled_completed.passed is False
|
||||
|
||||
|
||||
def _sample_cohort() -> CohortResult:
|
||||
fixtures = [
|
||||
FixtureResult(
|
||||
fixture_key="a",
|
||||
metrics=_metrics(
|
||||
final_status="completed",
|
||||
cycle_time_seconds=10.0,
|
||||
estimated_cost_usd=0.10,
|
||||
tokens_input=100,
|
||||
tokens_output=100,
|
||||
),
|
||||
judge=JudgeVerdict(score=5, rationale="great"),
|
||||
),
|
||||
FixtureResult(
|
||||
fixture_key="b",
|
||||
metrics=_metrics(
|
||||
final_status="needs_revision",
|
||||
stalled=True,
|
||||
cycle_time_seconds=30.0,
|
||||
estimated_cost_usd=0.20,
|
||||
tokens_input=200,
|
||||
tokens_output=200,
|
||||
),
|
||||
judge=JudgeVerdict(score=None, rationale="judge unavailable"),
|
||||
),
|
||||
]
|
||||
return CohortResult(role_slug="be-dev-1", cohort_name="baseline", fixtures=fixtures)
|
||||
|
||||
|
||||
def test_cohort_pass_rate_and_totals() -> None:
|
||||
cohort = _sample_cohort()
|
||||
|
||||
assert cohort.pass_rate == _HALF_PASS_RATE
|
||||
assert cohort.total_cost_usd == pytest.approx(0.3)
|
||||
assert cohort.total_tokens == _COHORT_TOTAL_TOKENS
|
||||
assert cohort.mean_cycle_time_seconds == _COHORT_MEAN_CYCLE_SECONDS
|
||||
# Only fixture "a" has a judge score; "b"'s None is excluded from the mean.
|
||||
assert cohort.mean_judge_score == _COHORT_MEAN_JUDGE_SCORE
|
||||
|
||||
|
||||
def test_cohort_mean_judge_score_is_none_when_no_fixture_was_scored() -> None:
|
||||
fixtures = [
|
||||
FixtureResult(
|
||||
fixture_key="a",
|
||||
metrics=_metrics(),
|
||||
judge=JudgeVerdict(score=None, rationale="judge unavailable"),
|
||||
)
|
||||
]
|
||||
cohort = CohortResult(role_slug="be-dev-1", cohort_name="x", fixtures=fixtures)
|
||||
assert cohort.mean_judge_score is None
|
||||
|
||||
|
||||
def test_cohort_with_no_fixtures_is_a_zero_result_not_a_crash() -> None:
|
||||
cohort = CohortResult(role_slug="be-dev-1", cohort_name="x", fixtures=[])
|
||||
assert cohort.pass_rate == 0.0
|
||||
assert cohort.total_cost_usd == 0.0
|
||||
assert cohort.total_tokens == 0
|
||||
assert cohort.mean_cycle_time_seconds == 0.0
|
||||
assert cohort.mean_judge_score is None
|
||||
|
||||
|
||||
def test_cohort_as_dict_round_trips_every_fixture() -> None:
|
||||
fixtures = [
|
||||
FixtureResult(
|
||||
fixture_key="a",
|
||||
metrics=_metrics(),
|
||||
judge=JudgeVerdict(_PASSING_JUDGE_SCORE, "solid"),
|
||||
),
|
||||
]
|
||||
cohort = CohortResult(role_slug="be-dev-1", cohort_name="x", fixtures=fixtures)
|
||||
payload = cohort.as_dict()
|
||||
|
||||
assert payload["role_slug"] == "be-dev-1"
|
||||
assert payload["cohort_name"] == "x"
|
||||
assert payload["aggregate"]["fixture_count"] == 1
|
||||
assert payload["aggregate"]["pass_rate"] == 1.0
|
||||
# Judge fields live under their own nested, explicitly-marked object —
|
||||
# never flat beside deterministic metrics — so a naive diff can't read
|
||||
# judge noise as a regression.
|
||||
assert "mean_judge_score" not in payload["aggregate"]
|
||||
assert payload["judge"] == {
|
||||
"mean_score": _PASSING_JUDGE_SCORE,
|
||||
"non_deterministic": True,
|
||||
}
|
||||
assert len(payload["fixtures"]) == 1
|
||||
assert payload["fixtures"][0]["fixture_key"] == "a"
|
||||
assert "judge_score" not in payload["fixtures"][0]
|
||||
assert payload["fixtures"][0]["judge"] == {
|
||||
"score": _PASSING_JUDGE_SCORE,
|
||||
"rationale": "solid",
|
||||
"non_deterministic": True,
|
||||
}
|
||||
|
||||
|
||||
def test_judge_score_regex_parses_the_required_reply_shape() -> None:
|
||||
reply = "Score: 4\nRationale: matches the expectation closely.\n"
|
||||
match = _JUDGE_SCORE_RE.search(reply)
|
||||
assert match is not None
|
||||
assert int(match.group(1)) == _PASSING_JUDGE_SCORE
|
||||
|
||||
|
||||
def test_judge_score_regex_is_case_insensitive_and_tolerates_spacing() -> None:
|
||||
assert _JUDGE_SCORE_RE.search("score:5") is not None
|
||||
assert _JUDGE_SCORE_RE.search("SCORE : 3") is not None
|
||||
|
||||
|
||||
def test_judge_score_regex_rejects_out_of_range_scores() -> None:
|
||||
assert _JUDGE_SCORE_RE.search("Score: 0") is None
|
||||
assert _JUDGE_SCORE_RE.search("Score: 6") is None
|
||||
|
||||
|
||||
def test_build_judge_prompt_includes_the_expectation_and_acceptance_criteria() -> None:
|
||||
fixture = FIXTURES[0]
|
||||
prompt = _build_judge_prompt(fixture, diff="+ fixed line", notes="dev notes here")
|
||||
|
||||
assert fixture.title in prompt
|
||||
assert fixture.expectations in prompt
|
||||
for criterion in fixture.acceptance_criteria:
|
||||
assert criterion in prompt
|
||||
assert "+ fixed line" in prompt
|
||||
assert "dev notes here" in prompt
|
||||
|
||||
|
||||
def test_build_judge_prompt_handles_empty_diff_and_notes() -> None:
|
||||
fixture = FIXTURES[0]
|
||||
prompt = _build_judge_prompt(fixture, diff="", notes="")
|
||||
assert "(empty diff)" in prompt
|
||||
assert "(no notes)" in prompt
|
||||
|
||||
|
||||
def test_orchestrator_stage_spawner_is_cut_and_refuses_to_construct() -> None:
|
||||
"""The real-spawn path is deliberately disabled this release (its MCP
|
||||
wiring would authenticate against the REAL production orchestrator) —
|
||||
this is the one runnable check that the cut stays in place."""
|
||||
with pytest.raises(NotImplementedError, match="cut from this release"):
|
||||
OrchestratorStageSpawner()
|
||||
Reference in New Issue
Block a user