From 64f3e2b3bd9c743a9c39bf33e02f81343008bf0d Mon Sep 17 00:00:00 2001 From: istos Date: Sun, 2 Aug 2026 13:42:38 +0200 Subject: [PATCH] A board whose agents cannot run anything says so MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit BOARD_AGENT_COMMANDS is the one setting a headless agent cannot work around, and since the install stopped asking for it a project the detector does not recognise starts with it empty — correctly, but silently, until a run ended with an agent explaining it could not verify its work. Say it twice, in the two places it is learnable: a quiet `no agent commands` chip in the header (`--idle`, like the drive's "no driver", never `--alarm` — nothing is failing, something is unconfigured), and a note appended to the ticker line of the launches that would have run those commands, work and act-pr. Neither blocks anything: an agent that only edits files is still useful. What counts as empty is answered once, by `config.agent_commands()`, which splits exactly as the adapters' own `split_commands()` does — so whitespace and a lone comma are nothing configured on the board as well as at the launch, and the page reads the server's boolean rather than the raw setting. --- AGENTS.md | 13 ++ manager/core/.env.example | 4 +- manager/core/agents.py | 26 ++- manager/core/board.html | 23 +++ manager/core/config.py | 10 ++ manager/core/httpd.py | 4 + tests/test_no_agent_commands.py | 282 ++++++++++++++++++++++++++++++++ 7 files changed, 358 insertions(+), 4 deletions(-) create mode 100644 tests/test_no_agent_commands.py diff --git a/AGENTS.md b/AGENTS.md index d52a02c..25d06cb 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -193,6 +193,19 @@ in. A wrong guess costs nothing an empty value would not, since a prefix that matches nothing denies exactly the same way. An existing value is never overwritten, so `--setup` cannot undo a hand-edit. +Empty is honest, but honest and invisible is a board that lets an agent +discover the problem on your behalf, slowly — so the board says it out +loud. With `BOARD_AGENT_COMMANDS` holding nothing (unset, whitespace, or +a lone comma, which is how the adapters read it too) the header carries a +quiet `no agent commands` chip naming the setting and the file that holds +it, `manager/local/.env`, and every launch of the two intents that would +have run those commands — **▸ start work** and **↻ act on PR** — appends +the same sentence to its line in the ticker. It is a configuration fact, +not an alarm: it wears the settled register rather than `--alarm`, it +appears only while the setting is empty, and it never refuses a launch. +An agent that only edits files is still useful, and bench does not +decline work because a project is unconfigured. + Bare Enter takes the default, Ctrl-D skips the rest, and what lands is `core/.env.example` with the answers substituted into their lines: every other key, every comment, so the written file is where the project reads diff --git a/manager/core/.env.example b/manager/core/.env.example index ca16723..450f50d 100644 --- a/manager/core/.env.example +++ b/manager/core/.env.example @@ -46,7 +46,9 @@ BOARD_AGENT_MODEL_REVIEW= # (each adapter renders them into its vendor's permission rules; the # git/gh grants per launch intent are built in). Headless runs have no # human at a permission prompt, so a test runner missing from this list -# is a test the work agent cannot run. +# is a test the work agent cannot run. Left empty — or holding only +# whitespace or a lone comma — the board says so: a quiet `no agent +# commands` chip in the header, and a note on every work launch. BOARD_AGENT_COMMANDS=python3 -m unittest # What counts as a definition-of-done check (the Focus view's CHECKS diff --git a/manager/core/agents.py b/manager/core/agents.py index 7b1ffb4..73a2c26 100644 --- a/manager/core/agents.py +++ b/manager/core/agents.py @@ -254,6 +254,21 @@ def _launch(mode: str, prompt: str, cwd: Path, agent_id: str, filename: str, log return proc, log_file, model +def _no_commands_note() -> str | None: + """What a launch owes the ticker when the project configured nothing for + its agents to run: this run can edit and commit, but it cannot check its + own work. Only the two intents that would have run the commands say it + (work and act-pr); a read-only kind never had them. + + A note, never a refusal — an agent that only edits files is still + useful, and bench does not decline work because a project is + unconfigured. The header says the same thing standing still.""" + if config.agent_commands(): + return None + return ("no project commands configured, so it cannot run this project's " + "tests — set BOARD_AGENT_COMMANDS in manager/local/.env") + + def _fresh_branch_point() -> tuple[str | None, str | None]: """Where a brand-new task branch should start: the newest main that exists. With an `origin` remote, fetch its main (bounded by @@ -429,8 +444,9 @@ def start_agent(filename: str, stage: str, takeover: bool = False) -> dict: summary = (f"{name} is back on {filename} — continuing branch {branch}" if continuing else f"{name} started on {filename} (branch {branch})") - if base_note: - summary += f" — {base_note}" + for note in (base_note, _no_commands_note()): + if note: + summary += f" — {note}" state.record_board_event({ "kind": "agent", "actor": "agent", "file": filename, "summary": summary, @@ -739,9 +755,13 @@ def start_pr_fix(filename: str, stage: str) -> dict: } with state.LOCK: state.AGENTS[agent_id] = record + summary = f"{name} is acting on the review of {filename}'s PR" + note = _no_commands_note() + if note: + summary += f" — {note}" state.record_board_event({ "kind": "agent", "actor": "agent", "file": filename, - "summary": f"{name} is acting on the review of {filename}'s PR", + "summary": summary, }) threading.Thread(target=_reap_pr_fix, args=(agent_id, proc, log_file), daemon=True).start() diff --git a/manager/core/board.html b/manager/core/board.html index b1ef4ea..6736540 100644 --- a/manager/core/board.html +++ b/manager/core/board.html @@ -661,6 +661,7 @@ +
@@ -1066,6 +1067,27 @@ function renderSync() { `${esc(detail.split(' — ')[0].replace(/^sync[^:]*:\s*/, ''))}`; } +/* The commands a headless agent may run are the one thing it cannot work + around: with none configured a work agent still edits and commits, but it + can never run this project's tests, and nothing said so until a run had + already ended saying it. Same shape as the drive's "no driver" — a quiet + statement of what this project has not set up, in the settled register, + never `--alarm`: nothing is failing here. A project that set it, by + detection or by hand, sees nothing at all. */ +function renderAgentCommands() { + const el = $('#cmdchip'); + const missing = S.state && S.state.hasAgentCommands === false; + el.hidden = !missing; + if (!missing) return; + el.title = 'Headless agents have no project commands to run here, so a ' + + "work agent cannot run this project's tests. Set BOARD_AGENT_COMMANDS " + + 'in manager/local/.env — comma-separated command prefixes, e.g. ' + + '"npm test". Work still runs; the agent simply cannot check itself.'; + el.innerHTML = `` + + `no agent commands` + + `BOARD_AGENT_COMMANDS`; +} + /* ── a phase in flight ────────────────────────────────────────────────── */ /* The phases the header has something to say about: one running, or one @@ -1214,6 +1236,7 @@ function render() { renderPhases(); renderViews(); renderSync(); + renderAgentCommands(); if (S.view === 'flight') renderFlight(); else if (S.view === 'focus') renderFocus(); else renderCards(); diff --git a/manager/core/config.py b/manager/core/config.py index 3e80f24..d74acd1 100644 --- a/manager/core/config.py +++ b/manager/core/config.py @@ -105,6 +105,16 @@ ADAPTER = setting("BOARD_AGENT_ADAPTER", "claude") # the adapter's own knowledge; this list is the project's half. AGENT_COMMANDS = setting("BOARD_AGENT_COMMANDS", "python3 -m unittest") + +def agent_commands() -> list[str]: + """The prefixes as an adapter reads them: comma-separated, blanks + dropped — so whitespace, or a lone comma, is exactly nothing + configured. The adapters split the same string in their own standalone + copies of `split_commands()`; this is core's, and it is what the board + asks so the page and the launch can never disagree about *empty*.""" + return [part.strip() for part in AGENT_COMMANDS.split(",") if part.strip()] + + # Model per launch intent — an opaque vendor-native name core passes to the # adapter untranslated (what names mean anything is vendor knowledge). Empty # = inherit the vendor's own default, exactly today's behaviour. A per-intent diff --git a/manager/core/httpd.py b/manager/core/httpd.py index 870cce1..ba87ad9 100644 --- a/manager/core/httpd.py +++ b/manager/core/httpd.py @@ -42,6 +42,10 @@ def state_payload() -> dict: "phases": phases.public_state(), "drive": drive.public(), "hasDriver": config.driver_path() is not None, + # whether this project gave its headless agents anything to run. The + # one setting an agent cannot work around, so the board says it is + # missing rather than letting a run discover it — see the header chip + "hasAgentCommands": bool(config.agent_commands()), "branches": github.task_branches(), "commands": config.commands(), "commandRuns": commands.public(), diff --git a/tests/test_no_agent_commands.py b/tests/test_no_agent_commands.py new file mode 100644 index 0000000..85be4bb --- /dev/null +++ b/tests/test_no_agent_commands.py @@ -0,0 +1,282 @@ +"""A board whose agents cannot run anything says so (task 46). + +`BOARD_AGENT_COMMANDS` is the one setting a headless agent cannot work +around: a test runner missing from it is a test the work agent cannot run. +Since the install stopped asking for it, a project the detector does not +recognise starts with it empty — and until this card nothing mentioned that +until a run had already ended with an agent explaining it could not verify +its work. + +Three places have to agree on *empty*: the splitter (core's, and the +standalone copies each adapter carries), the state payload the page reads, +and the launch that says it in the ticker. So the suite is one class each, +plus the page's own function run under node. + + python3 -m unittest discover -s tests -v +""" + +from __future__ import annotations + +import importlib.util +import json +import re +import shutil +import subprocess +import sys +import unittest +from pathlib import Path + +REPO = Path(__file__).resolve().parents[1] +CORE = REPO / "manager" / "core" +sys.path.insert(0, str(CORE)) + +import agents # noqa: E402 +import config # noqa: E402 + +from tests.test_phase_runs import PhaseCase, card, git # noqa: E402 + +BOARD = CORE / "board.html" +NODE = shutil.which("node") +ALONE = "77-alone.md" +PR_URL = "https://github.com/acme/widget/pull/7" + +# The forms of "nothing configured" a person can actually produce: never +# set, cleared, left as whitespace, or reduced to the separator. +NOTHING = ("", " ", "\t", ",", " , ", ",,", " ,\t,") +SOMETHING = {"npm test": ["npm test"], + " npm test ": ["npm test"], + "npm test, make check": ["npm test", "make check"], + "python3 -m unittest,": ["python3 -m unittest"]} + + +def _load(name: str, path: Path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class WhatCountsAsNothing(unittest.TestCase): + """`config.agent_commands()` is the board's half of the split the + adapters already do. If the two disagreed, a board could show a chip a + launch contradicts, or stay silent about a run with nothing to run.""" + + def commands(self, raw: str) -> list[str]: + saved = config.AGENT_COMMANDS + self.addCleanup(setattr, config, "AGENT_COMMANDS", saved) + config.AGENT_COMMANDS = raw + return config.agent_commands() + + def test_every_shape_of_empty_is_empty(self): + for raw in NOTHING: + self.assertEqual(self.commands(raw), [], + f"{raw!r} configures nothing") + + def test_a_value_survives_stripped(self): + for raw, expected in SOMETHING.items(): + self.assertEqual(self.commands(raw), expected) + + def test_it_agrees_with_the_adapters(self): + """Both shipped adapters carry a standalone `split_commands()` (the + hooks run outside the board's imports). Same answers, or the board + is guessing about somebody else's rules.""" + splitters = [ + _load("claude_hook_settings", + CORE / "adapters" / "claude" / "hook_settings.py").split_commands, + _load("opencode_permission_config", + CORE / "adapters" / "opencode" / "permission_config.py").split_commands, + ] + for raw in (*NOTHING, *SOMETHING): + for split in splitters: + self.assertEqual(self.commands(raw), split(raw), + f"core and an adapter disagree about {raw!r}") + + +class TheStatePayloadSaysWhichItIs(unittest.TestCase): + """One boolean, beside `hasDriver` — the same shape of fact, answered by + the server so the page never re-parses the setting itself.""" + + def flag(self, raw: str) -> bool: + import httpd + saved = config.AGENT_COMMANDS + self.addCleanup(setattr, config, "AGENT_COMMANDS", saved) + config.AGENT_COMMANDS = raw + return httpd.state_payload()["hasAgentCommands"] + + def test_empty_is_reported_as_empty(self): + for raw in NOTHING: + self.assertFalse(self.flag(raw), f"{raw!r} is nothing configured") + + def test_a_configured_board_reports_true(self): + self.assertTrue(self.flag("npm test")) + self.assertTrue(self.flag("python3 -m unittest")) + + def test_the_page_does_not_read_the_raw_setting(self): + html = BOARD.read_text(encoding="utf-8") + self.assertNotIn("state.agentCommands", html) + self.assertIn("hasAgentCommands", html) + + +@unittest.skipUnless(NODE, "node is needed to run the page's own rules") +class TheHeaderChip(unittest.TestCase): + """The indicator itself, run as the page runs it: a stub element in + place of the DOM, a state payload in place of the server.""" + + @classmethod + def setUpClass(cls): + cls.html = BOARD.read_text(encoding="utf-8") + match = re.search(r"function renderAgentCommands\(\) \{.*?\n\}", + cls.html, re.S) + if match is None: + raise AssertionError("board.html no longer defines renderAgentCommands") + cls.src = match.group(0) + + def chip(self, state: dict | None) -> dict: + script = ( + "const el = {hidden: null, title: '', innerHTML: ''};\n" + "function $(sel) { if (sel !== '#cmdchip') " + "throw new Error('unexpected ' + sel); return el; }\n" + "var S = " + json.dumps({"state": state}) + ";\n" + + self.src + "\nrenderAgentCommands();\n" + "console.log(JSON.stringify(el));\n") + out = subprocess.run([NODE, "-e", script], capture_output=True, text=True) + self.assertEqual(out.returncode, 0, out.stderr) + return json.loads(out.stdout) + + def test_an_empty_setting_is_said_once_naming_setting_and_file(self): + chip = self.chip({"hasAgentCommands": False}) + + self.assertFalse(chip["hidden"]) + self.assertIn("BOARD_AGENT_COMMANDS", chip["innerHTML"] + chip["title"]) + self.assertIn("manager/local/.env", chip["title"], + "a reader who has never opened that file has to be told " + "which file it is") + + def test_a_configured_board_shows_nothing_at_all(self): + chip = self.chip({"hasAgentCommands": True}) + + self.assertTrue(chip["hidden"]) + self.assertEqual(chip["innerHTML"], "", + "nothing is drawn, not even hidden") + + def test_a_payload_that_does_not_say_is_not_guessed_at(self): + """An older server, or the first frame before any state: the board + says nothing rather than accusing a project of being unconfigured.""" + self.assertTrue(self.chip(None)["hidden"]) + self.assertTrue(self.chip({})["hidden"]) + + def test_it_is_not_an_alarm(self): + """Nothing is failing — something is unconfigured. `--alarm` is + reserved for blocked, failed or HIGH.""" + self.assertNotIn("--alarm", self.src) + self.assertNotIn("--accent", self.src, "and nothing here is working") + self.assertIn("var(--idle)", self.src) + + def test_the_chip_exists_and_is_drawn_every_frame(self): + self.assertIn('id="cmdchip"', self.html) + render = re.search(r"function render\(\) \{.*?\n\}", self.html, re.S).group(0) + self.assertIn("renderAgentCommands();", render) + + +class ALaunchWithNothingToRun(PhaseCase): + """The sharper half: the moment it matters is the launch. It is a note + in the ticker beside the run's own line — never a refusal, because an + agent that only edits files is still useful.""" + + SETTING = "BOARD_AGENT_COMMANDS in manager/local/.env" + + def setUp(self): + super().setUp() + self.write(ALONE, card("77 — On its own", status="In Progress"), + "in-progress") + + def said(self) -> list[str]: + return [s for s in self.summaries() if self.SETTING in s] + + def test_a_work_launch_says_it_and_still_runs(self): + self.patch(AGENT_COMMANDS="") + + agents.start_agent(ALONE, "in-progress") + self.settle() + + self.assertEqual(len(self.said()), 1, self.summaries()) + self.assertIn("cannot run this project's tests", self.said()[0]) + self.assertEqual(self.stage_of(ALONE), "review", + "the launch was not blocked: it ran, committed and landed") + + def test_a_configured_board_never_mentions_it(self): + self.patch(AGENT_COMMANDS="npm test") + + agents.start_agent(ALONE, "in-progress") + self.settle() + + self.assertEqual(self.said(), []) + self.assertFalse([s for s in self.summaries() + if "BOARD_AGENT_COMMANDS" in s]) + + def test_whitespace_and_a_lone_comma_count_as_nothing_here_too(self): + for raw in (" ", ","): + with self.subTest(raw=raw): + self.patch(AGENT_COMMANDS=raw) + self.assertIsNotNone(agents._no_commands_note()) + + def test_the_note_rides_beside_the_branch_point_note(self): + """Two things worth saying about one launch, one line: the note is + appended, it does not replace what the launch already said.""" + self.patch(AGENT_COMMANDS="") + + agents.start_agent(ALONE, "in-progress") + self.settle() + + line = self.said()[0] + self.assertIn(f"started on {ALONE}", line) + self.assertIn(f"task/{ALONE[:-3]}", line) + + def test_acting_on_a_pr_says_it_too(self): + """↻ act on PR is a work agent with a push — same intent, same + commands, same silence to break.""" + self.patch(AGENT_COMMANDS="") + text = card("77 — On its own", status="Review").replace( + "**Priority:** High\n", f"**Priority:** High\n**PR:** {PR_URL}\n") + (self.tasks / "in-progress" / ALONE).unlink() + self.write(ALONE, text, "review") + git(self.repo, "branch", f"task/{ALONE[:-3]}") + + agents.start_pr_fix(ALONE, "review") + self.settle() + + self.assertEqual(len(self.said()), 1, self.summaries()) + self.assertIn("acting on the review", self.said()[0]) + + def test_a_read_only_kind_says_nothing(self): + """`◔ still true?` never had the project's commands: telling it + about them would be noise on every card in every stage.""" + self.patch(AGENT_COMMANDS="") + self.adapter_is("#!/usr/bin/env python3\n" + "print('RELEVANCE REVIEW: Still relevant')\n") + + agents.start_review(ALONE, "in-progress") + self.settle() + + self.assertEqual(self.said(), []) + + +class TheDocumentedTrade(unittest.TestCase): + """The install stopped asking on purpose; AGENTS.md carries the reason, + so it has to carry what the board now does about it.""" + + @classmethod + def setUpClass(cls): + doc = (REPO / "AGENTS.md").read_text(encoding="utf-8") + cls.flat = re.sub(r"\s+", " ", doc.replace("**", "").replace("`", "")) + + def test_the_indicator_is_written_down(self): + self.assertIn("no agent commands", self.flat) + self.assertIn("BOARD_AGENT_COMMANDS", self.flat) + + def test_it_says_the_launch_is_not_blocked(self): + self.assertIn("never refuses a launch", self.flat) + + +if __name__ == "__main__": + unittest.main()