feat(grok): convert interactive intake/secretary to the grok CLI; delete opencode

Move the last Grok runtime off opencode onto xAI's official `grok` CLI, for full
parity with the Claude path. The intake/secretary chat now runs per-turn headless
`grok -p` invocations that resume one session id (proven live: context carries
across runs), with streaming-json deltas mapped to the existing panel StreamChunk
kinds — the IntakeDriver loop, message source, relay, and idle reaper are reused
unchanged; only the SessionFactory differs (GrokCliSession replaces the
opencode-serve session).

- GrokCliSession + a pure, unit-tested streaming-json -> StreamChunk assembler
  (thought coalesced to one block, text streamed live, end captures the session
  id for -r, fenced-draft fallback, clear errors incl. rate-limit).
- intake propose_draft and secretary read_company_state/read_task/submit_directive
  are now FastMCP servers (roboco-intake / roboco-secretary) wired into
  ~/.grok/config.toml, launched via `uv run --directory /app` to resolve the
  installed package. The secretary tools reuse the shared backend helpers.
- Orchestrator: interactive spawn mounts the subscription auth + per-agent usage
  dir (no metered xAI key, no permission env — grok flags carry per-role perms);
  usage/cost now read a captured usage.json (drop the opencode.db reader, the
  _opencode_db_path/_grok_usage_from_opencode methods, and the cost-cap's
  opencode read). hosts["opencode"] -> hosts["grok_usage"]; OPENCODE_DATA_DIR ->
  GROK_USAGE_DATA_DIR.
- Fix one-shot usage capture: `-s` does not pin the session id (grok generates
  its own), so the entrypoint now reads the real id back from the JSON run log
  and the reader uses it; usage is captured per-turn on the interactive path.
- Delete the opencode layer: opencode_config/opencode_usage/opencode_session, the
  docker/grok/*.js plugins, the old one-shot entrypoint, and their tests.
- Compose (all three files), .env.example, and stale comments updated to the
  grok-CLI runtime; add the SuperGrok auth mount + grok-usage dir.

Gate green: ruff, mypy (296 files), xenon, tests. NAS build/verify pending.
This commit is contained in:
Renn F
2026-06-19 04:42:25 +02:00
parent 499f6fc509
commit a88045aacf
40 changed files with 1307 additions and 2200 deletions
+45 -67
View File
@@ -1,13 +1,15 @@
"""GROK agents capture token usage/cost from their opencode SQLite store.
"""GROK agents capture token usage/cost from their captured ``usage.json``.
A Grok agent runs opencode — no SDK /usage/status server and no Claude
transcript — so finalize must read opencode.db (mounted into the orchestrator)
instead. Reasoning folds into output (it bills at the output rate).
A Grok agent runs the grok CLI — no SDK /usage/status server and no Claude
transcript — so finalize reads the ``usage.json`` the entrypoint / interactive
driver wrote to the per-agent data dir (mounted into the orchestrator). grok
reports a single cumulative total with no input/output split, so it folds into
output (it bills at the output rate).
"""
from __future__ import annotations
import sqlite3
import json
from typing import TYPE_CHECKING
import pytest
@@ -18,82 +20,58 @@ if TYPE_CHECKING:
from pathlib import Path
def _make_db(path: Path, cols: dict[str, float]) -> None:
con = sqlite3.connect(path)
con.execute(
"CREATE TABLE session (id TEXT, tokens_input INT, tokens_output INT, "
"tokens_cache_read INT, tokens_cache_write INT, tokens_reasoning INT, "
"cost REAL)"
)
con.execute(
"INSERT INTO session (id, tokens_input, tokens_output, tokens_cache_read, "
"tokens_cache_write, tokens_reasoning, cost) VALUES (?,?,?,?,?,?,?)",
(
"s1",
cols["tokens_input"],
cols["tokens_output"],
cols["tokens_cache_read"],
cols["tokens_cache_write"],
cols["tokens_reasoning"],
cols["cost"],
def _write_usage(path: Path, total_tokens: int, cost_usd: float) -> None:
path.write_text(
json.dumps(
{"model": "grok-build", "total_tokens": total_tokens, "cost_usd": cost_usd}
),
encoding="utf-8",
)
con.commit()
con.close()
def test_grok_usage_folds_reasoning_into_output(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
db = tmp_path / "opencode.db"
_make_db(
db,
{
"tokens_input": 100,
"tokens_output": 50,
"tokens_reasoning": 30,
"tokens_cache_read": 10,
"tokens_cache_write": 5,
"cost": 0.02,
},
)
orch = AgentOrchestrator.__new__(AgentOrchestrator)
monkeypatch.setattr(orch, "_opencode_db_path", lambda _aid: str(db))
# reasoning (30) folded into output (50) → 80; bills at the output rate.
assert orch._grok_usage_from_opencode("be-dev-1") == (100, 80, 10, 5)
def test_grok_usage_zero_when_store_missing(
def test_grok_usage_folds_total_into_output(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
usage = tmp_path / "usage.json"
_write_usage(usage, total_tokens=180, cost_usd=0.02)
orch = AgentOrchestrator.__new__(AgentOrchestrator)
monkeypatch.setattr(
orch, "_opencode_db_path", lambda _aid: str(tmp_path / "absent.db")
orch, "_grok_usage_json", lambda _aid: json.loads(usage.read_text())
)
assert orch._grok_usage_from_opencode("be-dev-1") == (0, 0, 0, 0)
# The whole total folds into output (no input/output split from the CLI).
assert orch._grok_usage_tokens("be-dev-1") == (0, 180, 0, 0)
def test_grok_usage_zero_when_store_missing(monkeypatch: pytest.MonkeyPatch) -> None:
orch = AgentOrchestrator.__new__(AgentOrchestrator)
monkeypatch.setattr(orch, "_grok_usage_json", lambda _aid: None)
assert orch._grok_usage_tokens("be-dev-1") == (0, 0, 0, 0)
def test_grok_cost_read_from_usage_json(monkeypatch: pytest.MonkeyPatch) -> None:
captured_cost = 3.25
orch = AgentOrchestrator.__new__(AgentOrchestrator)
monkeypatch.setattr(
orch,
"_grok_usage_json",
lambda _aid: {"cost_usd": captured_cost, "total_tokens": 9},
)
assert orch._grok_cost_usd("be-dev-1") == captured_cost
monkeypatch.setattr(orch, "_grok_usage_json", lambda _aid: None)
assert orch._grok_cost_usd("be-dev-1") == 0.0
@pytest.mark.asyncio
async def test_resolve_final_usage_routes_grok_to_opencode(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
async def test_resolve_final_usage_routes_grok_to_usage_json(
monkeypatch: pytest.MonkeyPatch,
) -> None:
db = tmp_path / "opencode.db"
_make_db(
db,
{
"tokens_input": 7,
"tokens_output": 3,
"tokens_reasoning": 2,
"tokens_cache_read": 0,
"tokens_cache_write": 0,
"cost": 0.01,
},
)
orch = AgentOrchestrator.__new__(AgentOrchestrator)
monkeypatch.setattr(orch, "_opencode_db_path", lambda _aid: str(db))
monkeypatch.setattr(
orch, "_grok_usage_json", lambda _aid: {"total_tokens": 12, "cost_usd": 0.01}
)
cfg = type("C", (), {"provider_type": "grok"})()
orch._instances = {"be-dev-1": AgentInstance(agent_id="be-dev-1", config=cfg)}
# No SDK fetch / transcript read for GROK — usage comes from opencode.db.
assert await orch._resolve_final_token_usage("be-dev-1") == (7, 5, 0, 0)
# No SDK fetch / transcript read for GROK — usage comes from usage.json.
assert await orch._resolve_final_token_usage("be-dev-1") == (0, 12, 0, 0)