Files
Youssef 52aa127d60 The harness that scored the old skill had a third instance of the same bug
The runs archive copied a fixed filename list, index.html, tokens.css,
styles.css, and stampPresent scans that archive. A swept build whose stamp
lives in page.css scored skillLoaded=false with a correct stamp on disk,
which is the same stale assumption already fixed in stampPresent and in the
reference budget, one function lower. The archive now takes every top-level
html and css the build produced.

The three affected run.json files are rescored from the real artifacts:
only the buggy cell changed (inspo false to true); the bare control stays
false, which is the leak check holding. Also archives the second three-arm
run: bare 56 FAIL with the same six numbered eyebrows, both Hallmark arms
clean, and the control now provably offered zero MCP tools.
2026-08-06 15:17:36 +01:00

286 lines
15 KiB
JavaScript

#!/usr/bin/env node
// Tier A eval harness: drive the REAL Claude Code binary against a brief with
// the Hallmark skill installed, and measure conformance on the real harness
// (load-order discipline, stamp, files written) - not just bare-API
// instruction-following like Tier B (gen-direct.py).
//
// node gen-cli.mjs --brief b1|all --arm claude|bare|glm|kimi|all [--force] [--dry-run]
// [--model sonnet|opus|...] [--effort low|medium|high|max]
// [--budget-usd 5]
//
// --budget-usd caps spend per cell (default 5). A measured derived build on
// Opus lands near $3 and stops mid Step-7 at the old cap of 3, which scores a
// complete artifact as an error_max_budget_usd failure and never exercises the
// gate sweep. Raise it rather than reading a truncated run as a skill result.
//
// A/B: the "bare" arm is the no-skill control - same binary, same auth, plain
// brief prompt, NO skill copied, and --setting-sources project on the call so
// the user-scope ~/.claude/skills install cannot leak in (applied to BOTH arms
// so the skill arm always runs the project-scope copy). stampPresent on a bare
// cell doubles as the leak detector (must be false).
//
// IMPORTANT (measured): skills do NOT auto-trigger in `claude -p` headless mode,
// so we invoke the skill BY NAME ("/hallmark <brief>"). This is a CONFORMANCE
// harness (does the skill FOLLOW its discipline), not an auto-trigger test.
//
// Auth: the `claude` arm sets NO key and uses your Claude subscription -
// run it in a terminal where `claude` is logged in. The open-model arms
// (glm via Z.ai, kimi via Moonshot) are Anthropic-shaped and DISABLED until you
// add ZAI_AUTH_TOKEN / MOONSHOT_AUTH_TOKEN for them; Together/OpenRouter cannot back
// `claude -p` (it needs an Anthropic-shaped endpoint).
//
// Output: eval/runs/<brief>/<arm>-cli/{index.html?, transcript.jsonl, run.json}
// reusing the Tier B run.json schema plus Tier-A fields (skillLoaded, filesWritten,
// refReads, loadOrderOk).
import { readFileSync, writeFileSync, mkdirSync, existsSync, cpSync, rmSync, readdirSync } from "node:fs";
import { spawn } from "node:child_process";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
const ROOT = dirname(fileURLToPath(import.meta.url));
const REPO = dirname(ROOT);
const SKILL_SRC = join(REPO, "skills", "hallmark");
const RUNS = join(ROOT, "runs");
const SCRATCH = join(ROOT, "_cli-scratch");
// --- arms (Tier A only; Anthropic-shaped endpoints) ------------------------
const ARMS = {
"claude": { model: null, enabled: true, skill: true, note: "Hallmark arm; Claude subscription; set NO key, must be logged in." },
"bare": { model: null, enabled: true, skill: false, note: "No-skill control; same binary and auth." },
// Hallmark with a reference archive attached. Since fa12b99 signal 8 inverts the
// flow: the archive owns Steps 1-6 and Hallmark enters at Step 7 as a sweep, so
// this arm measures a different code path from "claude", not a better-resourced
// version of it. Local stdio server, no credentials in the config.
"inspo": { model: null, enabled: true, skill: true, mcp: true, note: "Hallmark + Inspo MCP; exercises the signal-8 sweep path." },
"glm": { model: "glm-5.2[1m]", base: "https://api.z.ai/api/anthropic", tokenEnv: "ZAI_AUTH_TOKEN", enabled: false },
"kimi": { model: "kimi-k3", base: "https://api.moonshot.ai/anthropic", tokenEnv: "MOONSHOT_AUTH_TOKEN", enabled: false },
};
function parseArgs(argv) {
const a = { brief: "all", arm: "all" };
for (let i = 0; i < argv.length; i++) {
const t = argv[i];
if (t === "--force" || t === "--dry-run") a[t.slice(2)] = true;
else if (t.startsWith("--")) a[t.slice(2)] = argv[++i];
}
return a;
}
const args = parseArgs(process.argv.slice(2));
const briefs = JSON.parse(readFileSync(join(ROOT, "briefs.json"), "utf8"));
const pickBriefs = args.brief === "all" ? briefs : briefs.filter((b) => b.id === args.brief);
const armIds = args.arm === "all" ? Object.keys(ARMS).filter((k) => ARMS[k].enabled) : args.arm.split(",");
// --- stream-json parsing: pull the signals we care about -------------------
function analyzeTranscript(lines, runDir) {
let cost = null, model = null, sessionInit = false, resultText = "", isError = false, earlyStop = false;
let usage = null, numTurns = null, stop = null;
const refReads = [], filesWritten = [];
for (const line of lines) {
let ev; try { ev = JSON.parse(line); } catch { continue; }
if (ev.type === "system" && ev.subtype === "init") { sessionInit = true; model = ev.model ?? model; }
if (ev.type === "result") {
cost = ev.total_cost_usd ?? cost; resultText = ev.result ?? resultText; isError = !!ev.is_error;
numTurns = ev.num_turns ?? numTurns; stop = ev.subtype ?? stop;
if (ev.usage && typeof ev.usage === "object" && "input_tokens" in ev.usage) usage = ev.usage;
else if (ev.modelUsage) {
usage = { input_tokens: 0, output_tokens: 0, cache_read_input_tokens: 0 };
for (const m of Object.values(ev.modelUsage)) {
usage.input_tokens += m.inputTokens ?? 0; usage.output_tokens += m.outputTokens ?? 0;
usage.cache_read_input_tokens += m.cacheReadInputTokens ?? 0;
}
}
}
const record = (name, inp) => {
const fp = (inp && (inp.file_path || inp.path)) || "";
if (!fp) return;
if (name === "Read") refReads.push(fp.replace(SKILL_SRC + "/", "").replace(runDir + "/", ""));
if (name === "Write" || name === "Edit" || name === "MultiEdit") filesWritten.push(fp.replace(runDir + "/", ""));
};
// --verbose emits tool calls as full `assistant` messages (message.content[] tool_use blocks)
if (ev.type === "assistant" && Array.isArray(ev?.message?.content)) {
for (const b of ev.message.content) if (b?.type === "tool_use") record(b.name, b.input);
}
// --include-partial-messages emits them as stream_event/content_block_start instead
const cb = ev?.event?.content_block;
if (ev.type === "stream_event" && ev?.event?.type === "content_block_start" && cb?.type === "tool_use") {
record(cb.name, cb.input);
}
}
return { cost, model, sessionInit, resultText, isError, refReads, filesWritten, earlyStop, usage, numTurns, stop };
}
// Reference-read budget. Was 10, which predated derivation becoming the default:
// a derived run reads direction.md AND theme-axes.md where a catalog run read one
// theme file, so the floor moved up by one, and a measured clean derived build
// lands at 12 (ritual, genre, hero-discipline, rejection table, then the eight
// universal files). 14 leaves room for two legitimate conditionals and still
// catches the failure this check exists for, which is defensive pre-loading of a
// forty-file reference tree.
const REF_BUDGET = 14;
function loadOrderCheck(refReads) {
// discipline: bounded reference reads; slop-test.md not read before the build (it is Step 7)
const refs = refReads.filter((r) => r.startsWith("references/") || r.endsWith(".md"));
const slopIdx = refs.findIndex((r) => r.includes("slop-test.md"));
return { refCount: refs.length, refsWithinBudget: refs.length <= REF_BUDGET, slopTestLast: slopIdx === -1 || slopIdx >= refs.length - 3, refs };
}
// What MCP surface did the child actually have, and did it use any of it? Recorded
// on every run so a control arm can be proven clean rather than assumed clean.
function mcpAudit(lines) {
let advertised = [], called = [];
for (const l of lines) {
let d; try { d = JSON.parse(l); } catch { continue; }
if (d.type === "system" && Array.isArray(d.tools)) {
advertised = d.tools.filter((t) => String(t).startsWith("mcp__"));
}
const blocks = d?.message?.content;
if (Array.isArray(blocks)) {
for (const b of blocks) {
if (b?.type === "tool_use" && String(b.name).startsWith("mcp__")) called.push(b.name);
}
}
}
return { advertised, called: [...new Set(called)] };
}
function stampPresent(runDir) {
// Scan every emitted artifact, not a fixed filename list: builds name their
// stylesheet whatever the page wants (page.css, style.css, main.css), and a
// measured derived run put the stamp in page.css while the old list checked
// index.html / styles.css / tokens.css only. The marker is also not always
// flush against the comment open, since a derived header reads
// "/* <Brand> · Hallmark derived system", so match Hallmark anywhere in a
// leading block comment rather than immediately after the slash-star.
let names;
try { names = readdirSync(runDir); } catch { return false; }
for (const name of names) {
if (!/\.(css|html)$/i.test(name)) continue;
const src = readFileSync(join(runDir, name), "utf8");
if (/\/\*[^*]{0,120}Hallmark/.test(src)) return true;
}
return false;
}
async function runCell(brief, armId) {
const arm = ARMS[armId];
const effortTag = args.effort ? `-${args.effort}` : "";
const cell = `${armId}${effortTag}-cli`;
const runDir = join(RUNS, brief.id, cell);
const runJson = join(runDir, "run.json");
if (existsSync(runJson) && !args.force) { console.log(`skip ${brief.id}/${cell} (run.json exists)`); return; }
const work = join(SCRATCH, `${brief.id}-${armId}${effortTag}`);
if (args["dry-run"]) { console.log(`would run ${brief.id}/${cell} in ${work}`); return; }
rmSync(work, { recursive: true, force: true });
if (arm.skill === false) {
mkdirSync(work, { recursive: true });
} else {
mkdirSync(join(work, ".claude", "skills"), { recursive: true });
cpSync(SKILL_SRC, join(work, ".claude", "skills", "hallmark"), { recursive: true });
}
mkdirSync(runDir, { recursive: true });
const env = { ...process.env };
delete env.ANTHROPIC_API_KEY; // never spend the API key; use the subscription
delete env.ANTHROPIC_BASE_URL; // a host session (e.g. running inside Claude Code) exports this and breaks child auth
for (const k of Object.keys(env)) if (/^CLAUDE(CODE$|_CODE_|_)/.test(k)) delete env[k]; // shed host-session vars
if (arm.base) {
env.ANTHROPIC_BASE_URL = arm.base;
env.ANTHROPIC_MODEL = arm.model;
const tok = process.env[arm.tokenEnv];
if (!tok) { console.log(`skip ${brief.id}/${armId}-cli (missing ${arm.tokenEnv})`); return; }
env.ANTHROPIC_AUTH_TOKEN = tok;
}
const prompt = arm.skill === false
? `${brief.brief}\nWrite the result as index.html (plus css files if you like) in the current directory. Infer what you need; do not ask questions.`
: `/hallmark ${brief.brief}\nGo ahead and infer audience, use, and tone from the brief; do not ask me questions.`;
const allowed = ["Read", "Write", "Edit", "Bash"];
const cliArgs = ["-p", prompt, "--output-format", "stream-json", "--verbose",
"--permission-mode", "acceptEdits",
"--max-turns", "60", "--max-budget-usd", String(args["budget-usd"] || 5), "--setting-sources", "project"];
if (arm.mcp) {
// Pointed at the user's own local Inspo checkout. Read from ~/.claude.json so
// the arm follows wherever that server actually lives rather than hardcoding it.
const home = JSON.parse(readFileSync(join(process.env.HOME, ".claude.json"), "utf8"));
let inspo = null;
(function walk(o) {
if (!o || typeof o !== "object" || inspo) return;
if (o.mcpServers && o.mcpServers.inspo) { inspo = o.mcpServers.inspo; return; }
for (const v of Object.values(o)) walk(v);
})(home);
if (!inspo) { console.log(`skip ${brief.id}/${cell} (no inspo server in ~/.claude.json)`); return; }
const cfg = join(work, ".mcp-inspo.json");
writeFileSync(cfg, JSON.stringify({ mcpServers: { inspo } }, null, 2));
cliArgs.push("--mcp-config", cfg, "--strict-mcp-config");
allowed.push("mcp__inspo");
} else {
// No archive arm means no MCP at all. Without this the child inherits whatever
// servers live in the user's ~/.claude.json: a measured bare run was advertised
// eight Google Drive tools it never asked for, which is not a clean control even
// though it never called them. --strict-mcp-config with no --mcp-config is zero.
cliArgs.push("--strict-mcp-config");
}
cliArgs.push("--allowedTools", allowed.join(","));
if (args.model) cliArgs.push("--model", args.model);
if (args.effort) cliArgs.push("--effort", args.effort);
const started = Date.now();
console.log(`run ${brief.id}/${cell} ...`);
const lines = await new Promise((resolve) => {
const p = spawn("claude", cliArgs, { cwd: work, env, stdio: ["ignore", "pipe", "pipe"] });
const out = [];
let buf = "";
p.stdout.on("data", (d) => {
buf += d.toString();
let nl; while ((nl = buf.indexOf("\n")) >= 0) { const l = buf.slice(0, nl); buf += ""; buf = buf.slice(nl + 1); if (l.trim()) out.push(l); }
});
p.stderr.on("data", () => {});
p.on("close", () => { if (buf.trim()) out.push(buf); resolve(out); });
p.on("error", () => resolve(out));
});
writeFileSync(join(runDir, "transcript.jsonl"), lines.join("\n") + "\n");
// Archive every top-level artifact, not a fixed filename list. Builds name
// their stylesheet whatever the page wants (page.css cost one cell its
// skillLoaded score when the stamp lived there and only tokens.css was
// copied), and stampPresent scans this archive, so the archive must be whole.
for (const f of readdirSync(work)) {
if (/\.(html|css)$/i.test(f)) cpSync(join(work, f), join(runDir, f));
}
const a = analyzeTranscript(lines, work);
const lo = loadOrderCheck(a.refReads);
const run = {
brief: brief.id, arm: cell, skill: arm.skill !== false, effort: args.effort ?? null,
tier: "A", model: a.model ?? arm.model ?? "claude-subscription",
ok: a.sessionInit && !a.isError, durationSec: +((Date.now() - started) / 1000).toFixed(1),
costUSD: a.cost, turns: a.numTurns, stopSubtype: a.stop,
tokens: a.usage ? { in: a.usage.input_tokens ?? null, out: a.usage.output_tokens ?? null,
cacheRead: a.usage.cache_read_input_tokens ?? null } : null,
skillLoaded: stampPresent(runDir),
mcpAdvertised: mcpAudit(lines).advertised,
mcpCalled: mcpAudit(lines).called,
filesWritten: [...new Set(a.filesWritten)].slice(0, 20),
refReadCount: lo.refCount,
loadOrderOk: arm.skill === false ? null : (lo.refsWithinBudget && lo.slopTestLast),
refReads: lo.refs.slice(0, 20),
resultPreview: (a.resultText || "").slice(0, 200), isError: a.isError,
generatedAt: new Date().toISOString(),
};
writeFileSync(runJson, JSON.stringify(run, null, 2) + "\n");
console.log(`done ${brief.id}/${cell} | ok=${run.ok} skillLoaded=${run.skillLoaded} turns=${run.turns} cost=${run.costUSD} ${run.durationSec}s`);
}
// --- main ------------------------------------------------------------------
if (armIds.length === 0) { console.log("no enabled arms. Enable an arm in ARMS or pass --arm claude-cli."); process.exit(0); }
console.log(`Tier A: ${pickBriefs.length} brief(s) x ${armIds.length} arm(s): ${armIds.join(", ")}`);
if (args["dry-run"]) console.log("(dry run)");
for (const brief of pickBriefs) for (const armId of armIds) {
if (!ARMS[armId]) { console.log(`unknown arm ${armId}`); continue; }
await runCell(brief, armId);
}