mirror of
https://github.com/Nutlope/hallmark.git
synced 2026-08-14 12:35:33 +02:00
The runs archive copied a fixed filename list, index.html, tokens.css, styles.css, and stampPresent scans that archive. A swept build whose stamp lives in page.css scored skillLoaded=false with a correct stamp on disk, which is the same stale assumption already fixed in stampPresent and in the reference budget, one function lower. The archive now takes every top-level html and css the build produced. The three affected run.json files are rescored from the real artifacts: only the buggy cell changed (inspo false to true); the bare control stays false, which is the leak check holding. Also archives the second three-arm run: bare 56 FAIL with the same six numbered eyebrows, both Hallmark arms clean, and the control now provably offered zero MCP tools.
286 lines
15 KiB
JavaScript
286 lines
15 KiB
JavaScript
#!/usr/bin/env node
|
|
// Tier A eval harness: drive the REAL Claude Code binary against a brief with
|
|
// the Hallmark skill installed, and measure conformance on the real harness
|
|
// (load-order discipline, stamp, files written) - not just bare-API
|
|
// instruction-following like Tier B (gen-direct.py).
|
|
//
|
|
// node gen-cli.mjs --brief b1|all --arm claude|bare|glm|kimi|all [--force] [--dry-run]
|
|
// [--model sonnet|opus|...] [--effort low|medium|high|max]
|
|
// [--budget-usd 5]
|
|
//
|
|
// --budget-usd caps spend per cell (default 5). A measured derived build on
|
|
// Opus lands near $3 and stops mid Step-7 at the old cap of 3, which scores a
|
|
// complete artifact as an error_max_budget_usd failure and never exercises the
|
|
// gate sweep. Raise it rather than reading a truncated run as a skill result.
|
|
//
|
|
// A/B: the "bare" arm is the no-skill control - same binary, same auth, plain
|
|
// brief prompt, NO skill copied, and --setting-sources project on the call so
|
|
// the user-scope ~/.claude/skills install cannot leak in (applied to BOTH arms
|
|
// so the skill arm always runs the project-scope copy). stampPresent on a bare
|
|
// cell doubles as the leak detector (must be false).
|
|
//
|
|
// IMPORTANT (measured): skills do NOT auto-trigger in `claude -p` headless mode,
|
|
// so we invoke the skill BY NAME ("/hallmark <brief>"). This is a CONFORMANCE
|
|
// harness (does the skill FOLLOW its discipline), not an auto-trigger test.
|
|
//
|
|
// Auth: the `claude` arm sets NO key and uses your Claude subscription -
|
|
// run it in a terminal where `claude` is logged in. The open-model arms
|
|
// (glm via Z.ai, kimi via Moonshot) are Anthropic-shaped and DISABLED until you
|
|
// add ZAI_AUTH_TOKEN / MOONSHOT_AUTH_TOKEN for them; Together/OpenRouter cannot back
|
|
// `claude -p` (it needs an Anthropic-shaped endpoint).
|
|
//
|
|
// Output: eval/runs/<brief>/<arm>-cli/{index.html?, transcript.jsonl, run.json}
|
|
// reusing the Tier B run.json schema plus Tier-A fields (skillLoaded, filesWritten,
|
|
// refReads, loadOrderOk).
|
|
|
|
import { readFileSync, writeFileSync, mkdirSync, existsSync, cpSync, rmSync, readdirSync } from "node:fs";
|
|
import { spawn } from "node:child_process";
|
|
import { join, dirname } from "node:path";
|
|
import { fileURLToPath } from "node:url";
|
|
|
|
const ROOT = dirname(fileURLToPath(import.meta.url));
|
|
const REPO = dirname(ROOT);
|
|
const SKILL_SRC = join(REPO, "skills", "hallmark");
|
|
const RUNS = join(ROOT, "runs");
|
|
const SCRATCH = join(ROOT, "_cli-scratch");
|
|
|
|
// --- arms (Tier A only; Anthropic-shaped endpoints) ------------------------
|
|
const ARMS = {
|
|
"claude": { model: null, enabled: true, skill: true, note: "Hallmark arm; Claude subscription; set NO key, must be logged in." },
|
|
"bare": { model: null, enabled: true, skill: false, note: "No-skill control; same binary and auth." },
|
|
// Hallmark with a reference archive attached. Since fa12b99 signal 8 inverts the
|
|
// flow: the archive owns Steps 1-6 and Hallmark enters at Step 7 as a sweep, so
|
|
// this arm measures a different code path from "claude", not a better-resourced
|
|
// version of it. Local stdio server, no credentials in the config.
|
|
"inspo": { model: null, enabled: true, skill: true, mcp: true, note: "Hallmark + Inspo MCP; exercises the signal-8 sweep path." },
|
|
"glm": { model: "glm-5.2[1m]", base: "https://api.z.ai/api/anthropic", tokenEnv: "ZAI_AUTH_TOKEN", enabled: false },
|
|
"kimi": { model: "kimi-k3", base: "https://api.moonshot.ai/anthropic", tokenEnv: "MOONSHOT_AUTH_TOKEN", enabled: false },
|
|
};
|
|
|
|
function parseArgs(argv) {
|
|
const a = { brief: "all", arm: "all" };
|
|
for (let i = 0; i < argv.length; i++) {
|
|
const t = argv[i];
|
|
if (t === "--force" || t === "--dry-run") a[t.slice(2)] = true;
|
|
else if (t.startsWith("--")) a[t.slice(2)] = argv[++i];
|
|
}
|
|
return a;
|
|
}
|
|
const args = parseArgs(process.argv.slice(2));
|
|
|
|
const briefs = JSON.parse(readFileSync(join(ROOT, "briefs.json"), "utf8"));
|
|
const pickBriefs = args.brief === "all" ? briefs : briefs.filter((b) => b.id === args.brief);
|
|
const armIds = args.arm === "all" ? Object.keys(ARMS).filter((k) => ARMS[k].enabled) : args.arm.split(",");
|
|
|
|
// --- stream-json parsing: pull the signals we care about -------------------
|
|
function analyzeTranscript(lines, runDir) {
|
|
let cost = null, model = null, sessionInit = false, resultText = "", isError = false, earlyStop = false;
|
|
let usage = null, numTurns = null, stop = null;
|
|
const refReads = [], filesWritten = [];
|
|
for (const line of lines) {
|
|
let ev; try { ev = JSON.parse(line); } catch { continue; }
|
|
if (ev.type === "system" && ev.subtype === "init") { sessionInit = true; model = ev.model ?? model; }
|
|
if (ev.type === "result") {
|
|
cost = ev.total_cost_usd ?? cost; resultText = ev.result ?? resultText; isError = !!ev.is_error;
|
|
numTurns = ev.num_turns ?? numTurns; stop = ev.subtype ?? stop;
|
|
if (ev.usage && typeof ev.usage === "object" && "input_tokens" in ev.usage) usage = ev.usage;
|
|
else if (ev.modelUsage) {
|
|
usage = { input_tokens: 0, output_tokens: 0, cache_read_input_tokens: 0 };
|
|
for (const m of Object.values(ev.modelUsage)) {
|
|
usage.input_tokens += m.inputTokens ?? 0; usage.output_tokens += m.outputTokens ?? 0;
|
|
usage.cache_read_input_tokens += m.cacheReadInputTokens ?? 0;
|
|
}
|
|
}
|
|
}
|
|
const record = (name, inp) => {
|
|
const fp = (inp && (inp.file_path || inp.path)) || "";
|
|
if (!fp) return;
|
|
if (name === "Read") refReads.push(fp.replace(SKILL_SRC + "/", "").replace(runDir + "/", ""));
|
|
if (name === "Write" || name === "Edit" || name === "MultiEdit") filesWritten.push(fp.replace(runDir + "/", ""));
|
|
};
|
|
// --verbose emits tool calls as full `assistant` messages (message.content[] tool_use blocks)
|
|
if (ev.type === "assistant" && Array.isArray(ev?.message?.content)) {
|
|
for (const b of ev.message.content) if (b?.type === "tool_use") record(b.name, b.input);
|
|
}
|
|
// --include-partial-messages emits them as stream_event/content_block_start instead
|
|
const cb = ev?.event?.content_block;
|
|
if (ev.type === "stream_event" && ev?.event?.type === "content_block_start" && cb?.type === "tool_use") {
|
|
record(cb.name, cb.input);
|
|
}
|
|
}
|
|
return { cost, model, sessionInit, resultText, isError, refReads, filesWritten, earlyStop, usage, numTurns, stop };
|
|
}
|
|
|
|
// Reference-read budget. Was 10, which predated derivation becoming the default:
|
|
// a derived run reads direction.md AND theme-axes.md where a catalog run read one
|
|
// theme file, so the floor moved up by one, and a measured clean derived build
|
|
// lands at 12 (ritual, genre, hero-discipline, rejection table, then the eight
|
|
// universal files). 14 leaves room for two legitimate conditionals and still
|
|
// catches the failure this check exists for, which is defensive pre-loading of a
|
|
// forty-file reference tree.
|
|
const REF_BUDGET = 14;
|
|
|
|
function loadOrderCheck(refReads) {
|
|
// discipline: bounded reference reads; slop-test.md not read before the build (it is Step 7)
|
|
const refs = refReads.filter((r) => r.startsWith("references/") || r.endsWith(".md"));
|
|
const slopIdx = refs.findIndex((r) => r.includes("slop-test.md"));
|
|
return { refCount: refs.length, refsWithinBudget: refs.length <= REF_BUDGET, slopTestLast: slopIdx === -1 || slopIdx >= refs.length - 3, refs };
|
|
}
|
|
|
|
// What MCP surface did the child actually have, and did it use any of it? Recorded
|
|
// on every run so a control arm can be proven clean rather than assumed clean.
|
|
function mcpAudit(lines) {
|
|
let advertised = [], called = [];
|
|
for (const l of lines) {
|
|
let d; try { d = JSON.parse(l); } catch { continue; }
|
|
if (d.type === "system" && Array.isArray(d.tools)) {
|
|
advertised = d.tools.filter((t) => String(t).startsWith("mcp__"));
|
|
}
|
|
const blocks = d?.message?.content;
|
|
if (Array.isArray(blocks)) {
|
|
for (const b of blocks) {
|
|
if (b?.type === "tool_use" && String(b.name).startsWith("mcp__")) called.push(b.name);
|
|
}
|
|
}
|
|
}
|
|
return { advertised, called: [...new Set(called)] };
|
|
}
|
|
|
|
function stampPresent(runDir) {
|
|
// Scan every emitted artifact, not a fixed filename list: builds name their
|
|
// stylesheet whatever the page wants (page.css, style.css, main.css), and a
|
|
// measured derived run put the stamp in page.css while the old list checked
|
|
// index.html / styles.css / tokens.css only. The marker is also not always
|
|
// flush against the comment open, since a derived header reads
|
|
// "/* <Brand> · Hallmark derived system", so match Hallmark anywhere in a
|
|
// leading block comment rather than immediately after the slash-star.
|
|
let names;
|
|
try { names = readdirSync(runDir); } catch { return false; }
|
|
for (const name of names) {
|
|
if (!/\.(css|html)$/i.test(name)) continue;
|
|
const src = readFileSync(join(runDir, name), "utf8");
|
|
if (/\/\*[^*]{0,120}Hallmark/.test(src)) return true;
|
|
}
|
|
return false;
|
|
}
|
|
|
|
async function runCell(brief, armId) {
|
|
const arm = ARMS[armId];
|
|
const effortTag = args.effort ? `-${args.effort}` : "";
|
|
const cell = `${armId}${effortTag}-cli`;
|
|
const runDir = join(RUNS, brief.id, cell);
|
|
const runJson = join(runDir, "run.json");
|
|
if (existsSync(runJson) && !args.force) { console.log(`skip ${brief.id}/${cell} (run.json exists)`); return; }
|
|
|
|
const work = join(SCRATCH, `${brief.id}-${armId}${effortTag}`);
|
|
if (args["dry-run"]) { console.log(`would run ${brief.id}/${cell} in ${work}`); return; }
|
|
|
|
rmSync(work, { recursive: true, force: true });
|
|
if (arm.skill === false) {
|
|
mkdirSync(work, { recursive: true });
|
|
} else {
|
|
mkdirSync(join(work, ".claude", "skills"), { recursive: true });
|
|
cpSync(SKILL_SRC, join(work, ".claude", "skills", "hallmark"), { recursive: true });
|
|
}
|
|
mkdirSync(runDir, { recursive: true });
|
|
|
|
const env = { ...process.env };
|
|
delete env.ANTHROPIC_API_KEY; // never spend the API key; use the subscription
|
|
delete env.ANTHROPIC_BASE_URL; // a host session (e.g. running inside Claude Code) exports this and breaks child auth
|
|
for (const k of Object.keys(env)) if (/^CLAUDE(CODE$|_CODE_|_)/.test(k)) delete env[k]; // shed host-session vars
|
|
if (arm.base) {
|
|
env.ANTHROPIC_BASE_URL = arm.base;
|
|
env.ANTHROPIC_MODEL = arm.model;
|
|
const tok = process.env[arm.tokenEnv];
|
|
if (!tok) { console.log(`skip ${brief.id}/${armId}-cli (missing ${arm.tokenEnv})`); return; }
|
|
env.ANTHROPIC_AUTH_TOKEN = tok;
|
|
}
|
|
|
|
const prompt = arm.skill === false
|
|
? `${brief.brief}\nWrite the result as index.html (plus css files if you like) in the current directory. Infer what you need; do not ask questions.`
|
|
: `/hallmark ${brief.brief}\nGo ahead and infer audience, use, and tone from the brief; do not ask me questions.`;
|
|
const allowed = ["Read", "Write", "Edit", "Bash"];
|
|
const cliArgs = ["-p", prompt, "--output-format", "stream-json", "--verbose",
|
|
"--permission-mode", "acceptEdits",
|
|
"--max-turns", "60", "--max-budget-usd", String(args["budget-usd"] || 5), "--setting-sources", "project"];
|
|
if (arm.mcp) {
|
|
// Pointed at the user's own local Inspo checkout. Read from ~/.claude.json so
|
|
// the arm follows wherever that server actually lives rather than hardcoding it.
|
|
const home = JSON.parse(readFileSync(join(process.env.HOME, ".claude.json"), "utf8"));
|
|
let inspo = null;
|
|
(function walk(o) {
|
|
if (!o || typeof o !== "object" || inspo) return;
|
|
if (o.mcpServers && o.mcpServers.inspo) { inspo = o.mcpServers.inspo; return; }
|
|
for (const v of Object.values(o)) walk(v);
|
|
})(home);
|
|
if (!inspo) { console.log(`skip ${brief.id}/${cell} (no inspo server in ~/.claude.json)`); return; }
|
|
const cfg = join(work, ".mcp-inspo.json");
|
|
writeFileSync(cfg, JSON.stringify({ mcpServers: { inspo } }, null, 2));
|
|
cliArgs.push("--mcp-config", cfg, "--strict-mcp-config");
|
|
allowed.push("mcp__inspo");
|
|
} else {
|
|
// No archive arm means no MCP at all. Without this the child inherits whatever
|
|
// servers live in the user's ~/.claude.json: a measured bare run was advertised
|
|
// eight Google Drive tools it never asked for, which is not a clean control even
|
|
// though it never called them. --strict-mcp-config with no --mcp-config is zero.
|
|
cliArgs.push("--strict-mcp-config");
|
|
}
|
|
cliArgs.push("--allowedTools", allowed.join(","));
|
|
if (args.model) cliArgs.push("--model", args.model);
|
|
if (args.effort) cliArgs.push("--effort", args.effort);
|
|
|
|
const started = Date.now();
|
|
console.log(`run ${brief.id}/${cell} ...`);
|
|
const lines = await new Promise((resolve) => {
|
|
const p = spawn("claude", cliArgs, { cwd: work, env, stdio: ["ignore", "pipe", "pipe"] });
|
|
const out = [];
|
|
let buf = "";
|
|
p.stdout.on("data", (d) => {
|
|
buf += d.toString();
|
|
let nl; while ((nl = buf.indexOf("\n")) >= 0) { const l = buf.slice(0, nl); buf += ""; buf = buf.slice(nl + 1); if (l.trim()) out.push(l); }
|
|
});
|
|
p.stderr.on("data", () => {});
|
|
p.on("close", () => { if (buf.trim()) out.push(buf); resolve(out); });
|
|
p.on("error", () => resolve(out));
|
|
});
|
|
|
|
writeFileSync(join(runDir, "transcript.jsonl"), lines.join("\n") + "\n");
|
|
// Archive every top-level artifact, not a fixed filename list. Builds name
|
|
// their stylesheet whatever the page wants (page.css cost one cell its
|
|
// skillLoaded score when the stamp lived there and only tokens.css was
|
|
// copied), and stampPresent scans this archive, so the archive must be whole.
|
|
for (const f of readdirSync(work)) {
|
|
if (/\.(html|css)$/i.test(f)) cpSync(join(work, f), join(runDir, f));
|
|
}
|
|
const a = analyzeTranscript(lines, work);
|
|
const lo = loadOrderCheck(a.refReads);
|
|
const run = {
|
|
brief: brief.id, arm: cell, skill: arm.skill !== false, effort: args.effort ?? null,
|
|
tier: "A", model: a.model ?? arm.model ?? "claude-subscription",
|
|
ok: a.sessionInit && !a.isError, durationSec: +((Date.now() - started) / 1000).toFixed(1),
|
|
costUSD: a.cost, turns: a.numTurns, stopSubtype: a.stop,
|
|
tokens: a.usage ? { in: a.usage.input_tokens ?? null, out: a.usage.output_tokens ?? null,
|
|
cacheRead: a.usage.cache_read_input_tokens ?? null } : null,
|
|
skillLoaded: stampPresent(runDir),
|
|
mcpAdvertised: mcpAudit(lines).advertised,
|
|
mcpCalled: mcpAudit(lines).called,
|
|
filesWritten: [...new Set(a.filesWritten)].slice(0, 20),
|
|
refReadCount: lo.refCount,
|
|
loadOrderOk: arm.skill === false ? null : (lo.refsWithinBudget && lo.slopTestLast),
|
|
refReads: lo.refs.slice(0, 20),
|
|
resultPreview: (a.resultText || "").slice(0, 200), isError: a.isError,
|
|
generatedAt: new Date().toISOString(),
|
|
};
|
|
writeFileSync(runJson, JSON.stringify(run, null, 2) + "\n");
|
|
console.log(`done ${brief.id}/${cell} | ok=${run.ok} skillLoaded=${run.skillLoaded} turns=${run.turns} cost=${run.costUSD} ${run.durationSec}s`);
|
|
}
|
|
|
|
// --- main ------------------------------------------------------------------
|
|
if (armIds.length === 0) { console.log("no enabled arms. Enable an arm in ARMS or pass --arm claude-cli."); process.exit(0); }
|
|
console.log(`Tier A: ${pickBriefs.length} brief(s) x ${armIds.length} arm(s): ${armIds.join(", ")}`);
|
|
if (args["dry-run"]) console.log("(dry run)");
|
|
for (const brief of pickBriefs) for (const armId of armIds) {
|
|
if (!ARMS[armId]) { console.log(`unknown arm ${armId}`); continue; }
|
|
await runCell(brief, armId);
|
|
}
|