feat(manifest): model-capability manifest — single source of truth for efforts, routes, labels

Rebased onto origin/main (5bf78671f4).

Conflict resolved in crates/buzz-agent/src/config.rs: kept our generated
anthropic_thinking_config_generated as the sole authority; absorbed main's
thinking shapes, ThinkingSummary enum + parse_thinking_summary, and the
thinking_summary field in Config wired through from_env / for_discovery.

Fixed CI job (model-capabilities): actions/checkout comment v6 → v6.0.3,
Swatinem/rust-cache SHA bumped to tagged v2.9.1 (c1937114) to resolve
zizmor ref-version-mismatch finding (r3730151005).

  62ebc9431 fix(model-capabilities): collapse empty segments in Rust gpt-version-segment matcher
  817b88790 test(file-size): fix fileOverrides integration test under GITHUB_ACTIONS
  1e9df5568 fix(manifest): enforce registry_label exclusion on family rules; test fileOverrides boundary
  0e6aacfd2 fix(buzz-agent): curate display name in configured_model_fallback
  77c24ef32 chore(manifest): remove dead registry_labels key

Co-authored-by: Will Pfleger <pfleger.will@gmail.com>
Signed-off-by: Will Pfleger <pfleger.will@gmail.com>
This commit is contained in:
Duncan
2026-08-10 11:10:45 -04:00
co-authored by Will Pfleger
parent 5e4c05f90b
commit 20609f1815
32 changed files with 10032 additions and 2702 deletions
+3 -2
View File
@@ -107,7 +107,7 @@ function readBaseFile(repoRoot, baseRef, filePath) {
}).toString("utf8");
}
export async function runFileSizeCheck({ projectRoot, rules, label }) {
export async function runFileSizeCheck({ projectRoot, rules, label, fileOverrides = {} }) {
// Every governed project is a direct child of the repository root. Derive
// these paths without Git so hook-provided repository environment variables
// cannot collapse the project pathspec to an empty string.
@@ -138,10 +138,11 @@ export async function runFileSizeCheck({ projectRoot, rules, label }) {
const baseContent =
change.status === "A" ? null : readBaseFile(repoRoot, baseRef, basePath);
const baseLines = baseContent == null ? null : countLines(baseContent);
const fileMaxLines = fileOverrides[relativePath] ?? rule.maxLines;
const result = evaluateFileSize({
baseLines,
candidateLines,
maxLines: rule.maxLines,
maxLines: fileMaxLines,
});
if (result.violates) {
+61 -1
View File
@@ -1,6 +1,6 @@
import assert from "node:assert/strict";
import { execFileSync } from "node:child_process";
import { mkdtempSync } from "node:fs";
import { mkdtempSync, writeFileSync, mkdirSync } from "node:fs";
import { tmpdir } from "node:os";
import path from "node:path";
import test from "node:test";
@@ -10,6 +10,7 @@ import {
evaluateFileSize,
parseChangedFiles,
resolveBaseRef,
runFileSizeCheck,
} from "./check-file-sizes-core.mjs";
function git(repo, ...args) {
@@ -107,3 +108,62 @@ test("an inherited oversized file may hold or shrink but not grow", () => {
true,
);
});
test("fileOverrides raises the ceiling for a named path without leaking to other files", async () => {
// Build a minimal temp git repo:
// repo/
// desktop/ ← projectRoot
// src/
// governed.ts ← 1 line over the 1000-line default ceiling
// exempted.ts ← same size, but named in fileOverrides (ceiling 1200)
// origin/main at the initial empty commit ← base for diff
const repoRoot = mkdtempSync(path.join(tmpdir(), "file-size-override-"));
const projectRoot = path.join(repoRoot, "desktop");
const srcDir = path.join(projectRoot, "src");
mkdirSync(srcDir, { recursive: true });
function gitRepo(...args) {
return execFileSync("git", args, { cwd: repoRoot, encoding: "utf8" });
}
gitRepo("init", "-b", "main");
gitRepo("config", "user.name", "Test");
gitRepo("config", "user.email", "test@example.com");
gitRepo("commit", "--allow-empty", "-m", "base");
gitRepo("remote", "add", "origin", repoRoot);
gitRepo("fetch", "origin", "main:refs/remotes/origin/main");
// Add a second commit so HEAD^1 resolves (the CI path uses HEAD^1 as base,
// and a parentless HEAD causes git cat-file -e HEAD^1^{commit} to throw).
gitRepo("commit", "--allow-empty", "-m", "branch commit");
// Write both files: 1001 lines each (1 over the 1000-line default ceiling)
const content = "x\n".repeat(1001);
writeFileSync(path.join(srcDir, "governed.ts"), content);
writeFileSync(path.join(srcDir, "exempted.ts"), content);
gitRepo("add", ".");
const prevExitCode = process.exitCode;
const errors = [];
const origConsoleError = console.error;
console.error = (...args) => errors.push(args.join(" "));
try {
await runFileSizeCheck({
projectRoot,
rules: [{ root: "src", extensions: new Set([".ts"]), maxLines: 1000 }],
label: "test",
fileOverrides: { "src/exempted.ts": 1200 },
});
// governed.ts (1001 lines, ceiling 1000) must violate
assert.equal(process.exitCode, 1, "governed.ts over default ceiling should fail");
const combined = errors.join("\n");
assert.ok(combined.includes("governed.ts"), "violation message should name governed.ts");
// exempted.ts (1001 lines, override ceiling 1200) must NOT be in the violation list
assert.ok(!combined.includes("exempted.ts"), "exempted.ts under override ceiling should not appear");
} finally {
process.exitCode = prevExitCode;
console.error = origConsoleError;
}
});
File diff suppressed because it is too large Load Diff
+897
View File
@@ -0,0 +1,897 @@
{
"_comment": "Hand-curated model capability manifest. Edit here; run scripts/generate-model-capabilities.mjs to regenerate artifacts.",
"_generated_by": "scripts/generate-model-capabilities.mjs",
"_sources": {
"models_dev": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0)",
"anthropic_thinking": "https://platform.claude.com/docs/en/build-with-claude/extended-thinking (July 2025)",
"anthropic_effort": "https://platform.claude.com/docs/en/build-with-claude/effort (July 2025)",
"openai_reasoning": "https://platform.openai.com/docs/guides/reasoning (July 2025)",
"goose_known_models": "goose revision 6789d4af (crates/goose-providers/src/databricks_v2.rs:41-42) \u2014 two IDs: databricks-gpt-5-5, databricks-claude-opus-4-7"
},
"family_tokens": [
"claude-",
"gpt-"
],
"family_rules": [
{
"id": "anthropic-manual-budget-claude3",
"match_kind": "prefix",
"match_value": "claude-3",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "manual-budget",
"supported_efforts": [
"low",
"medium",
"high"
],
"default_effort": null,
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-manual-budget-opus-4-5",
"match_kind": "exact",
"match_value": "claude-opus-4-5",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "manual-budget",
"supported_efforts": [
"low",
"medium",
"high"
],
"default_effort": null,
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-opus-4-7",
"match_kind": "prefix",
"match_value": "claude-opus-4-7",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-opus-4-8",
"match_kind": "prefix",
"match_value": "claude-opus-4-8",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-opus-5",
"match_kind": "prefix",
"match_value": "claude-opus-5",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-sonnet-5",
"match_kind": "prefix",
"match_value": "claude-sonnet-5",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-fable-5",
"match_kind": "prefix",
"match_value": "claude-fable-5",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-xhigh-mythos-5",
"match_kind": "prefix",
"match_value": "claude-mythos-5",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-no-xhigh-opus-4-6",
"match_kind": "prefix",
"match_value": "claude-opus-4-6",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-no-xhigh-sonnet-4-6",
"match_kind": "prefix",
"match_value": "claude-sonnet-4-6",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "anthropic-adaptive-no-xhigh-mythos-preview",
"match_kind": "prefix",
"match_value": "claude-mythos-preview",
"providers": [
"anthropic",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "openai-gpt5-pro",
"match_kind": "gpt5-token",
"match_value": "gpt-5-pro",
"match_aliases": [
"gpt5-pro"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 20,
"thinking_mode": "none",
"supported_efforts": [
"high"
],
"default_effort": "high",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "openai-gpt5-6",
"match_kind": "gpt5-token",
"match_value": "gpt-5.6",
"match_aliases": [
"gpt5.6",
"gpt-5-6",
"gpt5-6"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 15,
"thinking_mode": "none",
"supported_efforts": [
"none",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "openai-gpt5-5",
"match_kind": "gpt5-token",
"match_value": "gpt-5.5",
"match_aliases": [
"gpt5.5",
"gpt-5-5",
"gpt5-5"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 15,
"thinking_mode": "none",
"supported_efforts": [
"none",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "openai-gpt5-4",
"match_kind": "gpt5-token",
"match_value": "gpt-5.4",
"match_aliases": [
"gpt5.4",
"gpt-5-4",
"gpt5-4"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 15,
"thinking_mode": "none",
"supported_efforts": [
"none",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "openai-gpt5-1",
"match_kind": "gpt5-token",
"match_value": "gpt-5.1",
"match_aliases": [
"gpt5.1",
"gpt-5-1",
"gpt5-1"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 15,
"thinking_mode": "none",
"supported_efforts": [
"none",
"low",
"medium",
"high"
],
"default_effort": "none",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "openai-gpt5-base",
"match_kind": "gpt5-base",
"match_value": "gpt-5",
"match_aliases": [
"gpt5"
],
"providers": [
"openai",
"databricks",
"databricks_v2"
],
"match_priority": 10,
"thinking_mode": "none",
"supported_efforts": [
"minimal",
"low",
"medium",
"high"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-standard"
},
{
"id": "dbv2-claude-code-names-segment",
"_comment": "DBv2-only rule: endpoint names containing a Claude code-name segment (opus, sonnet, haiku, mythos, fable, claude) route via Anthropic Messages. This matches goose-opus-5 (segments: goose,opus,5) etc. Effort classification uses conservative defaults because prefix-stripped alias ('opus-5') is not a recognized Claude family.",
"match_kind": "segment",
"match_value": "claude",
"match_aliases": [
"opus",
"sonnet",
"haiku",
"mythos",
"fable"
],
"providers": [
"databricks_v2"
],
"match_priority": 5,
"thinking_mode": "omit-fields",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none"
},
{
"id": "dbv2-gpt-code-names-segment",
"_comment": "DBv2-only rule: models whose normalized alias has 'gpt' as a segment followed by a numeric segment (e.g. databricks-gpt-5.5 \u2192 'gpt' seg + '5' seg) route via OpenAI Responses. 'gpt5' dashless alias catches gpt5-custom forms. match_kind=gpt-version-segment: token must be an exact segment AND the next segment must start with a digit \u2014 prevents gptoss/gptj (no 'gpt' segment) AND gpt-neox-20b ('gpt' segment but next segment 'neox' is not numeric). Priority 6 > dbv2-claude's 5.",
"match_kind": "gpt-version-segment",
"match_value": "gpt",
"providers": [
"databricks_v2"
],
"match_priority": 6,
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-clamp-max-to-xhigh",
"match_aliases": [
"gpt5"
]
},
{
"id": "dbv2-sol-luna-terra-segment",
"_comment": "DBv2-only rule: sol/luna/terra are OpenAI code names. Route via OpenAI Responses. Must use segment match to avoid matching substrings (consolidated-llama has 'sol' but not as a segment).",
"match_kind": "segment",
"match_value": "sol",
"match_aliases": [
"luna",
"terra"
],
"providers": [
"databricks_v2"
],
"match_priority": 5,
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-clamp-max-to-xhigh"
}
],
"_comment_databricks_v2_known_models": "Authoritative list of Databricks v2 known model IDs. Mirrors goose DATABRICKS_V2_KNOWN_MODELS at revision 6789d4af (crates/goose-providers/src/databricks_v2.rs:41-42). Generated into DATABRICKS_V2_KNOWN_MODELS in both Rust and TS. Uniqueness enforced by the generator. Opt-in drift check: node scripts/generate-model-capabilities.mjs --check-goose",
"databricks_v2_known_models": [
"databricks-gpt-5-5",
"databricks-claude-opus-4-7"
],
"exact_records": [
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-4-mini",
"registry_label": "GPT-5.4 mini",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high (family rule adds none+xhigh \u2014 adopt provider-advertised)",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises low|medium|high. Family rule (gpt5-4) adds none+xhigh. Provider-advertised wins per plan F1 policy.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-4-mini\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-4-nano",
"registry_label": "GPT-5.4 nano",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises low|medium|high. Same as gpt-5-4-mini. Adopt.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-4-nano\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-sol",
"registry_label": "GPT-5.6 Sol",
"supported_efforts_override": [
"low",
"medium",
"high",
"max"
],
"source": "models.dev reasoning_options: low|medium|high|max (family rule adds none+xhigh \u2014 provider-advertised wins per plan F1)",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises [low, medium, high, max]. Family rule (gpt5-6) has none+xhigh+max; sol endpoint does not expose none or xhigh. Provider-advertised wins.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-6-sol\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\",\"max\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-5",
"registry_label": "GPT-5.5",
"source": "models.dev reasoning_options: low|medium|high (family rule adds none+xhigh \u2014 provider-advertised wins per plan F1)",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev (pinned payload) advertises [low, medium, high]. Family rule (gpt5-5) has none+xhigh; this Databricks endpoint does not expose none or xhigh. Provider-advertised wins.",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-5\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-opus-4-7",
"registry_label": "Claude Opus 4.7",
"source": "DATABRICKS_V2_KNOWN_MODELS; family rule anthropic-adaptive-xhigh-opus-4-7 applies",
"_reconciliation": "no-effort-divergence",
"_reconciliation_note": "models.dev advertises reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]. This is a different capability axis (extended thinking token budget), not an effort-level selector. No effort divergence to reconcile \u2014 efforts for this model come from the anthropic family rule (anthropic-adaptive-xhigh-opus-4-7).",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-claude-opus-4-7\"].reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-luna",
"registry_label": "GPT-5.6 Luna",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; luna endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-luna\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-terra",
"registry_label": "GPT-5.6 Terra",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; terra endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-terra\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-haiku-4-5",
"registry_label": "Claude Haiku 4.5 (latest)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-opus-4-1",
"registry_label": "Claude Opus 4.1 (latest)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-opus-4-5",
"registry_label": "Claude Opus 4.5 (latest)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-opus-4-6",
"registry_label": "Claude Opus 4.6",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-sonnet-4",
"registry_label": "Claude Sonnet 4.5",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-sonnet-4-5",
"registry_label": "Claude Sonnet 4.5 (latest)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-claude-sonnet-4-6",
"registry_label": "Claude Sonnet 4.6",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-2-5-flash",
"registry_label": "Gemini 2.5 Flash",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-2-5-pro",
"registry_label": "Gemini 2.5 Pro",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-3-1-flash-lite",
"registry_label": "Gemini 3.1 Flash Lite Preview",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-3-1-pro",
"registry_label": "Gemini 3.1 Pro Preview Custom Tools",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-3-flash",
"registry_label": "Gemini 3 Flash Preview",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-3-pro",
"registry_label": "Gemini 3 Pro Preview",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-glm-5-2",
"registry_label": "GLM-5.2",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5",
"registry_label": "GPT-5",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-1",
"registry_label": "GPT-5.1",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-2",
"registry_label": "GPT-5.2",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-4",
"registry_label": "GPT-5.4",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-mini",
"registry_label": "GPT-5 Mini",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-nano",
"registry_label": "GPT-5 Nano",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-oss-120b",
"registry_label": "GPT OSS 120B",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-oss-20b",
"registry_label": "GPT OSS 20B",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-kimi-k2-7-code",
"registry_label": "Kimi K2.7 Code",
"_source": "registry_labels"
}
],
"provider_fallbacks": {
"anthropic": {
"blank": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"normalization_policy": "none"
},
"concrete_unknown": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "omit-fields",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"normalization_policy": "none"
}
},
"openai": {
"blank": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
},
"concrete_unknown": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
}
},
"databricks_v2": {
"blank": {
"databricks_v2_wire_route": "route-unknown",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
},
"concrete_unknown": {
"databricks_v2_wire_route": "mlflow-chat",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
}
},
"databricks": {
"blank": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
},
"concrete_unknown": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"normalization_policy": "openai-clamp-max-to-xhigh"
}
},
"openrouter": {
"blank": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"normalization_policy": "none"
},
"concrete_unknown": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"normalization_policy": "none"
}
},
"_default": {
"blank": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"normalization_policy": "none"
},
"concrete_unknown": {
"databricks_v2_wire_route": "not-applicable",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "medium",
"normalization_policy": "none"
}
}
}
}
File diff suppressed because it is too large Load Diff
+114
View File
@@ -0,0 +1,114 @@
#!/usr/bin/env node
/**
* Normative corpus runner — validates the generated TS interpreter against
* scripts/normative-corpus.json.
*
* This is the JS side of the two-interpreter corpus check. The runner imports
* resolveModelCapabilities() directly from the generated TypeScript module via
* Node's --experimental-strip-types flag. The Rust side lives in
* crates/buzz-agent/src/generated_model_capabilities.rs (shared corpus harness).
*
* Usage: node --experimental-strip-types scripts/run-corpus.mjs [--verbose]
* Exits 0 on all pass, 1 on any failure.
*/
import { readFileSync } from "node:fs";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
const __dirname = dirname(fileURLToPath(import.meta.url));
const repoRoot = join(__dirname, "..");
const VERBOSE = process.argv.includes("--verbose");
// Import the generated TypeScript interpreter directly.
// Node 22+ --experimental-strip-types strips type annotations at load time; no build step needed.
const { resolveModelCapabilities } = await import(
join(repoRoot, "desktop", "src", "features", "agents", "ui", "modelCapabilities.ts")
);
const corpus = JSON.parse(
readFileSync(join(repoRoot, "scripts", "normative-corpus.json"), "utf8"),
);
// ----- Provider alias canonicalization -----
// Mirrors production canonicalizeProvider() in desktop/src/features/agents/lib/formatAgentModelLabel.ts.
// Applied before every generated lookup so alias vectors (e.g. "openai-compat") pass both interpreters.
const PROVIDER_ALIASES = new Map([
["databricks-v2", "databricks_v2"],
["openai-compat", "openai"],
]);
function canonicalizeProvider(provider) {
const normalized = (provider ?? "").trim().toLowerCase();
return PROVIDER_ALIASES.get(normalized) ?? normalized;
}
// ----- Run corpus -----
let passed = 0;
let failed = 0;
const REQUIRED_AXES = [
"thinking_mode",
"supported_efforts",
"default_effort",
"databricks_v2_wire_route",
"normalization_policy",
"registry_label",
];
for (const entry of corpus) {
// Skip group header entries
if (entry._group) continue;
if (!entry.expect) continue;
// Require all 6 axes on every executable vector — sparse vectors hide divergences.
const missingAxes = REQUIRED_AXES.filter((ax) => !(ax in entry.expect));
if (missingAxes.length > 0) {
failed++;
console.error(` FAIL ${entry.id}: sparse vector — missing axes: ${missingAxes.join(", ")}`);
console.error(` All six axes are required: ${REQUIRED_AXES.join(", ")}`);
continue;
}
// resolveModelCapabilities returns camelCase keys (registryLabel, thinkingMode, etc.)
const result = resolveModelCapabilities(canonicalizeProvider(entry.provider), entry.raw_model_id);
const expect = entry.expect;
const failures = [];
for (const [key, expectedVal] of Object.entries(expect)) {
// Corpus uses snake_case; generated TS uses camelCase — convert for lookup.
const camelKey = key.replace(/_([a-z])/g, (_, c) => c.toUpperCase());
const actualVal = camelKey in result ? result[camelKey] : result[key];
if (Array.isArray(expectedVal)) {
// Order-sensitive comparison for effort arrays
const actualArr = Array.isArray(actualVal) ? actualVal : [];
if (JSON.stringify(actualArr) !== JSON.stringify(expectedVal)) {
failures.push(
` ${key}: expected [${expectedVal.join(", ")}] got [${actualArr.join(", ")}]`,
);
}
} else {
if (actualVal !== expectedVal) {
failures.push(` ${key}: expected ${JSON.stringify(expectedVal)} got ${JSON.stringify(actualVal)}`);
}
}
}
if (failures.length === 0) {
passed++;
if (VERBOSE) {
console.log(` PASS ${entry.id}`);
}
} else {
failed++;
console.error(` FAIL ${entry.id} (${entry.provider} / "${entry.raw_model_id}")`);
for (const f of failures) console.error(f);
if (entry._note) console.error(` note: ${entry._note}`);
}
}
console.log(`\nCorpus: ${passed} passed, ${failed} failed`);
process.exit(failed > 0 ? 1 : 0);
+669
View File
@@ -0,0 +1,669 @@
#!/usr/bin/env node
/**
* Schema-negative validator tests for the manifest generator.
*
* Every validator rule in generate-model-capabilities.mjs must have a
* failing-input test here — if validation is missing, these tests would
* not catch the defect.
*
* Uses Node.js built-in test runner (node --test).
*/
import { test } from "node:test";
import assert from "node:assert/strict";
import { readFileSync, writeFileSync, unlinkSync, mkdirSync, mkdtempSync } from "node:fs";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
import { tmpdir } from "node:os";
import { execFileSync, spawnSync } from "node:child_process";
const __dirname = dirname(fileURLToPath(import.meta.url));
const repoRoot = join(__dirname, "..");
const manifestPath = join(repoRoot, "scripts", "model-capabilities.json");
const generatorPath = join(repoRoot, "scripts", "generate-model-capabilities.mjs");
/** Load the real manifest so we can mutate copies. */
const BASE_MANIFEST = JSON.parse(readFileSync(manifestPath, "utf8"));
/**
* Run the generator with a mutated manifest, returning { exitCode, stderr, stdout }.
* Writes the mutated manifest to a temp file and overrides the manifest path via env.
*/
function runGeneratorWithManifest(manifestOverride) {
// Write mutated manifest to a temp path
const tmpDir = mkdtempSync(join(tmpdir(), "test-validator-"));
const tmpManifest = join(tmpDir, "model-capabilities.json");
const tmpOutputDir = join(tmpDir, "out");
mkdirSync(tmpOutputDir, { recursive: true });
writeFileSync(tmpManifest, JSON.stringify(manifestOverride));
// Run the generator via node, pointing MANIFEST_PATH env at the temp file
// The generator reads from process.env.MANIFEST_PATH if set (we add this support)
const result = spawnSync(
process.execPath,
[generatorPath, "--manifest-path", tmpManifest, "--output-dir", tmpOutputDir],
{
encoding: "utf8",
env: { ...process.env },
},
);
// Cleanup
try { unlinkSync(tmpManifest); } catch {}
return { exitCode: result.status ?? 1, stderr: result.stderr, stdout: result.stdout };
}
/**
* Assert that the generator REJECTS the given manifest (exits non-zero).
* The optional `expectedMessage` is checked in stderr if provided.
*/
function assertRejects(label, manifest, expectedMessage) {
const { exitCode, stderr, stdout } = runGeneratorWithManifest(manifest);
assert.notEqual(exitCode, 0, `${label}: expected generator to fail but it succeeded.\nstdout: ${stdout}\nstderr: ${stderr}`);
if (expectedMessage) {
const combined = stderr + stdout;
assert.ok(
combined.includes(expectedMessage),
`${label}: expected error message "${expectedMessage}" not found.\nstdout: ${stdout}\nstderr: ${stderr}`,
);
}
}
/** Deep clone the base manifest and apply a mutator function. */
function mutate(fn) {
const clone = JSON.parse(JSON.stringify(BASE_MANIFEST));
fn(clone);
return clone;
}
// ---------------------------------------------------------------------------
// Rule: invalid enum value in family_rule.thinking_mode
// ---------------------------------------------------------------------------
test("schema-negative: invalid thinking_mode in family rule is rejected", () => {
assertRejects(
"invalid thinking_mode",
mutate((m) => {
m.family_rules[0].thinking_mode = "invalid-mode";
}),
"thinking_mode",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid enum value in family_rule.databricks_v2_wire_route
// ---------------------------------------------------------------------------
test("schema-negative: invalid databricks_v2_wire_route in family rule is rejected", () => {
assertRejects(
"invalid databricks_v2_wire_route",
mutate((m) => {
m.family_rules[0].databricks_v2_wire_route = "chat-completions";
}),
"databricks_v2_wire_route",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid enum value in family_rule.supported_efforts[]
// ---------------------------------------------------------------------------
test("schema-negative: invalid effort value in family rule supported_efforts is rejected", () => {
assertRejects(
"invalid supported_efforts value",
mutate((m) => {
m.family_rules[0].supported_efforts = ["low", "ultra-high"];
}),
"supported_efforts",
);
});
// ---------------------------------------------------------------------------
// Rule: empty supported_efforts array in family rule
// ---------------------------------------------------------------------------
test("schema-negative: empty supported_efforts in family rule is rejected", () => {
assertRejects(
"empty supported_efforts",
mutate((m) => {
m.family_rules[0].supported_efforts = [];
}),
"supported_efforts",
);
});
// ---------------------------------------------------------------------------
// Rule: default_effort not in supported_efforts (non-null)
// ---------------------------------------------------------------------------
test("schema-negative: default_effort not in supported_efforts is rejected", () => {
assertRejects(
"default_effort not in supported_efforts",
mutate((m) => {
m.family_rules[0].supported_efforts = ["low", "medium"];
m.family_rules[0].default_effort = "high"; // not in list
}),
"default_effort",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid normalization_policy in family rule
// ---------------------------------------------------------------------------
test("schema-negative: invalid normalization_policy in family rule is rejected", () => {
assertRejects(
"invalid normalization_policy",
mutate((m) => {
m.family_rules[0].normalization_policy = "pass-through-all";
}),
"normalization_policy",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid match_kind in family rule
// ---------------------------------------------------------------------------
test("schema-negative: invalid match_kind in family rule is rejected", () => {
assertRejects(
"invalid match_kind",
mutate((m) => {
m.family_rules[0].match_kind = "regex";
}),
"match_kind",
);
});
// ---------------------------------------------------------------------------
// Rule: duplicate family rule id
// ---------------------------------------------------------------------------
test("schema-negative: duplicate family rule id is rejected", () => {
assertRejects(
"duplicate family rule id",
mutate((m) => {
m.family_rules.push({ ...m.family_rules[0] }); // duplicate id
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: registry_label on a family rule is rejected (display rule (a))
// ---------------------------------------------------------------------------
test("schema-negative: registry_label on a family rule is rejected", () => {
assertRejects(
"family rule registry_label forbidden",
mutate((m) => {
m.family_rules[0].registry_label = "Family Masquerade";
}),
"registry_label is not allowed on family rules",
);
});
// ---------------------------------------------------------------------------
// Rule: duplicate exact_record (provider, raw_model_id) key
// ---------------------------------------------------------------------------
test("schema-negative: duplicate exact_record key is rejected", () => {
assertRejects(
"duplicate exact_record key",
mutate((m) => {
m.exact_records.push({ ...m.exact_records[0] }); // duplicate
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record missing provider
// ---------------------------------------------------------------------------
test("schema-negative: exact_record missing provider is rejected", () => {
assertRejects(
"exact_record missing provider",
mutate((m) => {
m.exact_records.push({ raw_model_id: "some-model" });
}),
"provider",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record missing raw_model_id
// ---------------------------------------------------------------------------
test("schema-negative: exact_record missing raw_model_id is rejected", () => {
assertRejects(
"exact_record missing raw_model_id",
mutate((m) => {
m.exact_records.push({ provider: "databricks_v2" });
}),
"raw_model_id",
);
});
// ---------------------------------------------------------------------------
// Rule: provider fallback record missing blank state
// ---------------------------------------------------------------------------
test("schema-negative: provider fallback missing blank state is rejected", () => {
assertRejects(
"provider fallback missing blank",
mutate((m) => {
delete m.provider_fallbacks.anthropic.blank;
}),
"blank",
);
});
// ---------------------------------------------------------------------------
// Rule: provider fallback record missing concrete_unknown state
// ---------------------------------------------------------------------------
test("schema-negative: provider fallback missing concrete_unknown state is rejected", () => {
assertRejects(
"provider fallback missing concrete_unknown",
mutate((m) => {
delete m.provider_fallbacks.anthropic.concrete_unknown;
}),
"concrete_unknown",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid thinking_mode in provider fallback
// ---------------------------------------------------------------------------
test("schema-negative: invalid thinking_mode in provider fallback is rejected", () => {
assertRejects(
"invalid thinking_mode in fallback",
mutate((m) => {
m.provider_fallbacks.anthropic.blank.thinking_mode = "always-on";
}),
"thinking_mode",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid databricks_v2_wire_route in provider fallback
// ---------------------------------------------------------------------------
test("schema-negative: invalid wire_route in provider fallback is rejected", () => {
assertRejects(
"invalid wire_route in fallback",
mutate((m) => {
m.provider_fallbacks.anthropic.blank.databricks_v2_wire_route = "http-sse";
}),
"databricks_v2_wire_route",
);
});
// ---------------------------------------------------------------------------
// Rule: invalid default_effort in provider fallback (not in supported_efforts)
// ---------------------------------------------------------------------------
test("schema-negative: default_effort not in supported_efforts in fallback is rejected", () => {
assertRejects(
"default_effort not in supported_efforts in fallback",
mutate((m) => {
m.provider_fallbacks.openai.blank.supported_efforts = ["low", "medium"];
m.provider_fallbacks.openai.blank.default_effort = "high"; // not in list
}),
"default_effort",
);
});
// ---------------------------------------------------------------------------
// Rule: family rule missing id
// ---------------------------------------------------------------------------
test("schema-negative: family rule missing id is rejected", () => {
assertRejects(
"family rule missing id",
mutate((m) => {
m.family_rules.push({
match_kind: "prefix",
match_value: "test-",
providers: ["anthropic"],
match_priority: 1,
thinking_mode: "none",
supported_efforts: ["low"],
default_effort: null,
databricks_v2_wire_route: "not-applicable",
normalization_policy: "none",
// id deliberately omitted
});
}),
"id",
);
});
// ---------------------------------------------------------------------------
// Rule: duplicate exact_record key (same provider + raw_model_id) is rejected
// ---------------------------------------------------------------------------
test("schema-negative: duplicate exact_record key is rejected", () => {
assertRejects(
"duplicate exact_record key",
mutate((m) => {
// Add a second record for the same (provider, raw_model_id) key
const existing = m.exact_records.find(
(r) => r.raw_model_id === "databricks-gpt-5-4-mini",
);
m.exact_records.push({ ...existing });
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record registry_label with empty label is rejected
// ---------------------------------------------------------------------------
test("schema-negative: exact_record registry_label with empty string is rejected", () => {
assertRejects(
"exact_record registry_label empty string",
mutate((m) => {
const rec = m.exact_records.find(
(r) => r.raw_model_id === "databricks-gpt-5-4-mini",
);
rec.registry_label = "";
}),
"nonempty",
);
});
// ---------------------------------------------------------------------------
// Rule: duplicate databricks_v2_known_models IDs
// ---------------------------------------------------------------------------
test("schema-negative: duplicate databricks_v2_known_models ID is rejected", () => {
assertRejects(
"duplicate known model ID",
mutate((m) => {
m.databricks_v2_known_models = ["databricks-gpt-5-5", "databricks-gpt-5-5"];
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: unsafe characters in match_value (family rule)
// ---------------------------------------------------------------------------
test("schema-negative: family rule match_value with unsafe chars is rejected", () => {
assertRejects(
"family rule match_value with backslash",
mutate((m) => {
// Inject a backslash into an existing rule's match_value — would break Rust string literal
const rule = m.family_rules.find((r) => r.id === "anthropic-manual-budget-claude3");
rule.match_value = "claude-3\\evil";
}),
"unsafe",
);
});
// ---------------------------------------------------------------------------
// Rule: unsafe characters in known-model ID
// ---------------------------------------------------------------------------
test("schema-negative: databricks_v2_known_models ID with unsafe chars is rejected", () => {
assertRejects(
"known-model ID with double-quote",
mutate((m) => {
m.databricks_v2_known_models = ['databricks-gpt-5-5', 'bad"id'];
}),
"unsafe",
);
});
// ---------------------------------------------------------------------------
// Rule: unsafe characters in exact_record registry_label
// ---------------------------------------------------------------------------
test("schema-negative: exact_record registry_label with unsafe chars is rejected", () => {
assertRejects(
"exact_record registry_label with backslash",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.registry_label = "GPT-5.4 Mini\\injected";
}),
"unsafe",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override must be non-empty
// ---------------------------------------------------------------------------
test("schema-negative: exact_record empty supported_efforts_override is rejected", () => {
assertRejects(
"exact_record empty supported_efforts_override",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = [];
}),
"supported_efforts_override",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override with invalid enum value
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override with bogus enum is rejected", () => {
assertRejects(
"exact_record bogus effort enum",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["ultra-high"];
}),
"supported_efforts_override",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override with duplicate effort
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override with duplicate effort is rejected", () => {
assertRejects(
"exact_record duplicate effort",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["low", "low"];
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override must follow canonical order
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override out of canonical order is rejected", () => {
assertRejects(
"exact_record efforts out of order",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
// Reverse order — [high, medium, low] is not canonical [low, medium, high]
rec.supported_efforts_override = ["high", "medium", "low"];
}),
"canonical order",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record default_effort not in supported_efforts_override
// ---------------------------------------------------------------------------
test("schema-negative: exact_record default_effort not in supported_efforts_override is rejected", () => {
assertRejects(
"exact_record default_effort outside override",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["low", "medium"];
rec.default_effort = "high"; // not in override
}),
"default_effort",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record match_priority must be a non-negative integer
// ---------------------------------------------------------------------------
test("schema-negative: exact_record match_priority non-integer is rejected", () => {
assertRejects(
"exact_record match_priority non-integer",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.match_priority = "five";
}),
"match_priority",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record thinking_mode must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid thinking_mode is rejected", () => {
assertRejects(
"exact_record invalid thinking_mode",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.thinking_mode = "turbo-thinking";
}),
"thinking_mode",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record databricks_v2_wire_route must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid databricks_v2_wire_route is rejected", () => {
assertRejects(
"exact_record invalid wire_route",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.databricks_v2_wire_route = "http-sse";
}),
"databricks_v2_wire_route",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record normalization_policy must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid normalization_policy is rejected", () => {
assertRejects(
"exact_record invalid normalization_policy",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.normalization_policy = "pass-through-all";
}),
"normalization_policy",
);
});
// ---------------------------------------------------------------------------
// Rule: family_rule match_priority must be a non-negative integer
// ---------------------------------------------------------------------------
test("schema-negative: family_rule match_priority string (injection vector) is rejected", () => {
assertRejects(
"family_rule string match_priority",
mutate((m) => {
// Inject a string that contains a compile_error! macro — must be rejected before emission
m.family_rules[0].match_priority = 'compile_error!("THUFIR_INJECTED")';
}),
"match_priority",
);
});
test("schema-negative: family_rule match_priority negative integer is rejected", () => {
assertRejects(
"family_rule negative match_priority",
mutate((m) => {
m.family_rules[0].match_priority = -1;
}),
"match_priority",
);
});
test("schema-negative: family_rule match_priority float is rejected", () => {
assertRejects(
"family_rule float match_priority",
mutate((m) => {
m.family_rules[0].match_priority = 1.5;
}),
"match_priority",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record uppercase duplicate key is rejected (after lowercasing)
// ---------------------------------------------------------------------------
test("schema-negative: exact_record uppercase duplicate key is rejected", () => {
assertRejects(
"exact_record uppercase duplicate key",
mutate((m) => {
// Add an uppercase copy of an existing exact record key
const existing = m.exact_records[0];
m.exact_records.push({
...existing,
provider: existing.provider.toUpperCase(),
raw_model_id: existing.raw_model_id.toUpperCase(),
});
}),
"duplicate exact_record key",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record inherited default_effort not in inherited supported_efforts
// ---------------------------------------------------------------------------
test("schema-negative: exact_record inherited default_effort outside materialized supported_efforts is rejected", () => {
assertRejects(
"exact_record inherited default out of materialized efforts",
mutate((m) => {
// Override supported_efforts_override to a single value that excludes the family default.
// For any exact record that inherits family default_effort, override efforts to exclude it.
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
// Family default for the gpt5-4 rule is "medium". Override to only ["low"] to force mismatch.
rec.supported_efforts_override = ["low"];
// No explicit default_effort — inherits "medium" from family, but "medium" is not in ["low"]
}),
"materialized default_effort",
);
});
// ---------------------------------------------------------------------------
// Rule: family_rule supported_efforts must have no duplicates
// ---------------------------------------------------------------------------
test("schema-negative: family_rule supported_efforts with duplicate is rejected", () => {
assertRejects(
"family_rule duplicate effort",
mutate((m) => {
m.family_rules[0].supported_efforts = ["low", "low", "medium"];
}),
"duplicate effort",
);
});
// ---------------------------------------------------------------------------
// Rule: family_rule supported_efforts must follow canonical order
// ---------------------------------------------------------------------------
test("schema-negative: family_rule supported_efforts out of canonical order is rejected", () => {
assertRejects(
"family_rule efforts out of order",
mutate((m) => {
m.family_rules[0].supported_efforts = ["high", "low", "medium"];
}),
"canonical order",
);
});
// ---------------------------------------------------------------------------
// Rule: provider_fallback supported_efforts must have no duplicates
// ---------------------------------------------------------------------------
test("schema-negative: provider_fallback supported_efforts with duplicate is rejected", () => {
assertRejects(
"provider_fallback duplicate effort",
mutate((m) => {
const provider = Object.keys(m.provider_fallbacks)[0];
m.provider_fallbacks[provider].blank.supported_efforts = ["low", "low", "medium"];
}),
"duplicate effort",
);
});
// ---------------------------------------------------------------------------
// Rule: provider_fallback supported_efforts must follow canonical order
// ---------------------------------------------------------------------------
test("schema-negative: provider_fallback supported_efforts out of canonical order is rejected", () => {
assertRejects(
"provider_fallback efforts out of order",
mutate((m) => {
const provider = Object.keys(m.provider_fallbacks)[0];
m.provider_fallbacks[provider].blank.supported_efforts = ["high", "low", "medium"];
}),
"canonical order",
);
});
console.log("\nSchema-negative validator tests complete.");