mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
feat(manifest): model-capability manifest — single source of truth for efforts, routes, labels
Rebased onto origin/main (5bf78671f4).
Conflict resolved in crates/buzz-agent/src/config.rs: kept our generated
anthropic_thinking_config_generated as the sole authority; absorbed main's
thinking shapes, ThinkingSummary enum + parse_thinking_summary, and the
thinking_summary field in Config wired through from_env / for_discovery.
Fixed CI job (model-capabilities): actions/checkout comment v6 → v6.0.3,
Swatinem/rust-cache SHA bumped to tagged v2.9.1 (c1937114) to resolve
zizmor ref-version-mismatch finding (r3730151005).
62ebc9431 fix(model-capabilities): collapse empty segments in Rust gpt-version-segment matcher
817b88790 test(file-size): fix fileOverrides integration test under GITHUB_ACTIONS
1e9df5568 fix(manifest): enforce registry_label exclusion on family rules; test fileOverrides boundary
0e6aacfd2 fix(buzz-agent): curate display name in configured_model_fallback
77c24ef32 chore(manifest): remove dead registry_labels key
Co-authored-by: Will Pfleger <pfleger.will@gmail.com>
Signed-off-by: Will Pfleger <pfleger.will@gmail.com>
This commit is contained in:
@@ -107,7 +107,7 @@ function readBaseFile(repoRoot, baseRef, filePath) {
|
||||
}).toString("utf8");
|
||||
}
|
||||
|
||||
export async function runFileSizeCheck({ projectRoot, rules, label }) {
|
||||
export async function runFileSizeCheck({ projectRoot, rules, label, fileOverrides = {} }) {
|
||||
// Every governed project is a direct child of the repository root. Derive
|
||||
// these paths without Git so hook-provided repository environment variables
|
||||
// cannot collapse the project pathspec to an empty string.
|
||||
@@ -138,10 +138,11 @@ export async function runFileSizeCheck({ projectRoot, rules, label }) {
|
||||
const baseContent =
|
||||
change.status === "A" ? null : readBaseFile(repoRoot, baseRef, basePath);
|
||||
const baseLines = baseContent == null ? null : countLines(baseContent);
|
||||
const fileMaxLines = fileOverrides[relativePath] ?? rule.maxLines;
|
||||
const result = evaluateFileSize({
|
||||
baseLines,
|
||||
candidateLines,
|
||||
maxLines: rule.maxLines,
|
||||
maxLines: fileMaxLines,
|
||||
});
|
||||
|
||||
if (result.violates) {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { execFileSync } from "node:child_process";
|
||||
import { mkdtempSync } from "node:fs";
|
||||
import { mkdtempSync, writeFileSync, mkdirSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
@@ -10,6 +10,7 @@ import {
|
||||
evaluateFileSize,
|
||||
parseChangedFiles,
|
||||
resolveBaseRef,
|
||||
runFileSizeCheck,
|
||||
} from "./check-file-sizes-core.mjs";
|
||||
|
||||
function git(repo, ...args) {
|
||||
@@ -107,3 +108,62 @@ test("an inherited oversized file may hold or shrink but not grow", () => {
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("fileOverrides raises the ceiling for a named path without leaking to other files", async () => {
|
||||
// Build a minimal temp git repo:
|
||||
// repo/
|
||||
// desktop/ ← projectRoot
|
||||
// src/
|
||||
// governed.ts ← 1 line over the 1000-line default ceiling
|
||||
// exempted.ts ← same size, but named in fileOverrides (ceiling 1200)
|
||||
// origin/main at the initial empty commit ← base for diff
|
||||
const repoRoot = mkdtempSync(path.join(tmpdir(), "file-size-override-"));
|
||||
const projectRoot = path.join(repoRoot, "desktop");
|
||||
const srcDir = path.join(projectRoot, "src");
|
||||
mkdirSync(srcDir, { recursive: true });
|
||||
|
||||
function gitRepo(...args) {
|
||||
return execFileSync("git", args, { cwd: repoRoot, encoding: "utf8" });
|
||||
}
|
||||
|
||||
gitRepo("init", "-b", "main");
|
||||
gitRepo("config", "user.name", "Test");
|
||||
gitRepo("config", "user.email", "test@example.com");
|
||||
gitRepo("commit", "--allow-empty", "-m", "base");
|
||||
gitRepo("remote", "add", "origin", repoRoot);
|
||||
gitRepo("fetch", "origin", "main:refs/remotes/origin/main");
|
||||
|
||||
// Add a second commit so HEAD^1 resolves (the CI path uses HEAD^1 as base,
|
||||
// and a parentless HEAD causes git cat-file -e HEAD^1^{commit} to throw).
|
||||
gitRepo("commit", "--allow-empty", "-m", "branch commit");
|
||||
|
||||
// Write both files: 1001 lines each (1 over the 1000-line default ceiling)
|
||||
const content = "x\n".repeat(1001);
|
||||
writeFileSync(path.join(srcDir, "governed.ts"), content);
|
||||
writeFileSync(path.join(srcDir, "exempted.ts"), content);
|
||||
gitRepo("add", ".");
|
||||
|
||||
const prevExitCode = process.exitCode;
|
||||
const errors = [];
|
||||
const origConsoleError = console.error;
|
||||
console.error = (...args) => errors.push(args.join(" "));
|
||||
|
||||
try {
|
||||
await runFileSizeCheck({
|
||||
projectRoot,
|
||||
rules: [{ root: "src", extensions: new Set([".ts"]), maxLines: 1000 }],
|
||||
label: "test",
|
||||
fileOverrides: { "src/exempted.ts": 1200 },
|
||||
});
|
||||
|
||||
// governed.ts (1001 lines, ceiling 1000) must violate
|
||||
assert.equal(process.exitCode, 1, "governed.ts over default ceiling should fail");
|
||||
const combined = errors.join("\n");
|
||||
assert.ok(combined.includes("governed.ts"), "violation message should name governed.ts");
|
||||
// exempted.ts (1001 lines, override ceiling 1200) must NOT be in the violation list
|
||||
assert.ok(!combined.includes("exempted.ts"), "exempted.ts under override ceiling should not appear");
|
||||
} finally {
|
||||
process.exitCode = prevExitCode;
|
||||
console.error = origConsoleError;
|
||||
}
|
||||
});
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,897 @@
|
||||
{
|
||||
"_comment": "Hand-curated model capability manifest. Edit here; run scripts/generate-model-capabilities.mjs to regenerate artifacts.",
|
||||
"_generated_by": "scripts/generate-model-capabilities.mjs",
|
||||
"_sources": {
|
||||
"models_dev": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0)",
|
||||
"anthropic_thinking": "https://platform.claude.com/docs/en/build-with-claude/extended-thinking (July 2025)",
|
||||
"anthropic_effort": "https://platform.claude.com/docs/en/build-with-claude/effort (July 2025)",
|
||||
"openai_reasoning": "https://platform.openai.com/docs/guides/reasoning (July 2025)",
|
||||
"goose_known_models": "goose revision 6789d4af (crates/goose-providers/src/databricks_v2.rs:41-42) \u2014 two IDs: databricks-gpt-5-5, databricks-claude-opus-4-7"
|
||||
},
|
||||
"family_tokens": [
|
||||
"claude-",
|
||||
"gpt-"
|
||||
],
|
||||
"family_rules": [
|
||||
{
|
||||
"id": "anthropic-manual-budget-claude3",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-3",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "manual-budget",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"default_effort": null,
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-manual-budget-opus-4-5",
|
||||
"match_kind": "exact",
|
||||
"match_value": "claude-opus-4-5",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "manual-budget",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"default_effort": null,
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-opus-4-7",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-opus-4-7",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-opus-4-8",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-opus-4-8",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-opus-5",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-opus-5",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-sonnet-5",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-sonnet-5",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-fable-5",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-fable-5",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-xhigh-mythos-5",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-mythos-5",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-no-xhigh-opus-4-6",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-opus-4-6",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-no-xhigh-sonnet-4-6",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-sonnet-4-6",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "anthropic-adaptive-no-xhigh-mythos-preview",
|
||||
"match_kind": "prefix",
|
||||
"match_value": "claude-mythos-preview",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-pro",
|
||||
"match_kind": "gpt5-token",
|
||||
"match_value": "gpt-5-pro",
|
||||
"match_aliases": [
|
||||
"gpt5-pro"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 20,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"high"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-6",
|
||||
"match_kind": "gpt5-token",
|
||||
"match_value": "gpt-5.6",
|
||||
"match_aliases": [
|
||||
"gpt5.6",
|
||||
"gpt-5-6",
|
||||
"gpt5-6"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 15,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-5",
|
||||
"match_kind": "gpt5-token",
|
||||
"match_value": "gpt-5.5",
|
||||
"match_aliases": [
|
||||
"gpt5.5",
|
||||
"gpt-5-5",
|
||||
"gpt5-5"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 15,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-4",
|
||||
"match_kind": "gpt5-token",
|
||||
"match_value": "gpt-5.4",
|
||||
"match_aliases": [
|
||||
"gpt5.4",
|
||||
"gpt-5-4",
|
||||
"gpt5-4"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 15,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-1",
|
||||
"match_kind": "gpt5-token",
|
||||
"match_value": "gpt-5.1",
|
||||
"match_aliases": [
|
||||
"gpt5.1",
|
||||
"gpt-5-1",
|
||||
"gpt5-1"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 15,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"default_effort": "none",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "openai-gpt5-base",
|
||||
"match_kind": "gpt5-base",
|
||||
"match_value": "gpt-5",
|
||||
"match_aliases": [
|
||||
"gpt5"
|
||||
],
|
||||
"providers": [
|
||||
"openai",
|
||||
"databricks",
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 10,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-standard"
|
||||
},
|
||||
{
|
||||
"id": "dbv2-claude-code-names-segment",
|
||||
"_comment": "DBv2-only rule: endpoint names containing a Claude code-name segment (opus, sonnet, haiku, mythos, fable, claude) route via Anthropic Messages. This matches goose-opus-5 (segments: goose,opus,5) etc. Effort classification uses conservative defaults because prefix-stripped alias ('opus-5') is not a recognized Claude family.",
|
||||
"match_kind": "segment",
|
||||
"match_value": "claude",
|
||||
"match_aliases": [
|
||||
"opus",
|
||||
"sonnet",
|
||||
"haiku",
|
||||
"mythos",
|
||||
"fable"
|
||||
],
|
||||
"providers": [
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 5,
|
||||
"thinking_mode": "omit-fields",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"databricks_v2_wire_route": "anthropic-messages",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
{
|
||||
"id": "dbv2-gpt-code-names-segment",
|
||||
"_comment": "DBv2-only rule: models whose normalized alias has 'gpt' as a segment followed by a numeric segment (e.g. databricks-gpt-5.5 \u2192 'gpt' seg + '5' seg) route via OpenAI Responses. 'gpt5' dashless alias catches gpt5-custom forms. match_kind=gpt-version-segment: token must be an exact segment AND the next segment must start with a digit \u2014 prevents gptoss/gptj (no 'gpt' segment) AND gpt-neox-20b ('gpt' segment but next segment 'neox' is not numeric). Priority 6 > dbv2-claude's 5.",
|
||||
"match_kind": "gpt-version-segment",
|
||||
"match_value": "gpt",
|
||||
"providers": [
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 6,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh",
|
||||
"match_aliases": [
|
||||
"gpt5"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "dbv2-sol-luna-terra-segment",
|
||||
"_comment": "DBv2-only rule: sol/luna/terra are OpenAI code names. Route via OpenAI Responses. Must use segment match to avoid matching substrings (consolidated-llama has 'sol' but not as a segment).",
|
||||
"match_kind": "segment",
|
||||
"match_value": "sol",
|
||||
"match_aliases": [
|
||||
"luna",
|
||||
"terra"
|
||||
],
|
||||
"providers": [
|
||||
"databricks_v2"
|
||||
],
|
||||
"match_priority": 5,
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"databricks_v2_wire_route": "openai-responses",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
}
|
||||
],
|
||||
"_comment_databricks_v2_known_models": "Authoritative list of Databricks v2 known model IDs. Mirrors goose DATABRICKS_V2_KNOWN_MODELS at revision 6789d4af (crates/goose-providers/src/databricks_v2.rs:41-42). Generated into DATABRICKS_V2_KNOWN_MODELS in both Rust and TS. Uniqueness enforced by the generator. Opt-in drift check: node scripts/generate-model-capabilities.mjs --check-goose",
|
||||
"databricks_v2_known_models": [
|
||||
"databricks-gpt-5-5",
|
||||
"databricks-claude-opus-4-7"
|
||||
],
|
||||
"exact_records": [
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-4-mini",
|
||||
"registry_label": "GPT-5.4 mini",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"source": "models.dev reasoning_options: low|medium|high (family rule adds none+xhigh \u2014 adopt provider-advertised)",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev advertises low|medium|high. Family rule (gpt5-4) adds none+xhigh. Provider-advertised wins per plan F1 policy.",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-4-mini\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-4-nano",
|
||||
"registry_label": "GPT-5.4 nano",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"source": "models.dev reasoning_options: low|medium|high",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev advertises low|medium|high. Same as gpt-5-4-mini. Adopt.",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-4-nano\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-6-sol",
|
||||
"registry_label": "GPT-5.6 Sol",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "models.dev reasoning_options: low|medium|high|max (family rule adds none+xhigh \u2014 provider-advertised wins per plan F1)",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev advertises [low, medium, high, max]. Family rule (gpt5-6) has none+xhigh+max; sol endpoint does not expose none or xhigh. Provider-advertised wins.",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-6-sol\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\",\"max\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-5",
|
||||
"registry_label": "GPT-5.5",
|
||||
"source": "models.dev reasoning_options: low|medium|high (family rule adds none+xhigh \u2014 provider-advertised wins per plan F1)",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev (pinned payload) advertises [low, medium, high]. Family rule (gpt5-5) has none+xhigh; this Databricks endpoint does not expose none or xhigh. Provider-advertised wins.",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-gpt-5-5\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-opus-4-7",
|
||||
"registry_label": "Claude Opus 4.7",
|
||||
"source": "DATABRICKS_V2_KNOWN_MODELS; family rule anthropic-adaptive-xhigh-opus-4-7 applies",
|
||||
"_reconciliation": "no-effort-divergence",
|
||||
"_reconciliation_note": "models.dev advertises reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]. This is a different capability axis (extended thinking token budget), not an effort-level selector. No effort divergence to reconcile \u2014 efforts for this model come from the anthropic family rule (anthropic-adaptive-xhigh-opus-4-7).",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-claude-opus-4-7\"].reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-6-luna",
|
||||
"registry_label": "GPT-5.6 Luna",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"source": "models.dev reasoning_options: low|medium|high",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; luna endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-luna\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-6-terra",
|
||||
"registry_label": "GPT-5.6 Terra",
|
||||
"supported_efforts_override": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"source": "models.dev reasoning_options: low|medium|high",
|
||||
"_reconciliation": "adopt",
|
||||
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; terra endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
|
||||
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-terra\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-haiku-4-5",
|
||||
"registry_label": "Claude Haiku 4.5 (latest)",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-opus-4-1",
|
||||
"registry_label": "Claude Opus 4.1 (latest)",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-opus-4-5",
|
||||
"registry_label": "Claude Opus 4.5 (latest)",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-opus-4-6",
|
||||
"registry_label": "Claude Opus 4.6",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-sonnet-4",
|
||||
"registry_label": "Claude Sonnet 4.5",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-sonnet-4-5",
|
||||
"registry_label": "Claude Sonnet 4.5 (latest)",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-claude-sonnet-4-6",
|
||||
"registry_label": "Claude Sonnet 4.6",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-2-5-flash",
|
||||
"registry_label": "Gemini 2.5 Flash",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-2-5-pro",
|
||||
"registry_label": "Gemini 2.5 Pro",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-3-1-flash-lite",
|
||||
"registry_label": "Gemini 3.1 Flash Lite Preview",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-3-1-pro",
|
||||
"registry_label": "Gemini 3.1 Pro Preview Custom Tools",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-3-flash",
|
||||
"registry_label": "Gemini 3 Flash Preview",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gemini-3-pro",
|
||||
"registry_label": "Gemini 3 Pro Preview",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-glm-5-2",
|
||||
"registry_label": "GLM-5.2",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5",
|
||||
"registry_label": "GPT-5",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-1",
|
||||
"registry_label": "GPT-5.1",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-2",
|
||||
"registry_label": "GPT-5.2",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-4",
|
||||
"registry_label": "GPT-5.4",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-mini",
|
||||
"registry_label": "GPT-5 Mini",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-5-nano",
|
||||
"registry_label": "GPT-5 Nano",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-oss-120b",
|
||||
"registry_label": "GPT OSS 120B",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-gpt-oss-20b",
|
||||
"registry_label": "GPT OSS 20B",
|
||||
"_source": "registry_labels"
|
||||
},
|
||||
{
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "databricks-kimi-k2-7-code",
|
||||
"registry_label": "Kimi K2.7 Code",
|
||||
"_source": "registry_labels"
|
||||
}
|
||||
],
|
||||
"provider_fallbacks": {
|
||||
"anthropic": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "adaptive",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "omit-fields",
|
||||
"supported_efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "high",
|
||||
"normalization_policy": "none"
|
||||
}
|
||||
},
|
||||
"openai": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
}
|
||||
},
|
||||
"databricks_v2": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "route-unknown",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "mlflow-chat",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
}
|
||||
},
|
||||
"databricks": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "openai-clamp-max-to-xhigh"
|
||||
}
|
||||
},
|
||||
"openrouter": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "none"
|
||||
}
|
||||
},
|
||||
"_default": {
|
||||
"blank": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "none"
|
||||
},
|
||||
"concrete_unknown": {
|
||||
"databricks_v2_wire_route": "not-applicable",
|
||||
"thinking_mode": "none",
|
||||
"supported_efforts": [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
],
|
||||
"default_effort": "medium",
|
||||
"normalization_policy": "none"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,114 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Normative corpus runner — validates the generated TS interpreter against
|
||||
* scripts/normative-corpus.json.
|
||||
*
|
||||
* This is the JS side of the two-interpreter corpus check. The runner imports
|
||||
* resolveModelCapabilities() directly from the generated TypeScript module via
|
||||
* Node's --experimental-strip-types flag. The Rust side lives in
|
||||
* crates/buzz-agent/src/generated_model_capabilities.rs (shared corpus harness).
|
||||
*
|
||||
* Usage: node --experimental-strip-types scripts/run-corpus.mjs [--verbose]
|
||||
* Exits 0 on all pass, 1 on any failure.
|
||||
*/
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { join, dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const repoRoot = join(__dirname, "..");
|
||||
const VERBOSE = process.argv.includes("--verbose");
|
||||
|
||||
// Import the generated TypeScript interpreter directly.
|
||||
// Node 22+ --experimental-strip-types strips type annotations at load time; no build step needed.
|
||||
const { resolveModelCapabilities } = await import(
|
||||
join(repoRoot, "desktop", "src", "features", "agents", "ui", "modelCapabilities.ts")
|
||||
);
|
||||
|
||||
const corpus = JSON.parse(
|
||||
readFileSync(join(repoRoot, "scripts", "normative-corpus.json"), "utf8"),
|
||||
);
|
||||
|
||||
// ----- Provider alias canonicalization -----
|
||||
// Mirrors production canonicalizeProvider() in desktop/src/features/agents/lib/formatAgentModelLabel.ts.
|
||||
// Applied before every generated lookup so alias vectors (e.g. "openai-compat") pass both interpreters.
|
||||
const PROVIDER_ALIASES = new Map([
|
||||
["databricks-v2", "databricks_v2"],
|
||||
["openai-compat", "openai"],
|
||||
]);
|
||||
|
||||
function canonicalizeProvider(provider) {
|
||||
const normalized = (provider ?? "").trim().toLowerCase();
|
||||
return PROVIDER_ALIASES.get(normalized) ?? normalized;
|
||||
}
|
||||
|
||||
// ----- Run corpus -----
|
||||
|
||||
let passed = 0;
|
||||
let failed = 0;
|
||||
|
||||
const REQUIRED_AXES = [
|
||||
"thinking_mode",
|
||||
"supported_efforts",
|
||||
"default_effort",
|
||||
"databricks_v2_wire_route",
|
||||
"normalization_policy",
|
||||
"registry_label",
|
||||
];
|
||||
|
||||
for (const entry of corpus) {
|
||||
// Skip group header entries
|
||||
if (entry._group) continue;
|
||||
if (!entry.expect) continue;
|
||||
|
||||
// Require all 6 axes on every executable vector — sparse vectors hide divergences.
|
||||
const missingAxes = REQUIRED_AXES.filter((ax) => !(ax in entry.expect));
|
||||
if (missingAxes.length > 0) {
|
||||
failed++;
|
||||
console.error(` FAIL ${entry.id}: sparse vector — missing axes: ${missingAxes.join(", ")}`);
|
||||
console.error(` All six axes are required: ${REQUIRED_AXES.join(", ")}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
// resolveModelCapabilities returns camelCase keys (registryLabel, thinkingMode, etc.)
|
||||
const result = resolveModelCapabilities(canonicalizeProvider(entry.provider), entry.raw_model_id);
|
||||
const expect = entry.expect;
|
||||
|
||||
const failures = [];
|
||||
|
||||
for (const [key, expectedVal] of Object.entries(expect)) {
|
||||
// Corpus uses snake_case; generated TS uses camelCase — convert for lookup.
|
||||
const camelKey = key.replace(/_([a-z])/g, (_, c) => c.toUpperCase());
|
||||
const actualVal = camelKey in result ? result[camelKey] : result[key];
|
||||
|
||||
if (Array.isArray(expectedVal)) {
|
||||
// Order-sensitive comparison for effort arrays
|
||||
const actualArr = Array.isArray(actualVal) ? actualVal : [];
|
||||
if (JSON.stringify(actualArr) !== JSON.stringify(expectedVal)) {
|
||||
failures.push(
|
||||
` ${key}: expected [${expectedVal.join(", ")}] got [${actualArr.join(", ")}]`,
|
||||
);
|
||||
}
|
||||
} else {
|
||||
if (actualVal !== expectedVal) {
|
||||
failures.push(` ${key}: expected ${JSON.stringify(expectedVal)} got ${JSON.stringify(actualVal)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (failures.length === 0) {
|
||||
passed++;
|
||||
if (VERBOSE) {
|
||||
console.log(` PASS ${entry.id}`);
|
||||
}
|
||||
} else {
|
||||
failed++;
|
||||
console.error(` FAIL ${entry.id} (${entry.provider} / "${entry.raw_model_id}")`);
|
||||
for (const f of failures) console.error(f);
|
||||
if (entry._note) console.error(` note: ${entry._note}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`\nCorpus: ${passed} passed, ${failed} failed`);
|
||||
process.exit(failed > 0 ? 1 : 0);
|
||||
@@ -0,0 +1,669 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Schema-negative validator tests for the manifest generator.
|
||||
*
|
||||
* Every validator rule in generate-model-capabilities.mjs must have a
|
||||
* failing-input test here — if validation is missing, these tests would
|
||||
* not catch the defect.
|
||||
*
|
||||
* Uses Node.js built-in test runner (node --test).
|
||||
*/
|
||||
|
||||
import { test } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { readFileSync, writeFileSync, unlinkSync, mkdirSync, mkdtempSync } from "node:fs";
|
||||
import { join, dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { tmpdir } from "node:os";
|
||||
import { execFileSync, spawnSync } from "node:child_process";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const repoRoot = join(__dirname, "..");
|
||||
const manifestPath = join(repoRoot, "scripts", "model-capabilities.json");
|
||||
const generatorPath = join(repoRoot, "scripts", "generate-model-capabilities.mjs");
|
||||
|
||||
/** Load the real manifest so we can mutate copies. */
|
||||
const BASE_MANIFEST = JSON.parse(readFileSync(manifestPath, "utf8"));
|
||||
|
||||
/**
|
||||
* Run the generator with a mutated manifest, returning { exitCode, stderr, stdout }.
|
||||
* Writes the mutated manifest to a temp file and overrides the manifest path via env.
|
||||
*/
|
||||
function runGeneratorWithManifest(manifestOverride) {
|
||||
// Write mutated manifest to a temp path
|
||||
const tmpDir = mkdtempSync(join(tmpdir(), "test-validator-"));
|
||||
const tmpManifest = join(tmpDir, "model-capabilities.json");
|
||||
const tmpOutputDir = join(tmpDir, "out");
|
||||
mkdirSync(tmpOutputDir, { recursive: true });
|
||||
|
||||
writeFileSync(tmpManifest, JSON.stringify(manifestOverride));
|
||||
|
||||
// Run the generator via node, pointing MANIFEST_PATH env at the temp file
|
||||
// The generator reads from process.env.MANIFEST_PATH if set (we add this support)
|
||||
const result = spawnSync(
|
||||
process.execPath,
|
||||
[generatorPath, "--manifest-path", tmpManifest, "--output-dir", tmpOutputDir],
|
||||
{
|
||||
encoding: "utf8",
|
||||
env: { ...process.env },
|
||||
},
|
||||
);
|
||||
|
||||
// Cleanup
|
||||
try { unlinkSync(tmpManifest); } catch {}
|
||||
|
||||
return { exitCode: result.status ?? 1, stderr: result.stderr, stdout: result.stdout };
|
||||
}
|
||||
|
||||
/**
|
||||
* Assert that the generator REJECTS the given manifest (exits non-zero).
|
||||
* The optional `expectedMessage` is checked in stderr if provided.
|
||||
*/
|
||||
function assertRejects(label, manifest, expectedMessage) {
|
||||
const { exitCode, stderr, stdout } = runGeneratorWithManifest(manifest);
|
||||
assert.notEqual(exitCode, 0, `${label}: expected generator to fail but it succeeded.\nstdout: ${stdout}\nstderr: ${stderr}`);
|
||||
if (expectedMessage) {
|
||||
const combined = stderr + stdout;
|
||||
assert.ok(
|
||||
combined.includes(expectedMessage),
|
||||
`${label}: expected error message "${expectedMessage}" not found.\nstdout: ${stdout}\nstderr: ${stderr}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/** Deep clone the base manifest and apply a mutator function. */
|
||||
function mutate(fn) {
|
||||
const clone = JSON.parse(JSON.stringify(BASE_MANIFEST));
|
||||
fn(clone);
|
||||
return clone;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid enum value in family_rule.thinking_mode
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid thinking_mode in family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid thinking_mode",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].thinking_mode = "invalid-mode";
|
||||
}),
|
||||
"thinking_mode",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid enum value in family_rule.databricks_v2_wire_route
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid databricks_v2_wire_route in family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid databricks_v2_wire_route",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].databricks_v2_wire_route = "chat-completions";
|
||||
}),
|
||||
"databricks_v2_wire_route",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid enum value in family_rule.supported_efforts[]
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid effort value in family rule supported_efforts is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid supported_efforts value",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].supported_efforts = ["low", "ultra-high"];
|
||||
}),
|
||||
"supported_efforts",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: empty supported_efforts array in family rule
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: empty supported_efforts in family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"empty supported_efforts",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].supported_efforts = [];
|
||||
}),
|
||||
"supported_efforts",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: default_effort not in supported_efforts (non-null)
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: default_effort not in supported_efforts is rejected", () => {
|
||||
assertRejects(
|
||||
"default_effort not in supported_efforts",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].supported_efforts = ["low", "medium"];
|
||||
m.family_rules[0].default_effort = "high"; // not in list
|
||||
}),
|
||||
"default_effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid normalization_policy in family rule
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid normalization_policy in family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid normalization_policy",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].normalization_policy = "pass-through-all";
|
||||
}),
|
||||
"normalization_policy",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid match_kind in family rule
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid match_kind in family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid match_kind",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].match_kind = "regex";
|
||||
}),
|
||||
"match_kind",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: duplicate family rule id
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: duplicate family rule id is rejected", () => {
|
||||
assertRejects(
|
||||
"duplicate family rule id",
|
||||
mutate((m) => {
|
||||
m.family_rules.push({ ...m.family_rules[0] }); // duplicate id
|
||||
}),
|
||||
"duplicate",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: registry_label on a family rule is rejected (display rule (a))
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: registry_label on a family rule is rejected", () => {
|
||||
assertRejects(
|
||||
"family rule registry_label forbidden",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].registry_label = "Family Masquerade";
|
||||
}),
|
||||
"registry_label is not allowed on family rules",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: duplicate exact_record (provider, raw_model_id) key
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: duplicate exact_record key is rejected", () => {
|
||||
assertRejects(
|
||||
"duplicate exact_record key",
|
||||
mutate((m) => {
|
||||
m.exact_records.push({ ...m.exact_records[0] }); // duplicate
|
||||
}),
|
||||
"duplicate",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record missing provider
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record missing provider is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record missing provider",
|
||||
mutate((m) => {
|
||||
m.exact_records.push({ raw_model_id: "some-model" });
|
||||
}),
|
||||
"provider",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record missing raw_model_id
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record missing raw_model_id is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record missing raw_model_id",
|
||||
mutate((m) => {
|
||||
m.exact_records.push({ provider: "databricks_v2" });
|
||||
}),
|
||||
"raw_model_id",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: provider fallback record missing blank state
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: provider fallback missing blank state is rejected", () => {
|
||||
assertRejects(
|
||||
"provider fallback missing blank",
|
||||
mutate((m) => {
|
||||
delete m.provider_fallbacks.anthropic.blank;
|
||||
}),
|
||||
"blank",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: provider fallback record missing concrete_unknown state
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: provider fallback missing concrete_unknown state is rejected", () => {
|
||||
assertRejects(
|
||||
"provider fallback missing concrete_unknown",
|
||||
mutate((m) => {
|
||||
delete m.provider_fallbacks.anthropic.concrete_unknown;
|
||||
}),
|
||||
"concrete_unknown",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid thinking_mode in provider fallback
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid thinking_mode in provider fallback is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid thinking_mode in fallback",
|
||||
mutate((m) => {
|
||||
m.provider_fallbacks.anthropic.blank.thinking_mode = "always-on";
|
||||
}),
|
||||
"thinking_mode",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid databricks_v2_wire_route in provider fallback
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: invalid wire_route in provider fallback is rejected", () => {
|
||||
assertRejects(
|
||||
"invalid wire_route in fallback",
|
||||
mutate((m) => {
|
||||
m.provider_fallbacks.anthropic.blank.databricks_v2_wire_route = "http-sse";
|
||||
}),
|
||||
"databricks_v2_wire_route",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: invalid default_effort in provider fallback (not in supported_efforts)
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: default_effort not in supported_efforts in fallback is rejected", () => {
|
||||
assertRejects(
|
||||
"default_effort not in supported_efforts in fallback",
|
||||
mutate((m) => {
|
||||
m.provider_fallbacks.openai.blank.supported_efforts = ["low", "medium"];
|
||||
m.provider_fallbacks.openai.blank.default_effort = "high"; // not in list
|
||||
}),
|
||||
"default_effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: family rule missing id
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: family rule missing id is rejected", () => {
|
||||
assertRejects(
|
||||
"family rule missing id",
|
||||
mutate((m) => {
|
||||
m.family_rules.push({
|
||||
match_kind: "prefix",
|
||||
match_value: "test-",
|
||||
providers: ["anthropic"],
|
||||
match_priority: 1,
|
||||
thinking_mode: "none",
|
||||
supported_efforts: ["low"],
|
||||
default_effort: null,
|
||||
databricks_v2_wire_route: "not-applicable",
|
||||
normalization_policy: "none",
|
||||
// id deliberately omitted
|
||||
});
|
||||
}),
|
||||
"id",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: duplicate exact_record key (same provider + raw_model_id) is rejected
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: duplicate exact_record key is rejected", () => {
|
||||
assertRejects(
|
||||
"duplicate exact_record key",
|
||||
mutate((m) => {
|
||||
// Add a second record for the same (provider, raw_model_id) key
|
||||
const existing = m.exact_records.find(
|
||||
(r) => r.raw_model_id === "databricks-gpt-5-4-mini",
|
||||
);
|
||||
m.exact_records.push({ ...existing });
|
||||
}),
|
||||
"duplicate",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record registry_label with empty label is rejected
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record registry_label with empty string is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record registry_label empty string",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find(
|
||||
(r) => r.raw_model_id === "databricks-gpt-5-4-mini",
|
||||
);
|
||||
rec.registry_label = "";
|
||||
}),
|
||||
"nonempty",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: duplicate databricks_v2_known_models IDs
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: duplicate databricks_v2_known_models ID is rejected", () => {
|
||||
assertRejects(
|
||||
"duplicate known model ID",
|
||||
mutate((m) => {
|
||||
m.databricks_v2_known_models = ["databricks-gpt-5-5", "databricks-gpt-5-5"];
|
||||
}),
|
||||
"duplicate",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: unsafe characters in match_value (family rule)
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: family rule match_value with unsafe chars is rejected", () => {
|
||||
assertRejects(
|
||||
"family rule match_value with backslash",
|
||||
mutate((m) => {
|
||||
// Inject a backslash into an existing rule's match_value — would break Rust string literal
|
||||
const rule = m.family_rules.find((r) => r.id === "anthropic-manual-budget-claude3");
|
||||
rule.match_value = "claude-3\\evil";
|
||||
}),
|
||||
"unsafe",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: unsafe characters in known-model ID
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: databricks_v2_known_models ID with unsafe chars is rejected", () => {
|
||||
assertRejects(
|
||||
"known-model ID with double-quote",
|
||||
mutate((m) => {
|
||||
m.databricks_v2_known_models = ['databricks-gpt-5-5', 'bad"id'];
|
||||
}),
|
||||
"unsafe",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: unsafe characters in exact_record registry_label
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record registry_label with unsafe chars is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record registry_label with backslash",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.registry_label = "GPT-5.4 Mini\\injected";
|
||||
}),
|
||||
"unsafe",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record supported_efforts_override must be non-empty
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record empty supported_efforts_override is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record empty supported_efforts_override",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.supported_efforts_override = [];
|
||||
}),
|
||||
"supported_efforts_override",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record supported_efforts_override with invalid enum value
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record supported_efforts_override with bogus enum is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record bogus effort enum",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.supported_efforts_override = ["ultra-high"];
|
||||
}),
|
||||
"supported_efforts_override",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record supported_efforts_override with duplicate effort
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record supported_efforts_override with duplicate effort is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record duplicate effort",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.supported_efforts_override = ["low", "low"];
|
||||
}),
|
||||
"duplicate",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record supported_efforts_override must follow canonical order
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record supported_efforts_override out of canonical order is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record efforts out of order",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
// Reverse order — [high, medium, low] is not canonical [low, medium, high]
|
||||
rec.supported_efforts_override = ["high", "medium", "low"];
|
||||
}),
|
||||
"canonical order",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record default_effort not in supported_efforts_override
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record default_effort not in supported_efforts_override is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record default_effort outside override",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.supported_efforts_override = ["low", "medium"];
|
||||
rec.default_effort = "high"; // not in override
|
||||
}),
|
||||
"default_effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record match_priority must be a non-negative integer
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record match_priority non-integer is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record match_priority non-integer",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.match_priority = "five";
|
||||
}),
|
||||
"match_priority",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record thinking_mode must be a valid enum
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record invalid thinking_mode is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record invalid thinking_mode",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.thinking_mode = "turbo-thinking";
|
||||
}),
|
||||
"thinking_mode",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record databricks_v2_wire_route must be a valid enum
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record invalid databricks_v2_wire_route is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record invalid wire_route",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.databricks_v2_wire_route = "http-sse";
|
||||
}),
|
||||
"databricks_v2_wire_route",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record normalization_policy must be a valid enum
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record invalid normalization_policy is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record invalid normalization_policy",
|
||||
mutate((m) => {
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
rec.normalization_policy = "pass-through-all";
|
||||
}),
|
||||
"normalization_policy",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: family_rule match_priority must be a non-negative integer
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: family_rule match_priority string (injection vector) is rejected", () => {
|
||||
assertRejects(
|
||||
"family_rule string match_priority",
|
||||
mutate((m) => {
|
||||
// Inject a string that contains a compile_error! macro — must be rejected before emission
|
||||
m.family_rules[0].match_priority = 'compile_error!("THUFIR_INJECTED")';
|
||||
}),
|
||||
"match_priority",
|
||||
);
|
||||
});
|
||||
|
||||
test("schema-negative: family_rule match_priority negative integer is rejected", () => {
|
||||
assertRejects(
|
||||
"family_rule negative match_priority",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].match_priority = -1;
|
||||
}),
|
||||
"match_priority",
|
||||
);
|
||||
});
|
||||
|
||||
test("schema-negative: family_rule match_priority float is rejected", () => {
|
||||
assertRejects(
|
||||
"family_rule float match_priority",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].match_priority = 1.5;
|
||||
}),
|
||||
"match_priority",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record uppercase duplicate key is rejected (after lowercasing)
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record uppercase duplicate key is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record uppercase duplicate key",
|
||||
mutate((m) => {
|
||||
// Add an uppercase copy of an existing exact record key
|
||||
const existing = m.exact_records[0];
|
||||
m.exact_records.push({
|
||||
...existing,
|
||||
provider: existing.provider.toUpperCase(),
|
||||
raw_model_id: existing.raw_model_id.toUpperCase(),
|
||||
});
|
||||
}),
|
||||
"duplicate exact_record key",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: exact_record inherited default_effort not in inherited supported_efforts
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: exact_record inherited default_effort outside materialized supported_efforts is rejected", () => {
|
||||
assertRejects(
|
||||
"exact_record inherited default out of materialized efforts",
|
||||
mutate((m) => {
|
||||
// Override supported_efforts_override to a single value that excludes the family default.
|
||||
// For any exact record that inherits family default_effort, override efforts to exclude it.
|
||||
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
|
||||
// Family default for the gpt5-4 rule is "medium". Override to only ["low"] to force mismatch.
|
||||
rec.supported_efforts_override = ["low"];
|
||||
// No explicit default_effort — inherits "medium" from family, but "medium" is not in ["low"]
|
||||
}),
|
||||
"materialized default_effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: family_rule supported_efforts must have no duplicates
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: family_rule supported_efforts with duplicate is rejected", () => {
|
||||
assertRejects(
|
||||
"family_rule duplicate effort",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].supported_efforts = ["low", "low", "medium"];
|
||||
}),
|
||||
"duplicate effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: family_rule supported_efforts must follow canonical order
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: family_rule supported_efforts out of canonical order is rejected", () => {
|
||||
assertRejects(
|
||||
"family_rule efforts out of order",
|
||||
mutate((m) => {
|
||||
m.family_rules[0].supported_efforts = ["high", "low", "medium"];
|
||||
}),
|
||||
"canonical order",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: provider_fallback supported_efforts must have no duplicates
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: provider_fallback supported_efforts with duplicate is rejected", () => {
|
||||
assertRejects(
|
||||
"provider_fallback duplicate effort",
|
||||
mutate((m) => {
|
||||
const provider = Object.keys(m.provider_fallbacks)[0];
|
||||
m.provider_fallbacks[provider].blank.supported_efforts = ["low", "low", "medium"];
|
||||
}),
|
||||
"duplicate effort",
|
||||
);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rule: provider_fallback supported_efforts must follow canonical order
|
||||
// ---------------------------------------------------------------------------
|
||||
test("schema-negative: provider_fallback supported_efforts out of canonical order is rejected", () => {
|
||||
assertRejects(
|
||||
"provider_fallback efforts out of order",
|
||||
mutate((m) => {
|
||||
const provider = Object.keys(m.provider_fallbacks)[0];
|
||||
m.provider_fallbacks[provider].blank.supported_efforts = ["high", "low", "medium"];
|
||||
}),
|
||||
"canonical order",
|
||||
);
|
||||
});
|
||||
|
||||
console.log("\nSchema-negative validator tests complete.");
|
||||
Reference in New Issue
Block a user