mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
refactor(models): Phase 3 — retire old hand tables and shrink manifest apparatus (#4589)
## Summary Retires the old hand-table authorities and transitional verification scaffolding from the model-capability manifest arc. All production routes now run exclusively through the generated interpreters introduced in Phase 1 ([#3821](https://github.com/block/buzz/pull/3821)) and wired in Phase 2 ([#3958](https://github.com/block/buzz/pull/3958)). Stack: [#3821](https://github.com/block/buzz/pull/3821) → [#3958](https://github.com/block/buzz/pull/3958) → this PR Base: [#3603](https://github.com/block/buzz/pull/3603) ## What the manifest system is now **Source of truth:** `scripts/model-capabilities.json` **Generator:** `scripts/generate-model-capabilities.mjs` — emits Rust and TS interpreters only (coverage JSON output removed) **Generated interpreters:** `crates/buzz-agent/src/generated_model_capabilities.rs` (+ normative tests), `desktop/src/features/agents/ui/modelCapabilities.ts` **Label registry:** `generate-databricks-model-names.py` → `databricks_model_names.rs` / `databricksModelNames.ts` **Permanent gates:** `scripts/normative-corpus.json` + `scripts/run-corpus.mjs` (both-interpreter equivalence, 51 vectors), `scripts/test-manifest-validator.mjs` (schema, 24 cases), regen-diff job inside `ci.yml` **One doc:** `scripts/MODEL_CAPABILITIES.md` ## Deleted **Old hand-table authorities (production):** - `getProviderEffortConfig_oldHandTable()` and all supporting helpers from `desktop/src/features/agents/ui/buzzAgentConfig.ts` - `normalize_effort_for_openai_route()`, `_old_anthropic_thinking_config_for_databricks_v2()`, test-only re-export wrappers from `crates/buzz-agent/src/config.rs` - `strip_catalog_prefix()`, `anthropic_thinking_config()`, `anthropic_model_supports_xhigh()`, `clamp_adaptive_effort()`, `anthropic_efforts_for_model()`, `is_manual_budget_model()`, `is_adaptive_thinking_model()`, `gpt5_token_matches()`, `gpt5_base_matches()`, `openai_efforts_for_model()` from `crates/buzz-agent/src/config.rs` — all superseded by generated interpreter - Old DBv2 body-level tests, `_OLD_DATABRICKS_V2_*` constants, `model_name_segments()`, `_old_databricks_v2_route_for_model()`, all Phase-2 behavioral differential test functions from `crates/buzz-agent/src/llm.rs` **Transitional scaffolding:** - `scripts/run-differential.mjs` — old-vs-new JS differential harness - `scripts/run-mutation-evidence.mjs` — one-time mutation evidence runner - `desktop/src/features/agents/ui/effortTable.fixture.json` — Phase-2 TS/Rust sync fixture - `desktop/src/features/agents/ui/effortTable.fixture.test.mjs` — fixture sync guard - `.github/workflows/model-capability-regen-diff.yml` — standalone workflow (steps folded into `ci.yml`) **One-time evidence and generated snapshots:** - `scripts/MUTATION_EVIDENCE.md`, `scripts/MODEL_CAPABILITIES_SCHEMA.md`, `scripts/MODELS_DEV_RECONCILIATION.md` — consolidated into `scripts/MODEL_CAPABILITIES.md` - `scripts/generated-model-capabilities-coverage.json` — full-table snapshot (generator no longer emits it) - `scripts/catalog-sample-fixture.json` — models.dev snapshot used only by the deleted differential harness ## Verification - `cargo test -p buzz-agent --lib` with `RUSTFLAGS="-D warnings"`: **337/337** (clean — no dead_code warnings) - `node --experimental-strip-types scripts/run-corpus.mjs`: **51/51** - `node scripts/generate-model-capabilities.mjs` + regen diff: **clean (exit 0)** - `node --test scripts/test-manifest-validator.mjs`: **24/24** - `just clippy`: zero warnings, zero errors --------- Signed-off-by: Will Pfleger <pfleger.will@gmail.com> Co-authored-by: npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7 <dcfd242e557282d7a1e2cf2e6877522682f1e5c6156dc92ca7d90eaedd3b0f95@buzz.block.builderlab.xyz>
This commit is contained in:
co-authored by
npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7
parent
ad9ae1dbb0
commit
899f55cc8d
@@ -130,6 +130,50 @@ jobs:
|
||||
- name: Unit tests
|
||||
run: just test-unit
|
||||
|
||||
model-capabilities:
|
||||
name: Model Capabilities (regen + corpus + schema + buzz-agent tests)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
||||
- uses: cashapp/activate-hermit@cea9af7913204a965fd488637a8d1811bba2e616 # v1
|
||||
- uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2
|
||||
with:
|
||||
save-if: ${{ github.event_name != 'pull_request' }}
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
|
||||
with:
|
||||
node-version: '22'
|
||||
package-manager-cache: false
|
||||
|
||||
- name: Regenerate artifacts
|
||||
run: node scripts/generate-model-capabilities.mjs
|
||||
|
||||
- name: Diff check — fail if generated files are stale
|
||||
run: |
|
||||
if ! git diff --exit-code \
|
||||
crates/buzz-agent/src/generated_model_capabilities.rs \
|
||||
desktop/src/features/agents/ui/modelCapabilities.ts; then
|
||||
echo ""
|
||||
echo "ERROR: Generated model-capability files are stale."
|
||||
echo "Run: node scripts/generate-model-capabilities.mjs"
|
||||
echo "Then commit the regenerated files."
|
||||
exit 1
|
||||
fi
|
||||
echo "✓ All generated files are up to date."
|
||||
|
||||
- name: Run corpus (TS interpreter via --experimental-strip-types)
|
||||
run: node --experimental-strip-types scripts/run-corpus.mjs
|
||||
|
||||
- name: Validate manifest (schema-negative tests)
|
||||
run: node --test scripts/test-manifest-validator.mjs
|
||||
|
||||
- name: Run buzz-agent unit tests (normative corpus + generated interpreter)
|
||||
run: cargo test -p buzz-agent --lib
|
||||
|
||||
desktop-core:
|
||||
name: Desktop Core
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
name: Model Capability Regenerate-Then-Diff
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'scripts/model-capabilities.json'
|
||||
- 'scripts/generate-model-capabilities.mjs'
|
||||
- 'crates/buzz-agent/src/generated_model_capabilities.rs'
|
||||
- 'desktop/src/features/agents/ui/modelCapabilities.ts'
|
||||
- 'scripts/generated-model-capabilities-coverage.json'
|
||||
- '.github/workflows/model-capability-regen-diff.yml'
|
||||
# Differential harness and fixtures — any change to old/new side or inputs re-runs.
|
||||
- 'scripts/run-differential.mjs'
|
||||
- 'scripts/normative-corpus.json'
|
||||
- 'scripts/catalog-sample-fixture.json'
|
||||
- 'desktop/src/features/agents/ui/effortTable.fixture.json'
|
||||
- 'desktop/src/features/agents/ui/buzzAgentConfig.ts'
|
||||
- 'crates/buzz-agent/src/config.rs'
|
||||
- 'crates/buzz-agent/src/llm.rs'
|
||||
push:
|
||||
branches: [main, release, 'duncan/databricks-model-label-registry']
|
||||
|
||||
jobs:
|
||||
regen-diff:
|
||||
name: Regenerate and diff model capability artifacts
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
|
||||
with:
|
||||
node-version: '22'
|
||||
package-manager-cache: false
|
||||
|
||||
- name: Regenerate artifacts
|
||||
run: node scripts/generate-model-capabilities.mjs
|
||||
|
||||
- name: Diff check — fail if generated files are stale
|
||||
run: |
|
||||
if ! git diff --exit-code \
|
||||
crates/buzz-agent/src/generated_model_capabilities.rs \
|
||||
desktop/src/features/agents/ui/modelCapabilities.ts \
|
||||
scripts/generated-model-capabilities-coverage.json; then
|
||||
echo ""
|
||||
echo "ERROR: Generated model-capability files are stale."
|
||||
echo "Run: node scripts/generate-model-capabilities.mjs"
|
||||
echo "Then commit the regenerated files."
|
||||
exit 1
|
||||
fi
|
||||
echo "✓ All generated files are up to date."
|
||||
|
||||
- name: Run corpus (TS interpreter via --experimental-strip-types)
|
||||
run: node --experimental-strip-types scripts/run-corpus.mjs
|
||||
|
||||
- name: Validate manifest (schema-negative tests)
|
||||
run: node --test scripts/test-manifest-validator.mjs
|
||||
|
||||
- name: Run differential harness (old vs new, all input sets)
|
||||
run: node --experimental-strip-types scripts/run-differential.mjs
|
||||
|
||||
rust-unit-tests:
|
||||
name: buzz-agent unit tests (normative corpus + behavioral differential)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
||||
|
||||
- name: Run buzz-agent unit tests
|
||||
run: cargo test -p buzz-agent --lib
|
||||
+116
-1852
File diff suppressed because it is too large
Load Diff
@@ -1336,7 +1336,6 @@ pub const DATABRICKS_MODEL_NAMES: &[(&str, &str)] = &[
|
||||
|
||||
/// Returns true if `model` contains `token` at a word boundary (end-of-string or "-").
|
||||
/// Does not match if followed immediately by a digit or letter.
|
||||
/// Mirrors gpt5_token_matches in config.rs.
|
||||
fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
|
||||
let lower = model;
|
||||
let tok_lower = token;
|
||||
|
||||
@@ -8,7 +8,9 @@
|
||||
//! cross-interpreter conformance gate; the JS runner executes the same file.
|
||||
//! 2. Handwritten supplement tests — adversarial cases and completeness checks that
|
||||
//! benefit from Rust-specific assertion ergonomics.
|
||||
//! 3. Per-interpreter mutation evidence — see scripts/MUTATION_EVIDENCE.md.
|
||||
//! 3. Per-interpreter mutation evidence — all 7 mutations were killed by both interpreters
|
||||
//! (2026-07-31, Phase 2). The mutation runner was deleted in Phase 3; see
|
||||
//! `scripts/MODEL_CAPABILITIES.md` for the historical record.
|
||||
|
||||
#[cfg(test)]
|
||||
mod shared_corpus_tests {
|
||||
|
||||
+24
-1076
File diff suppressed because it is too large
Load Diff
@@ -1,10 +1,8 @@
|
||||
/**
|
||||
* Source-of-truth constants for buzz-agent model-tuning configuration knobs.
|
||||
*
|
||||
* Phase 2b pass 2: getProviderEffortConfig() is now generated-backed (thin
|
||||
* wrapper over resolveModelCapabilities()). The legacy hand-table implementation
|
||||
* is preserved as getProviderEffortConfig_oldHandTable() for the differential
|
||||
* harness only — nothing user-facing imports the _old shim. Phase 3 retires it.
|
||||
* `getProviderEffortConfig()` is generated-backed (thin wrapper over
|
||||
* `resolveModelCapabilities()`).
|
||||
*/
|
||||
import { canonicalizeProvider } from "../lib/formatAgentModelLabel.ts";
|
||||
import { resolveModelCapabilities } from "./modelCapabilities.ts";
|
||||
@@ -50,10 +48,6 @@ export type ThinkingEffortValue =
|
||||
* thinking configuration entirely (i.e. "Inherit" is the natural default).
|
||||
* This applies to Anthropic manual-budget models where the effort level maps
|
||||
* to a budget_tokens count — there is no "default effort level" in the API.
|
||||
*
|
||||
* Mirrors the model-family tables in `crates/buzz-agent/src/config.rs`
|
||||
* (`openai_efforts_for_model`, `is_manual_budget_model`,
|
||||
* `is_adaptive_thinking_model`, `clamp_adaptive_effort`). Keep in sync.
|
||||
*/
|
||||
export type ProviderEffortConfig = {
|
||||
validValues: ReadonlyArray<ThinkingEffortValue>;
|
||||
@@ -61,8 +55,6 @@ export type ProviderEffortConfig = {
|
||||
defaultValue: ThinkingEffortValue | null;
|
||||
};
|
||||
|
||||
const ALL_VALUES = BUZZ_AGENT_THINKING_EFFORT_VALUES;
|
||||
|
||||
/**
|
||||
* Returns the valid thinking-effort values and semantic default for the
|
||||
* given provider and optional model string, resolved from the generated
|
||||
@@ -91,237 +83,6 @@ export function getProviderEffortConfig(
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Legacy hand-table implementation — differential harness shim only.
|
||||
*
|
||||
* Preserved for the run-differential.mjs old-vs-new comparison until Phase 3
|
||||
* retires it. Nothing user-facing should import this name.
|
||||
*
|
||||
* @deprecated Use getProviderEffortConfig() (generated-backed) instead.
|
||||
*/
|
||||
export function getProviderEffortConfig_oldHandTable(
|
||||
providerId: string,
|
||||
model?: string,
|
||||
): ProviderEffortConfig {
|
||||
const provider = providerId.toLowerCase();
|
||||
// Strip arbitrary endpoint-naming prefix before model-family matching.
|
||||
// Find the first occurrence of a known family token and drop everything before it.
|
||||
// e.g. "goose-claude-fable-5" → "claude-fable-5"
|
||||
// "team-x-gpt-5.5" → "gpt-5.5"
|
||||
// "databricks-claude-3" → "claude-3"
|
||||
// "claude-opus-4-7" → "claude-opus-4-7" (no prefix to strip)
|
||||
const rawModel = (model ?? "").trim().toLowerCase();
|
||||
const FAMILY_TOKENS = ["claude-", "gpt-"] as const;
|
||||
const firstFamilyIdx = Math.min(
|
||||
...FAMILY_TOKENS.map((tok) => {
|
||||
const idx = rawModel.indexOf(tok);
|
||||
return idx === -1 ? Infinity : idx;
|
||||
}),
|
||||
);
|
||||
const m =
|
||||
firstFamilyIdx === Infinity ? rawModel : rawModel.slice(firstFamilyIdx);
|
||||
|
||||
if (provider === "anthropic") {
|
||||
return anthropicConfig(m);
|
||||
}
|
||||
if (provider === "openai") {
|
||||
return openaiConfig(m);
|
||||
}
|
||||
if (provider === "databricks_v2") {
|
||||
// Route by model family: claude* → Anthropic tables, gpt-5* → OpenAI tables.
|
||||
// Non-Claude concrete models (e.g. llama-3) go through MlflowChatCompletions,
|
||||
// which applies normalize_effort_for_openai_route → clamps max to xhigh.
|
||||
// Route them through openaiConfig to exclude max. Only blank/unknown model
|
||||
// uses the all-7 fallback (can't know the route without a concrete model).
|
||||
if (m.startsWith("claude-")) {
|
||||
return anthropicConfig(m);
|
||||
}
|
||||
if (gpt5FamilyModel(m)) {
|
||||
return openaiConfig(m);
|
||||
}
|
||||
if (m.length > 0) {
|
||||
// Concrete non-Claude, non-GPT model → MLflow path clamps max → xhigh.
|
||||
return openaiConfig(m);
|
||||
}
|
||||
// Blank model — route unknown, show all 7.
|
||||
return { validValues: ALL_VALUES, defaultValue: "medium" };
|
||||
}
|
||||
if (provider === "databricks") {
|
||||
// databricks v1 uses OpenAI Chat Completions wire format.
|
||||
return openaiConfig(m);
|
||||
}
|
||||
if (provider === "openrouter") {
|
||||
return { validValues: ALL_VALUES, defaultValue: "medium" };
|
||||
}
|
||||
// openai-compat, unknown, empty — all values, default medium.
|
||||
return { validValues: ALL_VALUES, defaultValue: "medium" };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Anthropic family tables
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
function anthropicConfig(m: string): ProviderEffortConfig {
|
||||
// Manual-budget models: claude-3* and claude-opus-4-5.
|
||||
// These use budget_tokens — there is no "default effort level" in the API.
|
||||
if (m.startsWith("claude-3") || m === "claude-opus-4-5") {
|
||||
return {
|
||||
validValues: ["low", "medium", "high"],
|
||||
defaultValue: null,
|
||||
};
|
||||
}
|
||||
// Adaptive models that support xhigh: opus-4-7+, sonnet-5.x, fable-5, mythos-5.
|
||||
// mirrors clamp_adaptive_effort supports_xhigh check.
|
||||
if (
|
||||
m.startsWith("claude-opus-4-7") ||
|
||||
m.startsWith("claude-opus-4-8") ||
|
||||
m.startsWith("claude-sonnet-5") ||
|
||||
m.startsWith("claude-fable-5") ||
|
||||
m.startsWith("claude-mythos-5")
|
||||
) {
|
||||
return {
|
||||
validValues: ["low", "medium", "high", "xhigh", "max"],
|
||||
defaultValue: "high",
|
||||
};
|
||||
}
|
||||
// Adaptive models that do NOT support xhigh: opus-4-6, sonnet-4-6, mythos-preview.
|
||||
if (
|
||||
m.startsWith("claude-opus-4-6") ||
|
||||
m.startsWith("claude-sonnet-4-6") ||
|
||||
m.startsWith("claude-mythos-preview")
|
||||
) {
|
||||
return {
|
||||
validValues: ["low", "medium", "high", "max"],
|
||||
defaultValue: "high",
|
||||
};
|
||||
}
|
||||
// Unknown Anthropic model — assume adaptive with full support.
|
||||
return {
|
||||
validValues: ["low", "medium", "high", "xhigh", "max"],
|
||||
defaultValue: "high",
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// OpenAI family tables — mirrors openai_efforts_for_model in config.rs
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Returns true if `m` contains a GPT-5 family token at a word boundary
|
||||
* (not immediately followed by a digit or letter). Mirrors
|
||||
* `gpt5_token_matches` / `gpt5_base_matches` in config.rs.
|
||||
*/
|
||||
function gpt5TokenMatches(m: string, token: string): boolean {
|
||||
let start = 0;
|
||||
while (true) {
|
||||
const idx = m.indexOf(token, start);
|
||||
if (idx === -1) return false;
|
||||
const afterIdx = idx + token.length;
|
||||
const afterChar = afterIdx < m.length ? m[afterIdx] : "";
|
||||
// Boundary: end-of-string or a `-` separator (not a digit or letter).
|
||||
if (afterChar === "" || afterChar === "-") return true;
|
||||
start = afterIdx;
|
||||
}
|
||||
}
|
||||
|
||||
/** Like gpt5TokenMatches but also rejects short -<1-3 digit> suffixes (e.g. -5, -10). */
|
||||
function gpt5BaseMatches(m: string, token: string): boolean {
|
||||
let start = 0;
|
||||
while (true) {
|
||||
const idx = m.indexOf(token, start);
|
||||
if (idx === -1) return false;
|
||||
const afterIdx = idx + token.length;
|
||||
const suffix = m.slice(afterIdx);
|
||||
if (suffix === "") return true;
|
||||
if (!suffix.startsWith("-")) {
|
||||
start = afterIdx;
|
||||
continue;
|
||||
}
|
||||
// Has a `-` suffix — check if it looks like a 1-3 digit version number.
|
||||
const dashRest = suffix.slice(1);
|
||||
if (/^\d{1,3}(?:[^a-z\d]|$)/i.test(dashRest)) {
|
||||
start = afterIdx;
|
||||
continue;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Returns true if the model string belongs to any GPT-5 family. */
|
||||
function gpt5FamilyModel(m: string): boolean {
|
||||
return (
|
||||
gpt5TokenMatches(m, "gpt-5-pro") ||
|
||||
gpt5TokenMatches(m, "gpt5-pro") ||
|
||||
gpt5TokenMatches(m, "gpt-5.6") ||
|
||||
gpt5TokenMatches(m, "gpt5.6") ||
|
||||
gpt5TokenMatches(m, "gpt-5-6") ||
|
||||
gpt5TokenMatches(m, "gpt5-6") ||
|
||||
gpt5TokenMatches(m, "gpt-5.5") ||
|
||||
gpt5TokenMatches(m, "gpt5.5") ||
|
||||
gpt5TokenMatches(m, "gpt-5.4") ||
|
||||
gpt5TokenMatches(m, "gpt5.4") ||
|
||||
gpt5TokenMatches(m, "gpt-5.1") ||
|
||||
gpt5TokenMatches(m, "gpt5.1") ||
|
||||
gpt5BaseMatches(m, "gpt-5") ||
|
||||
gpt5BaseMatches(m, "gpt5")
|
||||
);
|
||||
}
|
||||
|
||||
function openaiConfig(m: string): ProviderEffortConfig {
|
||||
// Check -pro before versioned suffixes (gpt-5-pro contains "gpt-5").
|
||||
if (gpt5TokenMatches(m, "gpt-5-pro") || gpt5TokenMatches(m, "gpt5-pro")) {
|
||||
return { validValues: ["high"], defaultValue: "high" };
|
||||
}
|
||||
if (
|
||||
gpt5TokenMatches(m, "gpt-5.6") ||
|
||||
gpt5TokenMatches(m, "gpt5.6") ||
|
||||
gpt5TokenMatches(m, "gpt-5-6") ||
|
||||
gpt5TokenMatches(m, "gpt5-6")
|
||||
) {
|
||||
return {
|
||||
validValues: ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
defaultValue: "medium",
|
||||
};
|
||||
}
|
||||
if (
|
||||
gpt5TokenMatches(m, "gpt-5.5") ||
|
||||
gpt5TokenMatches(m, "gpt5.5") ||
|
||||
gpt5TokenMatches(m, "gpt-5-5") ||
|
||||
gpt5TokenMatches(m, "gpt5-5") ||
|
||||
gpt5TokenMatches(m, "gpt-5.4") ||
|
||||
gpt5TokenMatches(m, "gpt5.4") ||
|
||||
gpt5TokenMatches(m, "gpt-5-4") ||
|
||||
gpt5TokenMatches(m, "gpt5-4")
|
||||
) {
|
||||
return {
|
||||
validValues: ["none", "low", "medium", "high", "xhigh"],
|
||||
defaultValue: "medium",
|
||||
};
|
||||
}
|
||||
if (
|
||||
gpt5TokenMatches(m, "gpt-5.1") ||
|
||||
gpt5TokenMatches(m, "gpt5.1") ||
|
||||
gpt5TokenMatches(m, "gpt-5-1") ||
|
||||
gpt5TokenMatches(m, "gpt5-1")
|
||||
) {
|
||||
return {
|
||||
validValues: ["none", "low", "medium", "high"],
|
||||
defaultValue: "none",
|
||||
};
|
||||
}
|
||||
if (gpt5BaseMatches(m, "gpt-5") || gpt5BaseMatches(m, "gpt5")) {
|
||||
return {
|
||||
validValues: ["minimal", "low", "medium", "high"],
|
||||
defaultValue: "medium",
|
||||
};
|
||||
}
|
||||
// Unknown OpenAI model — conservative fallback; max is enabled only for families whose table includes it.
|
||||
return {
|
||||
validValues: ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
defaultValue: "medium",
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true when the given runtime id is buzz-agent, which is the only
|
||||
* runtime that supports the tier-1 model-tuning knobs above.
|
||||
@@ -329,7 +90,3 @@ function openaiConfig(m: string): ProviderEffortConfig {
|
||||
export function isBuzzAgentRuntime(runtimeId: string): boolean {
|
||||
return runtimeId === "buzz-agent";
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Differential harness support
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -1,261 +0,0 @@
|
||||
[
|
||||
{
|
||||
"note": "Anthropic manual-budget: claude-3 family",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-3-7-sonnet-20250219",
|
||||
"validValues": ["low", "medium", "high"],
|
||||
"defaultValue": null
|
||||
},
|
||||
{
|
||||
"note": "Anthropic manual-budget: claude-opus-4-5",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-5",
|
||||
"validValues": ["low", "medium", "high"],
|
||||
"defaultValue": null
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-opus-4-7",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-7",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-opus-4-8",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-8",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-sonnet-5",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-sonnet-5-20260101",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-fable-5",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-fable-5",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-opus-5",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-5",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive xhigh-capable: claude-mythos-5",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-mythos-5",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive no-xhigh: claude-opus-4-6",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-6",
|
||||
"validValues": ["low", "medium", "high", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive no-xhigh: claude-sonnet-4-6",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-sonnet-4-6",
|
||||
"validValues": ["low", "medium", "high", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic adaptive no-xhigh: claude-mythos-preview",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-mythos-preview",
|
||||
"validValues": ["low", "medium", "high", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "Anthropic unknown model: blank \u2014 assume full adaptive",
|
||||
"provider": "anthropic",
|
||||
"model": "",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5-pro: high only",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5-pro",
|
||||
"validValues": ["high"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5.6: none/low/medium/high/xhigh/max",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.6",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5.5: none/low/medium/high/xhigh",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.5",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5.4: same table as gpt-5.5",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.4",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5.1: none/low/medium/high",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.1",
|
||||
"validValues": ["none", "low", "medium", "high"],
|
||||
"defaultValue": "none"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI gpt-5 base: minimal/low/medium/high",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5",
|
||||
"validValues": ["minimal", "low", "medium", "high"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI unknown model (gpt-4o): all-except-max",
|
||||
"provider": "openai",
|
||||
"model": "gpt-4o",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "OpenAI empty model: all-except-max",
|
||||
"provider": "openai",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 claude route (claude-opus-4-7): xhigh-capable anthropic table",
|
||||
"provider": "databricks_v2",
|
||||
"model": "claude-opus-4-7",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 claude route with databricks- prefix stripped",
|
||||
"provider": "databricks_v2",
|
||||
"model": "databricks-claude-opus-4-7",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 gpt-5.6-sol route: OpenAI max-capable table",
|
||||
"provider": "databricks_v2",
|
||||
"model": "gpt-5.6-sol",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 gpt-5-6-sol route: dashed OpenAI max-capable table",
|
||||
"provider": "databricks_v2",
|
||||
"model": "gpt-5-6-sol",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 gpt-5.4 route: OpenAI gpt-5.5/5.4 table",
|
||||
"provider": "databricks_v2",
|
||||
"model": "gpt-5.4",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 gpt-5.1 with databricks- prefix: OpenAI gpt-5.1 table",
|
||||
"provider": "databricks_v2",
|
||||
"model": "databricks-gpt-5.1",
|
||||
"validValues": ["none", "low", "medium", "high"],
|
||||
"defaultValue": "none"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 concrete non-claude non-gpt5 (llama-3): MLflow path, all-except-max",
|
||||
"provider": "databricks_v2",
|
||||
"model": "llama-3",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "DatabricksV2 blank model: route unknown, all-7",
|
||||
"provider": "databricks_v2",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "databricks v1: routes like openai unknown, all-except-max",
|
||||
"provider": "databricks",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "openai-compat: canonicalizes to openai, empty model → all-except-max with medium default",
|
||||
"provider": "openai-compat",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "openai-compat/gpt-5-pro: canonicalizes to openai, gpt-5-pro → [high] only",
|
||||
"provider": "openai-compat",
|
||||
"model": "gpt-5-pro",
|
||||
"validValues": ["high"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "openrouter: all-7 with medium default",
|
||||
"provider": "openrouter",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "empty provider: all-7 with medium default",
|
||||
"provider": "",
|
||||
"model": "",
|
||||
"validValues": ["none", "minimal", "low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "databricks_v2 goose-claude-fable-5: strips goose- prefix, routes anthropic adaptive+xhigh, max valid",
|
||||
"provider": "databricks_v2",
|
||||
"model": "goose-claude-fable-5",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "databricks_v2 goose-gpt-5.5: strips goose- prefix, routes openai gpt-5.5 table (none+low-xhigh, no minimal)",
|
||||
"provider": "databricks_v2",
|
||||
"model": "goose-gpt-5.5",
|
||||
"validValues": ["none", "low", "medium", "high", "xhigh"],
|
||||
"defaultValue": "medium"
|
||||
},
|
||||
{
|
||||
"note": "databricks_v2 goose-claude-sonnet-5: strips goose- prefix, routes anthropic adaptive+xhigh",
|
||||
"provider": "databricks_v2",
|
||||
"model": "goose-claude-sonnet-5",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
},
|
||||
{
|
||||
"note": "databricks_v2 arbitrary prefix team-x-claude-opus-4-7: strips to claude-opus-4-7, routes anthropic adaptive+xhigh, max valid",
|
||||
"provider": "databricks_v2",
|
||||
"model": "team-x-claude-opus-4-7",
|
||||
"validValues": ["low", "medium", "high", "xhigh", "max"],
|
||||
"defaultValue": "high"
|
||||
}
|
||||
]
|
||||
@@ -1,52 +0,0 @@
|
||||
/**
|
||||
* Effort-table sync guard: TS side.
|
||||
*
|
||||
* Loads the checked-in fixture and asserts that `getProviderEffortConfig`
|
||||
* matches every entry. Drift between `buzzAgentConfig.ts` and the fixture
|
||||
* (e.g. a new model family added to one side but not the other) fails CI.
|
||||
* The companion Rust test in `crates/buzz-agent/src/config.rs` mirrors
|
||||
* this check so both sides of the mirror must stay in sync.
|
||||
*/
|
||||
|
||||
import assert from "node:assert/strict";
|
||||
import { readFileSync } from "node:fs";
|
||||
import test from "node:test";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import path from "node:path";
|
||||
|
||||
import { getProviderEffortConfig } from "./buzzAgentConfig.ts";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const fixture = JSON.parse(
|
||||
readFileSync(path.join(__dirname, "effortTable.fixture.json"), "utf8"),
|
||||
);
|
||||
|
||||
for (const entry of fixture) {
|
||||
const {
|
||||
note,
|
||||
provider,
|
||||
model,
|
||||
validValues: expectedValidValues,
|
||||
defaultValue: expectedDefault,
|
||||
} = entry;
|
||||
const label = note ?? `${provider}/${model}`;
|
||||
|
||||
test(`effort fixture: ${label}`, () => {
|
||||
const { validValues, defaultValue } = getProviderEffortConfig(
|
||||
provider,
|
||||
model,
|
||||
);
|
||||
|
||||
assert.deepEqual(
|
||||
[...validValues],
|
||||
expectedValidValues,
|
||||
`validValues mismatch for "${label}"`,
|
||||
);
|
||||
|
||||
assert.equal(
|
||||
defaultValue,
|
||||
expectedDefault,
|
||||
`defaultValue mismatch for "${label}"`,
|
||||
);
|
||||
});
|
||||
}
|
||||
@@ -1,115 +0,0 @@
|
||||
# models.dev Reasoning Options Reconciliation Table
|
||||
|
||||
**Source queried**: https://models.dev/api.json (2026-07-31)<br>
|
||||
**Payload SHA-256**: `d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0`<br>
|
||||
**Policy (plan v4 §Behavior policy)**: models.dev `reasoning_options` become exact overrides.
|
||||
Each divergence from the current family rule result is reconciled here: either (a) adopted as an
|
||||
intentional correction or (b) rejected with a curation note.
|
||||
|
||||
**Verbatim source snapshot**: `scripts/catalog-sample-fixture.json` — verbatim `id`, `name`, and
|
||||
nested `reasoning_options` objects captured from the live API without transformation.
|
||||
Re-verify hash: `curl -s https://models.dev/api.json | sha256sum`
|
||||
|
||||
## Divergences
|
||||
|
||||
### `databricks-gpt-5-4-mini`
|
||||
|
||||
| | Current family rule (gpt5-4) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
**Rationale**: The Databricks AI Gateway v2 endpoint for `databricks-gpt-5-4-mini` explicitly
|
||||
advertises only `[low, medium, high]` in its `reasoning_options`. The family rule's `none` and
|
||||
`xhigh` are derived from the upstream OpenAI GPT-5.4 spec, which this Databricks endpoint does
|
||||
not expose. Provider-advertised wins per plan F1 policy.
|
||||
|
||||
**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-mini"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`<br>
|
||||
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-4-mini"`<br>
|
||||
**Test vector**: `resolver-exact-raw-id-hit` in `scripts/normative-corpus.json`
|
||||
|
||||
---
|
||||
|
||||
### `databricks-gpt-5-4-nano`
|
||||
|
||||
| | Current family rule (gpt5-4) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
**Rationale**: Same as `databricks-gpt-5-4-mini`. The nano variant exposes the same restricted
|
||||
effort set. Provider-advertised wins.
|
||||
|
||||
**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-nano"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`<br>
|
||||
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-4-nano"`
|
||||
|
||||
---
|
||||
|
||||
### `databricks-gpt-5-6-sol`
|
||||
|
||||
| | Current family rule (gpt5-6) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh, max]` | `[low, medium, high, max]` | **ADOPT** |
|
||||
|
||||
**Rationale**: The Databricks AI Gateway v2 endpoint for `databricks-gpt-5-6-sol` advertises only
|
||||
`[low, medium, high, max]` in its `reasoning_options`. The family rule's `none` and `xhigh` are
|
||||
derived from the upstream OpenAI GPT-5.6 spec, which this Databricks endpoint does not expose.
|
||||
Provider-advertised wins per plan F1 policy.
|
||||
|
||||
**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-6-sol"].reasoning_options = [{"type":"effort","values":["low","medium","high","max"]}]`<br>
|
||||
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-6-sol"`
|
||||
|
||||
---
|
||||
|
||||
### `databricks-gpt-5-5`
|
||||
|
||||
| | Current family rule (gpt5-5) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
**Rationale**: The Databricks AI Gateway v2 endpoint for `databricks-gpt-5-5` advertises only
|
||||
`[low, medium, high]` in its `reasoning_options`. The family rule's `none` and `xhigh` are
|
||||
derived from the upstream OpenAI GPT-5.5 spec, which this Databricks endpoint does not expose.
|
||||
Provider-advertised wins per plan F1 policy.
|
||||
|
||||
**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-5"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`<br>
|
||||
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-5"`
|
||||
|
||||
---
|
||||
|
||||
### `databricks-claude-opus-4-7`
|
||||
|
||||
| | Current family rule (anthropic-adaptive-xhigh-opus-4-7) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `reasoning_options` type | effort-based | `budget_tokens` | **NO EFFORT DIVERGENCE** |
|
||||
|
||||
**Rationale**: models.dev advertises `reasoning_options=[{"type":"budget_tokens","min":1024}]` —
|
||||
a different capability axis (extended thinking token budget), not an effort-level selector.
|
||||
There is no effort divergence to reconcile. The effort capabilities for this model come from the
|
||||
`anthropic-adaptive-xhigh-opus-4-7` family rule (Anthropic extended-thinking support table).
|
||||
|
||||
**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-claude-opus-4-7"].reasoning_options = [{"type":"budget_tokens","min":1024}]`<br>
|
||||
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-claude-opus-4-7"`
|
||||
|
||||
---
|
||||
|
||||
## Non-divergences (confirmed consistent)
|
||||
|
||||
The following models were checked against models.dev or provider docs and found consistent with
|
||||
the manifest family rules. No exact records needed.
|
||||
|
||||
| Model family | Source | Checked against | Status |
|
||||
|---|---|---|---|
|
||||
| `claude-opus-4-7` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-opus-4-8` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-sonnet-5.*` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-fable-5` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-mythos-5` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-opus-4-6` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-sonnet-4-6` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-mythos-preview` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `claude-3*` | [https://platform.claude.com/docs/en/build-with-claude/extended-thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) | Anthropic extended-thinking support table (July 2025) | ✓ Consistent |
|
||||
| `gpt-5-pro` | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
| `gpt-5.6` | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
| `gpt-5.5` | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
| `gpt-5.4` | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
| `gpt-5.1` | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
| `gpt-5` (base) | [https://platform.openai.com/docs/guides/reasoning](https://platform.openai.com/docs/guides/reasoning) | OpenAI reasoning guide (July 2025) | ✓ Consistent |
|
||||
@@ -1,11 +1,19 @@
|
||||
# Model Capabilities Manifest — Schema Reference
|
||||
# Model Capabilities Manifest
|
||||
|
||||
**Source of truth**: `scripts/model-capabilities.json`
|
||||
**Generator**: `scripts/generate-model-capabilities.mjs`
|
||||
**Source of truth**: `scripts/model-capabilities.json`
|
||||
**Generator**: `scripts/generate-model-capabilities.mjs`
|
||||
**Emitted artifacts**:
|
||||
- `crates/buzz-agent/src/generated_model_capabilities.rs`
|
||||
- `desktop/src/features/agents/ui/modelCapabilities.ts`
|
||||
- `scripts/generated-model-capabilities-coverage.json` (test fixture)
|
||||
|
||||
## How to regenerate
|
||||
|
||||
```sh
|
||||
node scripts/generate-model-capabilities.mjs
|
||||
```
|
||||
|
||||
CI regenerates and diffs on every PR that touches the manifest, generator, or generated
|
||||
files. Any stale generated file fails the `model-capabilities` job in `ci.yml`.
|
||||
|
||||
## Resolver contract (plan v4)
|
||||
|
||||
@@ -38,10 +46,10 @@ from multiple tiers.
|
||||
|
||||
| Value | Meaning |
|
||||
|-------|---------|
|
||||
| `manual-budget` | `thinking:{type:"enabled", budget_tokens}` — claude-3*, claude-opus-4-5 |
|
||||
| `adaptive` | `thinking:{type:"adaptive"}` + `output_config:{effort}` — opus-4-6+, sonnet-4-6+, etc. |
|
||||
| `omit-fields` | Unknown Anthropic model — omit thinking fields rather than guess request shape |
|
||||
| `none` | Non-Anthropic-routed model — thinking fields not applicable |
|
||||
| `manual-budget` | `thinking:{type:"enabled", budget_tokens}` -- claude-3*, claude-opus-4-5 |
|
||||
| `adaptive` | `thinking:{type:"adaptive"}` + `output_config:{effort}` -- opus-4-6+, sonnet-4-6+, etc. |
|
||||
| `omit-fields` | Unknown Anthropic model -- omit thinking fields rather than guess request shape |
|
||||
| `none` | Non-Anthropic-routed model -- thinking fields not applicable |
|
||||
| `not-applicable` | Provider does not use Anthropic thinking API |
|
||||
|
||||
### `databricks_v2_wire_route` values
|
||||
@@ -55,7 +63,7 @@ Transport for pure OpenAI, legacy Databricks, and OpenRouter is selected by `Ope
|
||||
| `openai-responses` | `/ai-gateway/openai/v1/responses` |
|
||||
| `anthropic-messages` | `/ai-gateway/anthropic/v1/messages` |
|
||||
| `mlflow-chat` | `/ai-gateway/mlflow/v1/chat/completions` |
|
||||
| `route-unknown` | DBv2 blank model — route not yet determinable |
|
||||
| `route-unknown` | DBv2 blank model -- route not yet determinable |
|
||||
| `not-applicable` | Not a DBv2 provider |
|
||||
|
||||
## Family rule match kinds
|
||||
@@ -84,20 +92,87 @@ divergence from family rule results is reconciled against provider docs and eith
|
||||
- (a) **adopted** as an intentional correction with its own test + exact record, or
|
||||
- (b) **rejected** with a curation note in the exact record.
|
||||
|
||||
See reconciliation table: `scripts/MODELS_DEV_RECONCILIATION.md`.
|
||||
### models.dev reconciliation table
|
||||
|
||||
**Source queried**: https://models.dev/api.json (2026-07-31)
|
||||
**Payload SHA-256**: `d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0`
|
||||
**Verbatim source snapshot SHA-256** (catalog-sample-fixture.json, deleted in Phase 3):
|
||||
`dc4092a04392f258bea65de2cef53cb1902dce1779dc2b1b2e21fb56774f2d78`
|
||||
|
||||
#### `databricks-gpt-5-4-mini`
|
||||
|
||||
| | Family rule (gpt5-4) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
Provider-advertised wins per plan F1 policy. Source: providers.databricks.models["databricks-gpt-5-4-mini"].reasoning_options
|
||||
(retrieved 2026-07-31).
|
||||
|
||||
#### `databricks-gpt-5-4-nano`
|
||||
|
||||
| | Family rule (gpt5-4) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
Same as `databricks-gpt-5-4-mini`.
|
||||
|
||||
#### `databricks-gpt-5-6-sol`
|
||||
|
||||
| | Family rule (gpt5-6) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh, max]` | `[low, medium, high, max]` | **ADOPT** |
|
||||
|
||||
Provider-advertised wins. Source: providers.databricks.models["databricks-gpt-5-6-sol"].reasoning_options
|
||||
(retrieved 2026-07-31).
|
||||
|
||||
#### `databricks-gpt-5-5`
|
||||
|
||||
| | Family rule (gpt5-5) | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `supported_efforts` | `[none, low, medium, high, xhigh]` | `[low, medium, high]` | **ADOPT** |
|
||||
|
||||
Provider-advertised wins. Source: providers.databricks.models["databricks-gpt-5-5"].reasoning_options
|
||||
(retrieved 2026-07-31).
|
||||
|
||||
#### `databricks-claude-opus-4-7`
|
||||
|
||||
| | Family rule | models.dev | Disposition |
|
||||
|---|---|---|---|
|
||||
| `reasoning_options` type | effort-based | `budget_tokens` | **NO EFFORT DIVERGENCE** |
|
||||
|
||||
models.dev advertises a different capability axis (extended thinking token budget), not an
|
||||
effort-level selector. No effort divergence to reconcile. Source: providers.databricks.models
|
||||
["databricks-claude-opus-4-7"].reasoning_options (retrieved 2026-07-31).
|
||||
|
||||
### Non-divergences (confirmed consistent)
|
||||
|
||||
| Model family | Source | Status |
|
||||
|---|---|---|
|
||||
| `claude-opus-4-7`, `claude-opus-4-8` | Anthropic extended-thinking docs (July 2025) | ok |
|
||||
| `claude-sonnet-5.*`, `claude-fable-5`, `claude-mythos-5` | Anthropic extended-thinking docs (July 2025) | ok |
|
||||
| `claude-opus-4-6`, `claude-sonnet-4-6`, `claude-mythos-preview` | Anthropic extended-thinking docs (July 2025) | ok |
|
||||
| `claude-3*` | Anthropic extended-thinking docs (July 2025) | ok |
|
||||
| `gpt-5-pro`, `gpt-5.6`, `gpt-5.5`, `gpt-5.4`, `gpt-5.1`, `gpt-5` | OpenAI reasoning guide (July 2025) | ok |
|
||||
|
||||
## Mutation evidence (historical record)
|
||||
|
||||
Mutation testing was run at Phase 2 completion (2026-07-31). 7 generator mutations were applied
|
||||
in isolation against both TS and Rust interpreters. All 7 were killed by both interpreters (7/7).
|
||||
The mutation runner (`scripts/run-mutation-evidence.mjs`) was deleted in Phase 3; the normative
|
||||
corpus (`scripts/normative-corpus.json`) that kills these mutations continues to run in CI.
|
||||
|
||||
## Adding a new model family
|
||||
|
||||
1. Add a `family_rules` entry with a new unique `id`, appropriate `match_kind`, `providers`,
|
||||
`match_priority`, and all capability axes.
|
||||
2. Run `node scripts/generate-model-capabilities.mjs` to regenerate artifacts.
|
||||
3. CI `model-capability-regen-diff` job verifies byte-clean regeneration.
|
||||
3. CI verifies byte-clean regeneration.
|
||||
4. The normative corpus (`scripts/normative-corpus.json`) may need new vectors.
|
||||
|
||||
## Adding an exact model override
|
||||
|
||||
1. Add an `exact_records` entry with `provider` + `raw_model_id` (the full raw ID, no prefix
|
||||
stripping). Include a `_reconciliation` note and doc citation.
|
||||
2. Run `node scripts/generate-model-capabilities.mjs` — completeness validator will fail if any
|
||||
2. Run `node scripts/generate-model-capabilities.mjs` -- completeness validator will fail if any
|
||||
axis cannot be resolved.
|
||||
3. Regenerate and commit.
|
||||
@@ -1,45 +0,0 @@
|
||||
# Model-Capability Manifest — Mutation Evidence
|
||||
|
||||
**Interpreter coverage**: both generated interpreters are exercised per mutation fault.
|
||||
- **TypeScript**: `scripts/run-corpus.mjs` imports `resolveModelCapabilities()` from
|
||||
`desktop/src/features/agents/ui/modelCapabilities.ts` via `--experimental-strip-types`.
|
||||
- **Rust**: `cargo test -p buzz-agent -- generated_model_capabilities::tests::shared_corpus_tests`
|
||||
deserializes and executes every vector in `scripts/normative-corpus.json` against
|
||||
`resolve_model_capabilities()`.
|
||||
|
||||
## How to reproduce
|
||||
|
||||
```sh
|
||||
# Runs generator mutations; exercises both TS and Rust interpreters per fault
|
||||
node --experimental-strip-types scripts/run-mutation-evidence.mjs
|
||||
|
||||
# Run interpreters independently:
|
||||
node --experimental-strip-types scripts/run-corpus.mjs
|
||||
cargo test -p buzz-agent -- generated_model_capabilities::tests::shared_corpus_tests
|
||||
```
|
||||
|
||||
## Mutation run results (both interpreters)
|
||||
|
||||
All 7 mutations applied in isolation; manifest restored after each run.
|
||||
Each mutation must be detected (killed) by **both** interpreters for it to count as covered.
|
||||
|
||||
| ID | Mutation | Expected killer(s) | TS | Rust |
|
||||
|----|----------|--------------------|----|------|
|
||||
| M1 | Reduce `claude-opus-4-7` `supported_efforts` to `[low,medium,high]` (drops xhigh+max) | `anthropic-claude-opus-4-7`, `dbv2-claude-prefix-stripped`, `dbv2-claude-route-anthropic-messages` | **killed ✓** | **killed ✓** |
|
||||
| M2 | Add `xhigh` to `gpt5-base` `supported_efforts` | `openai-gpt5-base`, `openai-gpt5-1106-should-not-match-base`, `openai-gpt5-4o-matches-base`, `openai-gpt5-date-suffix` | **killed ✓** | **killed ✓** |
|
||||
| M3 | Change `gpt5-1` `default_effort` to `"high"` instead of `"none"` | `openai-gpt5.1` | **killed ✓** | **killed ✓** |
|
||||
| M4 | Swap `dbv2-claude-code-names-segment` route from `anthropic-messages` to `openai-responses` | `dbv2-goose-opus-5-is-anthropic` | **killed ✓** | **killed ✓** |
|
||||
| M5 | Remove all three DBv2 segment rules | `dbv2-goose-opus-5-is-anthropic`, `dbv2-consolidated-llama-not-sol`, `dbv2-terraform-coder-not-terra` | **killed ✓** | **killed ✓** |
|
||||
| M6 | Change `databricks_v2` concrete-unknown fallback route from `mlflow-chat` to `openai-responses` | `dbv2-concrete-unknown-mlflow-no-max` | **killed ✓** | **killed ✓** |
|
||||
| M7 | Remove `xhigh` from `gpt5-4` `supported_efforts` | `resolver-prefixed-alias-misses-exact` | **killed ✓** | **killed ✓** |
|
||||
|
||||
**Summary: 7/7 mutations killed in both TS and Rust interpreters.**
|
||||
|
||||
## Coverage gaps
|
||||
|
||||
- Provider fallback mutations for `anthropic`, `openai`, `databricks`, `openrouter`, and
|
||||
`_default` are not individually mutated. These are covered by explicit fallback vectors
|
||||
in the corpus for `anthropic`, `openai`, and `databricks_v2`.
|
||||
- Rust mutations are run by recompiling the mutated generated file per fault (via `cargo
|
||||
test` after `node generate-model-capabilities.mjs`). Compile time is acceptable for
|
||||
offline mutation runs; CI only runs the already-compiled shared corpus harness.
|
||||
@@ -1,134 +0,0 @@
|
||||
{
|
||||
"_comment": "Verbatim models.dev snapshot for differential harness (plan v4 §Oracle). Contains exact records captured from the live API for exact-override entries. Verbatim: name and reasoning_options are reproduced without transformation.",
|
||||
"_source_url": "https://models.dev/api.json",
|
||||
"_retrieval_date": "2026-07-31",
|
||||
"_payload_sha256": "d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0",
|
||||
"_retrieval_note": "Full payload SHA-256 computed over the raw response body of GET https://models.dev/api.json (no transforms). Re-verify: curl -s https://models.dev/api.json | sha256sum",
|
||||
"_models_dev_records": {
|
||||
"databricks-gpt-5-5": {
|
||||
"id": "databricks-gpt-5-5",
|
||||
"name": "GPT-5.5",
|
||||
"reasoning_options": [
|
||||
{
|
||||
"type": "effort",
|
||||
"values": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"databricks-gpt-5-4-mini": {
|
||||
"id": "databricks-gpt-5-4-mini",
|
||||
"name": "GPT-5.4 mini",
|
||||
"reasoning_options": [
|
||||
{
|
||||
"type": "effort",
|
||||
"values": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"databricks-gpt-5-4-nano": {
|
||||
"id": "databricks-gpt-5-4-nano",
|
||||
"name": "GPT-5.4 nano",
|
||||
"reasoning_options": [
|
||||
{
|
||||
"type": "effort",
|
||||
"values": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"databricks-gpt-5-6-sol": {
|
||||
"id": "databricks-gpt-5-6-sol",
|
||||
"name": "GPT-5.6 Sol",
|
||||
"reasoning_options": [
|
||||
{
|
||||
"type": "effort",
|
||||
"values": [
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"max"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"databricks-claude-opus-4-7": {
|
||||
"id": "databricks-claude-opus-4-7",
|
||||
"name": "Claude Opus 4.7",
|
||||
"reasoning_options": [
|
||||
{
|
||||
"type": "budget_tokens",
|
||||
"min": 1024
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"endpoints": [
|
||||
{
|
||||
"name": "databricks-gpt-5-5",
|
||||
"note": "DATABRICKS_V2_KNOWN_MODELS entry; gpt5-5 family; openai-responses route"
|
||||
},
|
||||
{
|
||||
"name": "databricks-gpt-5-4-mini",
|
||||
"note": "exact record; models.dev override: low|medium|high (not family rule none+xhigh)"
|
||||
},
|
||||
{
|
||||
"name": "databricks-gpt-5-4-nano",
|
||||
"note": "exact record; models.dev override: low|medium|high"
|
||||
},
|
||||
{
|
||||
"name": "databricks-gpt-5-6-sol",
|
||||
"note": "exact record; models.dev source: low|medium|high|max (adopted as-is)"
|
||||
},
|
||||
{
|
||||
"name": "databricks-claude-opus-4-7",
|
||||
"note": "DATABRICKS_V2_KNOWN_MODELS entry; anthropic adaptive xhigh-capable; anthropic-messages route"
|
||||
},
|
||||
{
|
||||
"name": "goose-claude-fable-5",
|
||||
"note": "goose- prefix stripped; claude-fable-5 → anthropic adaptive xhigh-capable; anthropic-messages"
|
||||
},
|
||||
{
|
||||
"name": "goose-claude-sonnet-5-20260101",
|
||||
"note": "goose- prefix stripped; claude-sonnet-5 family; anthropic adaptive xhigh-capable"
|
||||
},
|
||||
{
|
||||
"name": "goose-opus-5",
|
||||
"note": "'opus' segment → anthropic-messages route; effort: fallback (prefix-stripped alias 'opus-5' not recognized Claude family)"
|
||||
},
|
||||
{
|
||||
"name": "consolidated-llama",
|
||||
"note": "segment test: 'sol' is substring of 'consolidated', NOT a segment → mlflow-chat"
|
||||
},
|
||||
{
|
||||
"name": "terraform-coder",
|
||||
"note": "segment test: 'terra' is prefix of 'terraform', NOT a segment → mlflow-chat"
|
||||
},
|
||||
{
|
||||
"name": "corpus-reranker",
|
||||
"note": "segment test: 'opus' is NOT a segment of 'corpus-reranker' → mlflow-chat"
|
||||
},
|
||||
{
|
||||
"name": "octopus-model",
|
||||
"note": "segment test: 'opus' is NOT a segment of 'octopus-model' → mlflow-chat"
|
||||
},
|
||||
{
|
||||
"name": "llama-3-70b",
|
||||
"note": "concrete non-Claude non-GPT → mlflow-chat; effort: all-except-max"
|
||||
},
|
||||
{
|
||||
"name": "",
|
||||
"note": "blank model → route-unknown; all 7 efforts; default medium"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -5,7 +5,6 @@
|
||||
* Reads `scripts/model-capabilities.json` and emits:
|
||||
* - `crates/buzz-agent/src/generated_model_capabilities.rs`
|
||||
* - `desktop/src/features/agents/ui/modelCapabilities.ts`
|
||||
* - `scripts/generated-model-capabilities-coverage.json` (snapshot/drift fixture — full-table resolver output, diff-checked by CI)
|
||||
*
|
||||
* The generator performs three ordered resolution steps (resolver contract, plan v4):
|
||||
* 1. Provider-qualified raw exact lookup — key is (provider, raw_model_id), matched
|
||||
@@ -570,68 +569,6 @@ function getProviderFallback(provider, isBlank) {
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Build full-table snapshot/drift fixture
|
||||
// Every (provider, model) pair that can be reached by any manifest rule is resolved
|
||||
// and written here. CI diffs this against the committed copy — any resolver output
|
||||
// change for any input shows up as a diff, catching silent behavior shifts.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const allEntries = [];
|
||||
|
||||
// All family rule canonical model IDs
|
||||
for (const rule of manifest.family_rules) {
|
||||
for (const provider of rule.providers) {
|
||||
const result = resolve(provider, rule.match_value);
|
||||
allEntries.push({
|
||||
note: `family rule ${rule.id} / provider ${provider}`,
|
||||
provider,
|
||||
model: rule.match_value,
|
||||
resolved: result,
|
||||
});
|
||||
// Also test aliases
|
||||
for (const alias of rule.match_aliases ?? []) {
|
||||
const r2 = resolve(provider, alias);
|
||||
allEntries.push({
|
||||
note: `family rule ${rule.id} alias ${alias} / provider ${provider}`,
|
||||
provider,
|
||||
model: alias,
|
||||
resolved: r2,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// All exact_records
|
||||
for (const rec of manifest.exact_records ?? []) {
|
||||
const result = resolve(rec.provider, rec.raw_model_id);
|
||||
allEntries.push({
|
||||
note: `exact record ${rec.provider}::${rec.raw_model_id}`,
|
||||
provider: rec.provider,
|
||||
model: rec.raw_model_id,
|
||||
resolved: result,
|
||||
});
|
||||
}
|
||||
|
||||
// All provider fallbacks (blank + concrete unknown examples)
|
||||
for (const [provider] of Object.entries(manifest.provider_fallbacks)) {
|
||||
if (provider === "_default") continue;
|
||||
const blankResult = resolve(provider, "");
|
||||
allEntries.push({
|
||||
note: `fallback ${provider} blank`,
|
||||
provider,
|
||||
model: "",
|
||||
resolved: blankResult,
|
||||
});
|
||||
const unknownResult = resolve(provider, "some-unknown-model-xyz");
|
||||
allEntries.push({
|
||||
note: `fallback ${provider} concrete_unknown`,
|
||||
provider,
|
||||
model: "some-unknown-model-xyz",
|
||||
resolved: unknownResult,
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rust code generation
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -1079,7 +1016,6 @@ const rustGpt5Helpers = `
|
||||
|
||||
/// Returns true if \`model\` contains \`token\` at a word boundary (end-of-string or "-").
|
||||
/// Does not match if followed immediately by a digit or letter.
|
||||
/// Mirrors gpt5_token_matches in config.rs.
|
||||
fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
|
||||
let lower = model;
|
||||
let tok_lower = token;
|
||||
@@ -1484,13 +1420,6 @@ const outputs = [
|
||||
content: tsContent,
|
||||
label: "TypeScript",
|
||||
},
|
||||
{
|
||||
path: outputDirOverride
|
||||
? join(outputDirOverride, "generated-model-capabilities-coverage.json")
|
||||
: join(repoRoot, "scripts", "generated-model-capabilities-coverage.json"),
|
||||
content: JSON.stringify(allEntries, null, 2) + "\n",
|
||||
label: "Coverage snapshot/drift fixture",
|
||||
},
|
||||
];
|
||||
|
||||
let checkFailed = false;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,239 +0,0 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Phase-2 differential harness — compare old buzzAgentConfig.ts effort logic with
|
||||
* the new generated modelCapabilities.ts interpreter over:
|
||||
* 1. The 36-entry effortTable.fixture.json (cross-boundary Rust/TS fixture)
|
||||
* 2. The 45-vector normative corpus (scripts/normative-corpus.json)
|
||||
* 3. The catalog-sample fixture (scripts/catalog-sample-fixture.json)
|
||||
*
|
||||
* Equality is required except for entries in the committed allowlist of intentional
|
||||
* F1 corrections (models.dev provider-capability reconciliations).
|
||||
*
|
||||
* Usage: node --experimental-strip-types scripts/run-differential.mjs [--verbose]
|
||||
* Exits 0 on all-pass (modulo allowlist), 1 on unexpected divergence or unexercised allowlist entry.
|
||||
*/
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { join, dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const repoRoot = join(__dirname, "..");
|
||||
const VERBOSE = process.argv.includes("--verbose");
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Import both interpreters
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// NEW: generated capability module
|
||||
const { resolveModelCapabilities: resolveNew } = await import(
|
||||
join(repoRoot, "desktop", "src", "features", "agents", "ui", "modelCapabilities.ts")
|
||||
);
|
||||
|
||||
// OLD: buzzAgentConfig.ts effort config
|
||||
const { getProviderEffortConfig_oldHandTable: getOldEffortConfig } = await import(
|
||||
join(repoRoot, "desktop", "src", "features", "agents", "ui", "buzzAgentConfig.ts")
|
||||
);
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Intentional corrections allowlist (Phase 1 F1 reconciliations)
|
||||
// Each entry: { provider, raw_model_id, reason }
|
||||
// ---------------------------------------------------------------------------
|
||||
const ALLOWLIST = [
|
||||
{
|
||||
provider: "databricks_v2",
|
||||
raw_model_id: "databricks-gpt-5-5",
|
||||
axes: ["supported_efforts"],
|
||||
reason: "Phase 1 ADOPT: models.dev d5a4974c advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]",
|
||||
},
|
||||
{
|
||||
provider: "databricks_v2",
|
||||
raw_model_id: "databricks-gpt-5-4-mini",
|
||||
axes: ["supported_efforts"],
|
||||
reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]",
|
||||
},
|
||||
{
|
||||
provider: "databricks_v2",
|
||||
raw_model_id: "databricks-gpt-5-4-nano",
|
||||
axes: ["supported_efforts"],
|
||||
reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]",
|
||||
},
|
||||
{
|
||||
provider: "databricks_v2",
|
||||
raw_model_id: "databricks-gpt-5-6-sol",
|
||||
axes: ["supported_efforts"],
|
||||
reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high,max]; old returns [none,low,medium,high,xhigh,max]",
|
||||
},
|
||||
{
|
||||
provider: "databricks_v2",
|
||||
raw_model_id: "goose-opus-5",
|
||||
axes: ["supported_efforts", "default_effort"],
|
||||
reason: "Phase 1 correction: 'opus' is a named DBv2 segment → anthropic-messages route; old config.rs disagreed with llm.rs (corpus note dbv2-goose-opus-5-is-anthropic). Generated adopts anthropic adaptive-xhigh capabilities consistent with the wire route.",
|
||||
},
|
||||
];
|
||||
|
||||
// Track which allowlist entries are actually exercised (suppressed a divergence).
|
||||
// Keyed as "provider:raw_model_id:axis".
|
||||
const allowlistHits = new Set();
|
||||
|
||||
function isAllowlisted(provider, rawModelId, axis) {
|
||||
const entry = ALLOWLIST.find(
|
||||
(e) =>
|
||||
e.provider === provider &&
|
||||
e.raw_model_id === rawModelId &&
|
||||
e.axes.includes(axis),
|
||||
);
|
||||
if (entry) {
|
||||
allowlistHits.add(`${provider}:${rawModelId}:${axis}`);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Comparison helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Compare effort axes from both interpreters for one (provider, model) pair.
|
||||
* Returns array of divergence objects.
|
||||
*/
|
||||
function compareEffortAxes(provider, model) {
|
||||
const newResult = resolveNew(provider, model);
|
||||
const oldResult = getOldEffortConfig(provider, model);
|
||||
|
||||
const divergences = [];
|
||||
|
||||
// supported_efforts
|
||||
const newEfforts = newResult.supportedEfforts ?? [];
|
||||
const oldEfforts = oldResult?.validValues ?? [];
|
||||
if (JSON.stringify(newEfforts) !== JSON.stringify(oldEfforts)) {
|
||||
if (!isAllowlisted(provider, model, "supported_efforts")) {
|
||||
divergences.push({
|
||||
axis: "supported_efforts",
|
||||
old: oldEfforts,
|
||||
new: newEfforts,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// default_effort
|
||||
const newDefault = newResult.defaultEffort ?? null;
|
||||
const oldDefault = oldResult?.defaultValue ?? null;
|
||||
if (newDefault !== oldDefault) {
|
||||
if (!isAllowlisted(provider, model, "default_effort")) {
|
||||
divergences.push({
|
||||
axis: "default_effort",
|
||||
old: oldDefault,
|
||||
new: newDefault,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return divergences;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test suites
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
let totalChecks = 0;
|
||||
let totalDivergences = 0;
|
||||
|
||||
function runCheck(label, provider, model) {
|
||||
totalChecks++;
|
||||
const divs = compareEffortAxes(provider, model);
|
||||
if (divs.length > 0) {
|
||||
totalDivergences += divs.length;
|
||||
for (const d of divs) {
|
||||
console.error(
|
||||
`DIVERGE [${label}] provider=${provider} model=${model} axis=${d.axis}\n` +
|
||||
` old: ${JSON.stringify(d.old)}\n` +
|
||||
` new: ${JSON.stringify(d.new)}`,
|
||||
);
|
||||
}
|
||||
} else if (VERBOSE) {
|
||||
console.log(`OK [${label}] provider=${provider} model=${model}`);
|
||||
}
|
||||
}
|
||||
|
||||
// 1. effortTable.fixture.json
|
||||
console.log("--- effortTable.fixture.json ---");
|
||||
const fixture = JSON.parse(
|
||||
readFileSync(
|
||||
join(repoRoot, "desktop", "src", "features", "agents", "ui", "effortTable.fixture.json"),
|
||||
"utf8",
|
||||
),
|
||||
);
|
||||
for (const entry of fixture) {
|
||||
if (!entry.provider) continue;
|
||||
runCheck("fixture", entry.provider, entry.model ?? "");
|
||||
}
|
||||
|
||||
// 2. normative-corpus.json (effort axes only)
|
||||
console.log("--- normative-corpus.json ---");
|
||||
const corpus = JSON.parse(
|
||||
readFileSync(join(repoRoot, "scripts", "normative-corpus.json"), "utf8"),
|
||||
);
|
||||
for (const entry of corpus) {
|
||||
if (entry._group) continue;
|
||||
if (!entry.provider || !entry.expect) continue;
|
||||
if (!entry.expect.supported_efforts && !entry.expect.default_effort) continue;
|
||||
runCheck("corpus", entry.provider, entry.raw_model_id ?? "");
|
||||
}
|
||||
|
||||
// 3. catalog-sample-fixture.json (exact records from pinned models.dev payload)
|
||||
console.log("--- catalog-sample-fixture.json ---");
|
||||
const catalogFixture = JSON.parse(
|
||||
readFileSync(join(repoRoot, "scripts", "catalog-sample-fixture.json"), "utf8"),
|
||||
);
|
||||
for (const ep of catalogFixture.endpoints ?? []) {
|
||||
if (!ep.name) continue;
|
||||
// All catalog endpoints are databricks_v2 provider
|
||||
runCheck("catalog-sample", "databricks_v2", ep.name);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Summary
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Count total allowlist axis slots expected to be hit
|
||||
const totalAllowlistSlots = ALLOWLIST.reduce((n, e) => n + e.axes.length, 0);
|
||||
const allowlistHitCount = allowlistHits.size;
|
||||
|
||||
// Detect stale allowlist entries (declared but never actually suppressed a divergence)
|
||||
const staleEntries = [];
|
||||
for (const entry of ALLOWLIST) {
|
||||
for (const axis of entry.axes) {
|
||||
const key = `${entry.provider}:${entry.raw_model_id}:${axis}`;
|
||||
if (!allowlistHits.has(key)) {
|
||||
staleEntries.push({ ...entry, axis });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log(
|
||||
`\nDifferential: ${totalChecks} checks, ${totalDivergences} unexpected divergences, ${allowlistHitCount}/${totalAllowlistSlots} allowlist slots exercised`,
|
||||
);
|
||||
|
||||
if (staleEntries.length > 0) {
|
||||
for (const e of staleEntries) {
|
||||
console.error(
|
||||
`STALE_ALLOWLIST provider=${e.provider} model=${e.raw_model_id} axis=${e.axis} — entry never fired; remove or update it`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (totalDivergences > 0) {
|
||||
console.error(
|
||||
`FAIL: ${totalDivergences} unexpected divergence(s) — see output above`,
|
||||
);
|
||||
process.exit(1);
|
||||
} else if (staleEntries.length > 0) {
|
||||
console.error(
|
||||
`FAIL: ${staleEntries.length} stale allowlist entry(ies) — entries that never suppress a divergence mask future regressions`,
|
||||
);
|
||||
process.exit(1);
|
||||
} else {
|
||||
console.log("PASS: old and new effort logic agree on all non-allowlisted entries");
|
||||
}
|
||||
@@ -1,261 +0,0 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Per-interpreter mutation evidence runner.
|
||||
*
|
||||
* Introduces deliberate resolver faults into the manifest, regenerates artifacts,
|
||||
* and verifies the shared normative corpus detects every fault in BOTH the generated
|
||||
* TypeScript interpreter (via --experimental-strip-types import) and the Rust
|
||||
* interpreter (via cargo test shared-corpus harness). Exits 0 if all mutations are
|
||||
* killed in both interpreters; exits 1 if any survive.
|
||||
*
|
||||
* Usage:
|
||||
* node --experimental-strip-types scripts/run-mutation-evidence.mjs [--verbose]
|
||||
*
|
||||
* This script is non-CI (run manually to generate MUTATION_EVIDENCE.md). It writes
|
||||
* its findings to stdout in a format suitable for copy-paste into the evidence doc.
|
||||
*
|
||||
* Mutations applied (each in isolation, manifest restored after each run):
|
||||
* M1: Swap anthropic-adaptive-xhigh-opus-4-7 efforts from [low,medium,high,xhigh,max]
|
||||
* to [low,medium,high] — kills corpus vectors that check xhigh/max.
|
||||
* M2: Change gpt5-base supported_efforts to include "xhigh" — kills vectors that
|
||||
* check gpt5-base resolves minimal-only, not xhigh.
|
||||
* M3: Change openai-gpt5-1 default_effort to "high" instead of "none" — kills
|
||||
* the gpt5.1 corpus vector that checks default_effort=none.
|
||||
* M4: Swap databricks_v2_wire_route in dbv2-claude-code-names-segment from
|
||||
* "anthropic-messages" to "openai-responses" — kills segment-route corpus vectors.
|
||||
* M5: Remove all three DBv2 segment rules — kills goose-opus-5 and terraform/consolidated
|
||||
* segment collision vectors.
|
||||
* M6: Change databricks_v2 concrete_unknown fallback route from "mlflow-chat" to
|
||||
* "openai-responses" — kills dbv2-concrete-unknown-mlflow-no-max vector.
|
||||
* M7: Change gpt5-4 supported_efforts to remove "xhigh" — kills
|
||||
* resolver-prefixed-alias-misses-exact vector.
|
||||
*/
|
||||
|
||||
import { readFileSync, writeFileSync } from "node:fs";
|
||||
import { join, dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { execSync, spawnSync } from "node:child_process";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const repoRoot = join(__dirname, "..");
|
||||
const VERBOSE = process.argv.includes("--verbose");
|
||||
|
||||
const manifestPath = join(repoRoot, "scripts", "model-capabilities.json");
|
||||
const generatorPath = join(repoRoot, "scripts", "generate-model-capabilities.mjs");
|
||||
const jsRunnerPath = join(repoRoot, "scripts", "run-corpus.mjs");
|
||||
|
||||
const originalManifest = readFileSync(manifestPath, "utf8");
|
||||
|
||||
/**
|
||||
* Run the TS corpus and return:
|
||||
* { kind: "passed" } — all vectors pass (mutation survived)
|
||||
* { kind: "killed", output } — nonzero exit AND at least one expectedKiller ID
|
||||
* appears in stdout/stderr ("FAIL <id>" line)
|
||||
* { kind: "error", output } — nonzero exit but NO expected corpus output
|
||||
* (import error, missing file, syntax error, etc.)
|
||||
*/
|
||||
function runTsCorpus(expectedKillers) {
|
||||
let result;
|
||||
try {
|
||||
result = spawnSync(
|
||||
process.execPath,
|
||||
["--experimental-strip-types", jsRunnerPath],
|
||||
{ cwd: repoRoot, encoding: "utf8" },
|
||||
);
|
||||
} catch (e) {
|
||||
return { kind: "error", output: `spawn error: ${e.message}` };
|
||||
}
|
||||
if (result.status === 0) return { kind: "passed" };
|
||||
const output = (result.stdout ?? "") + (result.stderr ?? "");
|
||||
// A genuine corpus kill produces "FAIL <vector-id>" lines.
|
||||
// An infrastructure failure (import error, syntax error) produces no such lines.
|
||||
const hasCorpusFailure = expectedKillers.some((id) => output.includes(`FAIL ${id}`));
|
||||
if (hasCorpusFailure) return { kind: "killed", output };
|
||||
return { kind: "error", output };
|
||||
}
|
||||
|
||||
/**
|
||||
* Run the Rust corpus and return:
|
||||
* { kind: "passed" } — all vectors pass
|
||||
* { kind: "killed", output } — nonzero exit AND at least one expectedKiller ID
|
||||
* appears in the panic output ("[<id>]" format)
|
||||
* { kind: "error", output } — nonzero exit but NO expected corpus output
|
||||
* (compile error, missing cargo, linker error, etc.)
|
||||
*/
|
||||
function runRustCorpus(expectedKillers) {
|
||||
let result;
|
||||
try {
|
||||
result = spawnSync(
|
||||
"cargo",
|
||||
[
|
||||
"test",
|
||||
"-p", "buzz-agent",
|
||||
"--",
|
||||
"generated_model_capabilities::tests::shared_corpus_tests",
|
||||
"--nocapture",
|
||||
],
|
||||
{ cwd: repoRoot, encoding: "utf8", env: { ...process.env, RUST_BACKTRACE: "0" } },
|
||||
);
|
||||
} catch (e) {
|
||||
return { kind: "error", output: `spawn error: ${e.message}` };
|
||||
}
|
||||
if (result.status === 0) return { kind: "passed" };
|
||||
const output = (result.stdout ?? "") + (result.stderr ?? "");
|
||||
// The Rust corpus runner panics with "[<vector-id>] <axis>: got..." messages.
|
||||
const hasCorpusFailure = expectedKillers.some((id) => output.includes(`[${id}]`));
|
||||
if (hasCorpusFailure) return { kind: "killed", output };
|
||||
return { kind: "error", output };
|
||||
}
|
||||
|
||||
function regen() {
|
||||
execSync(`"${process.execPath}" "${generatorPath}"`, {
|
||||
cwd: repoRoot,
|
||||
stdio: VERBOSE ? "inherit" : "pipe",
|
||||
});
|
||||
}
|
||||
|
||||
function restore() {
|
||||
writeFileSync(manifestPath, originalManifest, "utf8");
|
||||
}
|
||||
|
||||
function applyMutation(mutFn) {
|
||||
const manifest = JSON.parse(originalManifest);
|
||||
mutFn(manifest);
|
||||
writeFileSync(manifestPath, JSON.stringify(manifest, null, 2) + "\n", "utf8");
|
||||
}
|
||||
|
||||
const mutations = [
|
||||
{
|
||||
id: "M1",
|
||||
description: "Reduce claude-opus-4-7 supported_efforts to [low,medium,high] (drops xhigh+max)",
|
||||
expectedKillers: ["anthropic-claude-opus-4-7", "dbv2-claude-prefix-stripped", "dbv2-claude-route-anthropic-messages"],
|
||||
mutate(manifest) {
|
||||
const rule = manifest.family_rules.find(r => r.id === "anthropic-adaptive-xhigh-opus-4-7");
|
||||
rule.supported_efforts = ["low", "medium", "high"];
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M2",
|
||||
description: "Add xhigh to gpt5-base supported_efforts [minimal,low,medium,high,xhigh]",
|
||||
expectedKillers: ["openai-gpt5-base", "openai-gpt5-1106-should-not-match-base", "openai-gpt5-4o-matches-base", "openai-gpt5-date-suffix"],
|
||||
mutate(manifest) {
|
||||
const rule = manifest.family_rules.find(r => r.id === "openai-gpt5-base");
|
||||
rule.supported_efforts = ["minimal", "low", "medium", "high", "xhigh"];
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M3",
|
||||
description: "Change gpt5-1 default_effort to 'high' instead of 'none'",
|
||||
expectedKillers: ["openai-gpt5.1"],
|
||||
mutate(manifest) {
|
||||
const rule = manifest.family_rules.find(r => r.id === "openai-gpt5-1");
|
||||
rule.default_effort = "high";
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M4",
|
||||
description: "Swap dbv2-claude-code-names-segment route from anthropic-messages to openai-responses",
|
||||
expectedKillers: ["dbv2-goose-opus-5-is-anthropic"],
|
||||
mutate(manifest) {
|
||||
const rule = manifest.family_rules.find(r => r.id === "dbv2-claude-code-names-segment");
|
||||
rule.databricks_v2_wire_route = "openai-responses";
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M5",
|
||||
description: "Remove all three DBv2 segment rules (dbv2-claude-code-names-segment, dbv2-gpt-code-names-segment, dbv2-sol-luna-terra-segment)",
|
||||
expectedKillers: ["dbv2-goose-opus-5-is-anthropic", "dbv2-consolidated-llama-not-sol", "dbv2-terraform-coder-not-terra"],
|
||||
mutate(manifest) {
|
||||
manifest.family_rules = manifest.family_rules.filter(
|
||||
r => !["dbv2-claude-code-names-segment", "dbv2-gpt-code-names-segment", "dbv2-sol-luna-terra-segment"].includes(r.id)
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M6",
|
||||
description: "Change databricks_v2 concrete_unknown fallback route from mlflow-chat to openai-responses",
|
||||
expectedKillers: ["dbv2-concrete-unknown-mlflow-no-max"],
|
||||
mutate(manifest) {
|
||||
manifest.provider_fallbacks.databricks_v2.concrete_unknown.databricks_v2_wire_route = "openai-responses";
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "M7",
|
||||
description: "Remove xhigh from gpt5-4 supported_efforts [none,low,medium,high]",
|
||||
expectedKillers: ["resolver-prefixed-alias-misses-exact"],
|
||||
mutate(manifest) {
|
||||
const rule = manifest.family_rules.find(r => r.id === "openai-gpt5-4");
|
||||
rule.supported_efforts = ["none", "low", "medium", "high"];
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
let allKilled = true;
|
||||
const results = [];
|
||||
|
||||
for (const mut of mutations) {
|
||||
process.stdout.write(` ${mut.id}: ${mut.description}\n`);
|
||||
try {
|
||||
applyMutation(mut.mutate);
|
||||
regen();
|
||||
|
||||
// TS interpreter
|
||||
process.stdout.write(` TS ... `);
|
||||
const tsResult = runTsCorpus(mut.expectedKillers);
|
||||
const tsKilled = tsResult.kind === "killed";
|
||||
const tsError = tsResult.kind === "error";
|
||||
if (tsKilled) {
|
||||
process.stdout.write("killed ✓\n");
|
||||
} else if (tsError) {
|
||||
process.stdout.write(`ERROR (infrastructure failure — not a corpus kill)\n`);
|
||||
if (VERBOSE) process.stdout.write(` ${tsResult.output}\n`);
|
||||
} else {
|
||||
process.stdout.write("SURVIVED ✗\n");
|
||||
if (VERBOSE) process.stdout.write(` ${tsResult.output ?? ""}\n`);
|
||||
}
|
||||
|
||||
// Rust interpreter
|
||||
process.stdout.write(` Rust... `);
|
||||
const rustResult = runRustCorpus(mut.expectedKillers);
|
||||
const rustKilled = rustResult.kind === "killed";
|
||||
const rustError = rustResult.kind === "error";
|
||||
if (rustKilled) {
|
||||
process.stdout.write("killed ✓\n");
|
||||
} else if (rustError) {
|
||||
process.stdout.write(`ERROR (infrastructure failure — not a corpus kill)\n`);
|
||||
if (VERBOSE) process.stdout.write(` ${rustResult.output}\n`);
|
||||
} else {
|
||||
process.stdout.write("SURVIVED ✗\n");
|
||||
if (VERBOSE) process.stdout.write(` ${rustResult.output ?? ""}\n`);
|
||||
}
|
||||
|
||||
const killed = tsKilled && rustKilled;
|
||||
if (!killed) allKilled = false;
|
||||
results.push({ ...mut, killed, tsKilled, rustKilled, tsError, rustError });
|
||||
} catch (e) {
|
||||
process.stdout.write(` ERROR: ${e.message}\n`);
|
||||
allKilled = false;
|
||||
results.push({ ...mut, killed: false, tsKilled: false, rustKilled: false, tsError: true, rustError: true, output: e.message });
|
||||
} finally {
|
||||
restore();
|
||||
regen(); // restore generated files
|
||||
}
|
||||
}
|
||||
|
||||
console.log("");
|
||||
const killed = results.filter(r => r.killed).length;
|
||||
const errored = results.filter(r => r.tsError || r.rustError).length;
|
||||
console.log(`Mutation results: ${killed}/${results.length} killed (both interpreters)` +
|
||||
(errored > 0 ? `, ${errored} ERROR (infrastructure failure — see output above)` : ""));
|
||||
|
||||
if (!allKilled) {
|
||||
const hasErrors = results.some(r => r.tsError || r.rustError);
|
||||
if (hasErrors) {
|
||||
console.error("ERROR: Infrastructure failures prevented some mutations from being verified as killed.");
|
||||
console.error(" Run with --verbose to see the full output for ERROR entries.");
|
||||
}
|
||||
console.error("ERROR: Some mutations survived — corpus does not kill all resolver faults.");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log("All mutations killed in both TS and Rust interpreters.");
|
||||
Reference in New Issue
Block a user