diff --git a/crates/buzz-agent/src/catalog.rs b/crates/buzz-agent/src/catalog.rs index 659cbd76f..d9ba11632 100644 --- a/crates/buzz-agent/src/catalog.rs +++ b/crates/buzz-agent/src/catalog.rs @@ -47,8 +47,10 @@ pub struct ModelEntry { /// Known Databricks AI Gateway v2 models — used as a fallback when the /// `api/ai-gateway/v2/endpoints` call returns an empty list. /// Mirrors goose's `DATABRICKS_V2_KNOWN_MODELS`. -pub const DATABRICKS_V2_KNOWN_MODELS: &[&str] = - &["databricks-gpt-5-5", "databricks-claude-opus-4-7"]; +/// +/// Phase 2 cutover: this is now a re-export of the generated constant in +/// `generated_model_capabilities`. Phase 3 removes the old hand-maintained list. +pub use crate::generated_model_capabilities::DATABRICKS_V2_KNOWN_MODELS; /// Returns the discovery-failure fallback catalog for a Databricks provider. /// diff --git a/crates/buzz-agent/src/config.rs b/crates/buzz-agent/src/config.rs index a0e64f1a9..2ad2c8be0 100644 --- a/crates/buzz-agent/src/config.rs +++ b/crates/buzz-agent/src/config.rs @@ -2738,6 +2738,69 @@ mod tests { } } + /// Phase-2 differential: new generated effort config matches old hand-coded helper + /// for every entry in effortTable.fixture.json. This gate ensures Phase 2 cutover + /// is behavior-preserving except where the allowlist explicitly covers a correction. + #[test] + fn effort_table_fixture_differential_old_vs_new() { + use crate::generated_model_capabilities::resolve_model_capabilities; + + // Intentional corrections: models where the generated capability deliberately + // diverges from the old implementation. Each entry must cite its source. + // + // "databricks_v2/databricks-gpt-5-5": Phase 1 ADOPT — models.dev payload + // d5a4974c advertises [low,medium,high]; old code returns [none,low,medium,high,xhigh]. + // Provider-advertised wins per plan F1. + // "databricks_v2/databricks-gpt-5-4-mini": Phase 1 ADOPT — models.dev advertises + // [low,medium,high]; old code returns [none,low,medium,high,xhigh]. + // "databricks_v2/databricks-gpt-5-4-nano": Phase 1 ADOPT — same as mini. + // "databricks_v2/databricks-gpt-5-6-sol": Phase 1 ADOPT — models.dev advertises + // [low,medium,high,max]; old code returns [none,low,medium,high,xhigh,max]. + let allowlist: &[(&str, &str)] = &[ + ("databricks_v2", "databricks-gpt-5-5"), + ("databricks_v2", "databricks-gpt-5-4-mini"), + ("databricks_v2", "databricks-gpt-5-4-nano"), + ("databricks_v2", "databricks-gpt-5-6-sol"), + ]; + + let fixture_json = + include_str!("../../../desktop/src/features/agents/ui/effortTable.fixture.json"); + let entries: Vec = + serde_json::from_str(fixture_json).expect("fixture must be valid JSON"); + + for entry in &entries { + let label = entry.note.as_deref().unwrap_or(entry.model.as_str()); + let in_allowlist = allowlist + .iter() + .any(|(p, m)| *p == entry.provider && *m == entry.model); + + let old_result = valid_effort_values_for_provider_model(&entry.provider, &entry.model); + + // Build new result from generated module. + let cap = resolve_model_capabilities(&entry.provider, &entry.model); + let new_values: Vec<&'static str> = cap + .supported_efforts + .iter() + .map(|e| e.openai_effort_str()) + .collect(); + let new_default: Option<&'static str> = + cap.default_effort.map(|e| e.openai_effort_str()); + let new_result = (new_values, new_default); + + if in_allowlist { + // Intentional divergence — skip equality check. + continue; + } + + assert_eq!( + old_result, new_result, + "effort differential divergence for fixture entry \"{label}\" \ + (provider={}, model={}): old={old_result:?} new={new_result:?}", + entry.provider, entry.model, + ); + } + } + #[test] fn resolve_provider_openrouter_with_key() { assert_eq!( diff --git a/crates/buzz-agent/src/llm.rs b/crates/buzz-agent/src/llm.rs index f595a165e..0b33b260e 100644 --- a/crates/buzz-agent/src/llm.rs +++ b/crates/buzz-agent/src/llm.rs @@ -1095,25 +1095,24 @@ fn is_responses_required_error(body: &str) -> bool { || b.contains("use the responses api") } -/// OpenAI-family code names that appear as their own segment in a Databricks v2 -/// endpoint name (the GPT-5 launch aliases). The `gpt` family itself is matched -/// separately by segment prefix so `gpt`, `gpt5`, and the `gpt` of a split -/// `gpt-5` all qualify. -const DATABRICKS_V2_OPENAI_CODE_NAMES: &[&str] = &["sol", "luna", "terra"]; +/// OpenAI-family code names used by the OLD segment-based route classifier. +/// Preserved for the Phase-2 differential harness and Phase-3 cleanup. +/// Production routing now delegates to `resolve_model_capabilities` (see +/// `databricks_v2_route_for_model` below). +#[cfg(test)] +const _OLD_DATABRICKS_V2_OPENAI_CODE_NAMES: &[&str] = &["sol", "luna", "terra"]; -/// Anthropic (Claude) family and release code names that appear as their own -/// segment in a Databricks v2 endpoint name — the `claude` prefix, the family -/// names (`opus`, `sonnet`, `haiku`), and the release code names (`mythos`, -/// `fable`). Getting a Claude model onto the Anthropic Messages route is what -/// lets it carry a `cache_control` breakpoint; an endpoint that matches none of -/// these falls through to the MLflow (OpenAI-wire) path, where Anthropic prompt -/// caching is structurally impossible and the discount is silently lost. -const DATABRICKS_V2_CLAUDE_NAMES: &[&str] = +/// Anthropic (Claude) family and release code names used by the OLD classifier. +/// Preserved for the Phase-2 differential harness and Phase-3 cleanup. +#[cfg(test)] +const _OLD_DATABRICKS_V2_CLAUDE_NAMES: &[&str] = &["claude", "opus", "sonnet", "haiku", "mythos", "fable"]; /// Split a Databricks v2 endpoint name into its lowercase alphanumeric segments, /// breaking on any non-alphanumeric delimiter (`-`, `_`, `.`, `/`, …). E.g. /// `Databricks-Claude-Opus-5` -> `["databricks", "claude", "opus", "5"]`. +/// Used by the old classifier (differential harness). Phase 3 removes this. +#[cfg(test)] fn model_name_segments(model: &str) -> Vec { model .split(|c: char| !c.is_ascii_alphanumeric()) @@ -1122,32 +1121,48 @@ fn model_name_segments(model: &str) -> Vec { .collect() } -fn databricks_v2_route_for_model(model: &str) -> DatabricksV2Route { - // The v2 catalog exposes no family field, so the wire format is inferred - // from the endpoint name. Discovery deliberately keeps arbitrary custom - // aliases, so we match whole name *segments* rather than raw substrings: a - // substring test would misroute unrelated names — `consolidated-llama` - // (`sol`), `terraform-coder` (`terra`), `corpus-reranker`/`octopus-model` - // (`opus`) — onto a wire whose request shape their backend can't parse, - // turning a caching optimization into a hard request/parse failure. Segment - // matching still accepts real prefixed names like `goose-opus-5`. +/// OLD segment-based route classifier — preserved for the Phase-2 differential +/// harness. Production routing now delegates to `databricks_v2_route_for_model`. +/// Phase 3 removes this function. +#[cfg(test)] +fn _old_databricks_v2_route_for_model(model: &str) -> DatabricksV2Route { let segments = model_name_segments(model); let has_named_segment = |names: &[&str]| segments.iter().any(|seg| names.contains(&seg.as_str())); - // `gpt` family: any segment beginning with `gpt` — covers `gpt`, `gpt5`, and - // the `gpt` segment of a split `gpt-5`, without matching mid-word. let is_gpt_family = segments.iter().any(|seg| seg.starts_with("gpt")); - // OpenAI is checked before Claude so a name carrying both markers resolves - // to the OpenAI wire (preserving the prior `gpt-5`-first precedence). - if is_gpt_family || has_named_segment(DATABRICKS_V2_OPENAI_CODE_NAMES) { + if is_gpt_family || has_named_segment(_OLD_DATABRICKS_V2_OPENAI_CODE_NAMES) { DatabricksV2Route::OpenAiResponses - } else if has_named_segment(DATABRICKS_V2_CLAUDE_NAMES) { + } else if has_named_segment(_OLD_DATABRICKS_V2_CLAUDE_NAMES) { DatabricksV2Route::AnthropicMessages } else { DatabricksV2Route::MlflowChatCompletions } } +/// Returns the Databricks v2 wire route for a model name. +/// +/// Phase 2 cutover: delegates to `resolve_model_capabilities` from the generated +/// capability module. The generated resolver uses the same segment-based matching +/// logic, now derived from the manifest single source of truth. +/// +/// `RouteUnknown` (blank model) and `NotApplicable` (non-DBv2 provider) are not +/// reachable here — this function is only called for DBv2 requests with an +/// effective model string — both map to `MlflowChatCompletions` as a safe fallback. +fn databricks_v2_route_for_model(model: &str) -> DatabricksV2Route { + use crate::generated_model_capabilities::{ + resolve_model_capabilities, DatabricksV2Route as GenRoute, + }; + match resolve_model_capabilities("databricks_v2", model).databricks_v2_wire_route { + GenRoute::OpenAiResponses => DatabricksV2Route::OpenAiResponses, + GenRoute::AnthropicMessages => DatabricksV2Route::AnthropicMessages, + // RouteUnknown (blank model) and NotApplicable (non-DBv2) are structurally + // unreachable from this call site; fall through to the mlflow path. + GenRoute::MlflowChatCompletions | GenRoute::RouteUnknown | GenRoute::NotApplicable => { + DatabricksV2Route::MlflowChatCompletions + } + } +} + fn databricks_v2_path(route: DatabricksV2Route) -> &'static str { match route { DatabricksV2Route::OpenAiResponses => "/ai-gateway/openai/v1/responses", @@ -3335,6 +3350,51 @@ mod tests { } } + /// Phase-2 differential: new generated route matches old segment-based route + /// for all models in the existing test suite. Documents where they agree and + /// flags unexpected divergence. Any intentional divergence must be added to + /// the allowlist in this test. + #[test] + fn databricks_v2_route_differential_old_vs_new() { + // Models where old and new are intentionally expected to differ. + // (Empty: the generated manifest uses identical segment logic.) + let allowlist: &[&str] = &[]; + + for model in [ + "databricks-gpt-5-5", + "gpt-4o", + "gpt5", + "databricks-gpt-5-6-luna", + "databricks-gpt-5-6-sol", + "databricks-terra", + "databricks-claude-opus-4-7", + "goose-opus-5", + "databricks-sonnet-5", + "databricks-haiku-4-5", + "databricks-mythos-5", + "databricks-fable-5", + "Databricks-Claude-Opus-5", + "custom-tool-model", + "databricks-gemini-3-pro", + "consolidated-llama", + "terraform-coder", + "corpus-reranker", + "octopus-model", + "", + ] { + let old_route = _old_databricks_v2_route_for_model(model); + let new_route = databricks_v2_route_for_model(model); + if allowlist.contains(&model) { + // Intentional divergence — just document, don't assert equality. + continue; + } + assert_eq!( + old_route, new_route, + "route divergence for model={model:?}: old={old_route:?} new={new_route:?}" + ); + } + } + #[test] fn parse_responses_rejects_malformed_function_arguments() { let v = serde_json::json!({ diff --git a/scripts/MODELS_DEV_RECONCILIATION.md b/scripts/MODELS_DEV_RECONCILIATION.md index 3fff89a9c..59722e97e 100644 --- a/scripts/MODELS_DEV_RECONCILIATION.md +++ b/scripts/MODELS_DEV_RECONCILIATION.md @@ -1,7 +1,7 @@ # models.dev Reasoning Options Reconciliation Table -**Source queried**: https://models.dev/api.json (2026-07-31) -**Payload SHA-256**: `d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0` +**Source queried**: https://models.dev/api.json (2026-07-31)
+**Payload SHA-256**: `d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0`
**Policy (plan v4 §Behavior policy)**: models.dev `reasoning_options` become exact overrides. Each divergence from the current family rule result is reconciled here: either (a) adopted as an intentional correction or (b) rejected with a curation note. @@ -23,8 +23,8 @@ advertises only `[low, medium, high]` in its `reasoning_options`. The family rul `xhigh` are derived from the upstream OpenAI GPT-5.4 spec, which this Databricks endpoint does not expose. Provider-advertised wins per plan F1 policy. -**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-mini"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]` -**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-4-mini"` +**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-mini"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`
+**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-4-mini"`
**Test vector**: `resolver-exact-raw-id-hit` in `scripts/normative-corpus.json` --- @@ -38,7 +38,7 @@ not expose. Provider-advertised wins per plan F1 policy. **Rationale**: Same as `databricks-gpt-5-4-mini`. The nano variant exposes the same restricted effort set. Provider-advertised wins. -**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-nano"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]` +**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-4-nano"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-4-nano"` --- @@ -54,7 +54,7 @@ effort set. Provider-advertised wins. derived from the upstream OpenAI GPT-5.6 spec, which this Databricks endpoint does not expose. Provider-advertised wins per plan F1 policy. -**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-6-sol"].reasoning_options = [{"type":"effort","values":["low","medium","high","max"]}]` +**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-6-sol"].reasoning_options = [{"type":"effort","values":["low","medium","high","max"]}]`
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-6-sol"` --- @@ -70,7 +70,7 @@ Provider-advertised wins per plan F1 policy. derived from the upstream OpenAI GPT-5.5 spec, which this Databricks endpoint does not expose. Provider-advertised wins per plan F1 policy. -**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-5"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]` +**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-gpt-5-5"].reasoning_options = [{"type":"effort","values":["low","medium","high"]}]`
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-gpt-5-5"` --- @@ -86,7 +86,7 @@ a different capability axis (extended thinking token budget), not an effort-leve There is no effort divergence to reconcile. The effort capabilities for this model come from the `anthropic-adaptive-xhigh-opus-4-7` family rule (Anthropic extended-thinking support table). -**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-claude-opus-4-7"].reasoning_options = [{"type":"budget_tokens","min":1024}]` +**Source**: [https://models.dev/api.json](https://models.dev/api.json) — retrieved 2026-07-31; `providers.databricks.models["databricks-claude-opus-4-7"].reasoning_options = [{"type":"budget_tokens","min":1024}]`
**Snapshot**: `scripts/catalog-sample-fixture.json` key `"databricks-claude-opus-4-7"` --- diff --git a/scripts/run-differential.mjs b/scripts/run-differential.mjs new file mode 100755 index 000000000..c1d9fe63e --- /dev/null +++ b/scripts/run-differential.mjs @@ -0,0 +1,201 @@ +#!/usr/bin/env node +/** + * Phase-2 differential harness — compare old buzzAgentConfig.ts effort logic with + * the new generated modelCapabilities.ts interpreter over: + * 1. The 36-entry effortTable.fixture.json (cross-boundary Rust/TS fixture) + * 2. The 45-vector normative corpus (scripts/normative-corpus.json) + * 3. The catalog-sample fixture (scripts/catalog-sample-fixture.json) + * + * Equality is required except for entries in the committed allowlist of intentional + * F1 corrections (models.dev provider-capability reconciliations). + * + * Usage: node --experimental-strip-types scripts/run-differential.mjs [--verbose] + * Exits 0 on all-pass (modulo allowlist), 1 on unexpected divergence. + */ + +import { readFileSync } from "node:fs"; +import { join, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const repoRoot = join(__dirname, ".."); +const VERBOSE = process.argv.includes("--verbose"); + +// --------------------------------------------------------------------------- +// Import both interpreters +// --------------------------------------------------------------------------- + +// NEW: generated capability module +const { resolveModelCapabilities: resolveNew } = await import( + join(repoRoot, "desktop", "src", "features", "agents", "ui", "modelCapabilities.ts") +); + +// OLD: buzzAgentConfig.ts effort config +const { getProviderEffortConfig: getOldEffortConfig } = await import( + join(repoRoot, "desktop", "src", "features", "agents", "ui", "buzzAgentConfig.ts") +); + +// --------------------------------------------------------------------------- +// Intentional corrections allowlist (Phase 1 F1 reconciliations) +// Each entry: { provider, raw_model_id, reason } +// --------------------------------------------------------------------------- +const ALLOWLIST = [ + { + provider: "databricks_v2", + raw_model_id: "databricks-gpt-5-5", + axes: ["supported_efforts"], + reason: "Phase 1 ADOPT: models.dev d5a4974c advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]", + }, + { + provider: "databricks_v2", + raw_model_id: "databricks-gpt-5-4-mini", + axes: ["supported_efforts"], + reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]", + }, + { + provider: "databricks_v2", + raw_model_id: "databricks-gpt-5-4-nano", + axes: ["supported_efforts"], + reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high]; old returns [none,low,medium,high,xhigh]", + }, + { + provider: "databricks_v2", + raw_model_id: "databricks-gpt-5-6-sol", + axes: ["supported_efforts"], + reason: "Phase 1 ADOPT: models.dev advertises [low,medium,high,max]; old returns [none,low,medium,high,xhigh,max]", + }, + { + provider: "databricks_v2", + raw_model_id: "goose-opus-5", + axes: ["supported_efforts", "default_effort"], + reason: "Phase 1 correction: 'opus' is a named DBv2 segment → anthropic-messages route; old config.rs disagreed with llm.rs (corpus note dbv2-goose-opus-5-is-anthropic). Generated adopts anthropic adaptive-xhigh capabilities consistent with the wire route.", + }, +]; + +function isAllowlisted(provider, rawModelId, axis) { + return ALLOWLIST.some( + (e) => + e.provider === provider && + e.raw_model_id === rawModelId && + e.axes.includes(axis), + ); +} + +// --------------------------------------------------------------------------- +// Comparison helpers +// --------------------------------------------------------------------------- + +/** + * Compare effort axes from both interpreters for one (provider, model) pair. + * Returns array of divergence objects. + */ +function compareEffortAxes(provider, model) { + const newResult = resolveNew(provider, model); + const oldResult = getOldEffortConfig(provider, model); + + const divergences = []; + + // supported_efforts + const newEfforts = newResult.supportedEfforts ?? []; + const oldEfforts = oldResult?.validValues ?? []; + if (JSON.stringify(newEfforts) !== JSON.stringify(oldEfforts)) { + if (!isAllowlisted(provider, model, "supported_efforts")) { + divergences.push({ + axis: "supported_efforts", + old: oldEfforts, + new: newEfforts, + }); + } + } + + // default_effort + const newDefault = newResult.defaultEffort ?? null; + const oldDefault = oldResult?.defaultValue ?? null; + if (newDefault !== oldDefault) { + if (!isAllowlisted(provider, model, "default_effort")) { + divergences.push({ + axis: "default_effort", + old: oldDefault, + new: newDefault, + }); + } + } + + return divergences; +} + +// --------------------------------------------------------------------------- +// Test suites +// --------------------------------------------------------------------------- + +let totalChecks = 0; +let totalDivergences = 0; +let totalAllowlisted = 0; + +function runCheck(label, provider, model) { + totalChecks++; + const divs = compareEffortAxes(provider, model); + if (divs.length > 0) { + totalDivergences += divs.length; + for (const d of divs) { + console.error( + `DIVERGE [${label}] provider=${provider} model=${model} axis=${d.axis}\n` + + ` old: ${JSON.stringify(d.old)}\n` + + ` new: ${JSON.stringify(d.new)}`, + ); + } + } else if (VERBOSE) { + console.log(`OK [${label}] provider=${provider} model=${model}`); + } +} + +// 1. effortTable.fixture.json +console.log("--- effortTable.fixture.json ---"); +const fixture = JSON.parse( + readFileSync( + join(repoRoot, "desktop", "src", "features", "agents", "ui", "effortTable.fixture.json"), + "utf8", + ), +); +for (const entry of fixture) { + if (!entry.provider) continue; + runCheck("fixture", entry.provider, entry.model ?? ""); +} + +// 2. normative-corpus.json (effort axes only) +console.log("--- normative-corpus.json ---"); +const corpus = JSON.parse( + readFileSync(join(repoRoot, "scripts", "normative-corpus.json"), "utf8"), +); +for (const entry of corpus) { + if (entry._group) continue; + if (!entry.provider || !entry.expect) continue; + if (!entry.expect.supported_efforts && !entry.expect.default_effort) continue; + runCheck("corpus", entry.provider, entry.raw_model_id ?? ""); +} + +// 3. catalog-sample-fixture.json (exact records from pinned models.dev payload) +console.log("--- catalog-sample-fixture.json ---"); +const catalogFixture = JSON.parse( + readFileSync(join(repoRoot, "scripts", "catalog-sample-fixture.json"), "utf8"), +); +for (const ep of catalogFixture.endpoints ?? []) { + if (!ep.name) continue; + // All catalog endpoints are databricks_v2 provider + runCheck("catalog-sample", "databricks_v2", ep.name); +} + +// --------------------------------------------------------------------------- +// Summary +// --------------------------------------------------------------------------- +console.log( + `\nDifferential: ${totalChecks} checks, ${totalDivergences} unexpected divergences, ${totalAllowlisted} allowlisted`, +); +if (totalDivergences > 0) { + console.error( + `FAIL: ${totalDivergences} unexpected divergence(s) — see output above`, + ); + process.exit(1); +} else { + console.log("PASS: old and new effort logic agree on all non-allowlisted entries"); +}