From 60fc24b0c6f604cf85cb74727f3b5f403da04a39 Mon Sep 17 00:00:00 2001 From: npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7 Date: Tue, 4 Aug 2026 15:30:06 -0400 Subject: [PATCH] =?UTF-8?q?fix(models):=20round-3=20corrective=20pass=20?= =?UTF-8?q?=E2=80=94=20boundary=20strip,=20gpt-version-segment,=206-axis?= =?UTF-8?q?=20corpus,=20validation=20symmetry,=20prototype-leak?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - stripCatalogPrefix boundary-aligned: customgpt/sgpt/mygpt no longer false-strip to gpt-* forms; boundary check requires position 0 or non-alphanumeric precursor - New match_kind gpt-version-segment: gpt-neox-20b routes mlflow-chat (was residual collision); gpt segment match requires next segment to start with digit or be dashless gpt5 form - PROVIDER_FALLBACKS Record→Map: constructor/__proto__ prototype-key leak closed; resolveModelCapabilities/getProviderEffortConfig return complete records for all provider strings - Corpus 69→77 vectors, all 6-axis mandatory; JS and Rust runners hard-fail on missing axes; prototype-key vectors added; boundary-negative vectors added (sgpt-5-5, mygpt-5, customgpt-5-5-endpoint all route mlflow-chat); gpt-neox-20b pinned as mlflow-chat (corrected behavior, not residual collision) - Manifest validator 33→42 tests: family match_priority integer guard, lowercase duplicate-key detection, post-inheritance materialized default∈supported check, assertEfforts shared helper covering family/fallback/exact - Clippy: map_or(false)→is_some_and in emitter template; fmt clean - Desktop regression tests: constructor/__proto__ prototype-key stability Co-authored-by: Will Pfleger Signed-off-by: Will Pfleger --- .../src/generated_model_capabilities.rs | 83 +- .../src/generated_model_capabilities_tests.rs | 63 +- .../agents/ui/buzzAgentConfig.test.mjs | 37 + .../features/agents/ui/modelCapabilities.ts | 63 +- scripts/generate-model-capabilities.mjs | 249 ++++-- scripts/model-capabilities.json | 4 +- scripts/normative-corpus.json | 745 +++++++++++++++--- scripts/run-corpus.mjs | 18 + scripts/test-manifest-validator.mjs | 125 +++ 9 files changed, 1163 insertions(+), 224 deletions(-) diff --git a/crates/buzz-agent/src/generated_model_capabilities.rs b/crates/buzz-agent/src/generated_model_capabilities.rs index 7afcf2e4f..782aaa006 100644 --- a/crates/buzz-agent/src/generated_model_capabilities.rs +++ b/crates/buzz-agent/src/generated_model_capabilities.rs @@ -258,18 +258,48 @@ pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option &str { const FAMILY_TOKENS: &[&str] = &["claude-", "gpt-"]; let lower = model.to_ascii_lowercase(); - let first_idx = FAMILY_TOKENS.iter().filter_map(|tok| lower.find(tok)).min(); + let mut first_idx: Option = None; + for tok in FAMILY_TOKENS { + let tok_bytes = tok.as_bytes(); + let lower_bytes = lower.as_bytes(); + let mut start = 0usize; + loop { + match lower_bytes[start..] + .windows(tok_bytes.len()) + .position(|w| w == tok_bytes) + { + None => break, + Some(rel) => { + let idx = start + rel; + // Boundary check: position 0 or preceded by a non-alphanumeric byte + let at_boundary = idx == 0 || { + let prev = lower_bytes[idx - 1]; + !prev.is_ascii_alphanumeric() + }; + if at_boundary { + first_idx = Some(first_idx.map_or(idx, |f| f.min(idx))); + break; + } + start = idx + 1; + } + } + } + } match first_idx { Some(idx) => &model[idx..], None => model, @@ -1004,12 +1034,8 @@ pub fn lookup_by_family_rules(provider: &str, normalized: &str) -> Option bool { let first_non_digit = dash_rest .find(|c: char| !c.is_ascii_digit()) .unwrap_or(dash_rest.len()); - let is_short_version = first_non_digit >= 1 - && first_non_digit <= 3 + let is_short_version = (1..=3).contains(&first_non_digit) && (first_non_digit == dash_rest.len() || !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric()); if is_short_version { @@ -1462,6 +1487,36 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { } } +// --------------------------------------------------------------------------- +// gpt-version-segment helper (used by generated family resolver) +// --------------------------------------------------------------------------- + +/// gpt-version-segment match: token is an exact segment AND (token itself starts with a digit, +/// OR the next segment after it starts with a digit). Prevents "gpt-neox-20b" from matching +/// "gpt" because "neox" starts with a letter, while "gpt-5.5" matches because next seg "5" is digit. +fn gpt_version_segment_matches_rs(model: &str, token: &str) -> bool { + let segs: Vec<&str> = model.split(|c: char| !c.is_ascii_alphanumeric()).collect(); + for (i, seg) in segs.iter().enumerate() { + if *seg == token { + // Dashless numeric form (e.g. "gpt5"): token length > 3 or starts with digit + if token.len() > 3 || token.as_bytes().first().is_some_and(|b| b.is_ascii_digit()) { + return true; + } + // For short alpha tokens like "gpt": require the next segment to start with a digit + if let Some(next_seg) = segs.get(i + 1) { + if next_seg + .as_bytes() + .first() + .is_some_and(|b| b.is_ascii_digit()) + { + return true; + } + } + } + } + false +} + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- diff --git a/crates/buzz-agent/src/generated_model_capabilities_tests.rs b/crates/buzz-agent/src/generated_model_capabilities_tests.rs index 12392997c..49f1a19de 100644 --- a/crates/buzz-agent/src/generated_model_capabilities_tests.rs +++ b/crates/buzz-agent/src/generated_model_capabilities_tests.rs @@ -40,14 +40,16 @@ mod shared_corpus_tests { expect: Option, } + /// All six axes are required on every corpus vector. + /// A missing axis is a schema error that hard-fails the test. #[derive(Deserialize)] struct CorpusExpect { - thinking_mode: Option, - supported_efforts: Option>, - default_effort: Option, // string or null - databricks_v2_wire_route: Option, - normalization_policy: Option, - registry_label: Option, // string or null + thinking_mode: String, + supported_efforts: Vec, + default_effort: serde_json::Value, // string or null + databricks_v2_wire_route: String, + normalization_policy: String, + registry_label: serde_json::Value, // string or null } // --------------------------------------------------------------------------- @@ -162,9 +164,10 @@ mod shared_corpus_tests { let result = resolve_model_capabilities(canonical_provider, raw_model_id); ran += 1; - // Check thinking_mode if present in expect - if let Some(expected_mode) = &expect.thinking_mode { - let expected = parse_thinking_mode(expected_mode); + // All six axes are required — no optional field checks. + // thinking_mode + { + let expected = parse_thinking_mode(&expect.thinking_mode); if result.thinking_mode != expected { failures.push(format!( "[{id}] thinking_mode: got {:?}, expected {:?}", @@ -173,10 +176,13 @@ mod shared_corpus_tests { } } - // Check supported_efforts if present - if let Some(expected_efforts) = &expect.supported_efforts { - let expected: Vec = - expected_efforts.iter().map(|s| parse_effort(s)).collect(); + // supported_efforts + { + let expected: Vec = expect + .supported_efforts + .iter() + .map(|s| parse_effort(s)) + .collect(); let actual: Vec = result.supported_efforts.iter().cloned().collect(); if actual != expected { @@ -186,9 +192,9 @@ mod shared_corpus_tests { } } - // Check default_effort if present - if let Some(expected_de) = &expect.default_effort { - let expected_parsed = match expected_de { + // default_effort (string or null) + { + let expected_parsed: Option = match &expect.default_effort { serde_json::Value::Null => None, serde_json::Value::String(s) => Some(parse_effort(s)), other => { @@ -203,9 +209,9 @@ mod shared_corpus_tests { } } - // Check databricks_v2_wire_route if present - if let Some(expected_route) = &expect.databricks_v2_wire_route { - let expected = parse_route(expected_route); + // databricks_v2_wire_route + { + let expected = parse_route(&expect.databricks_v2_wire_route); if result.databricks_v2_wire_route != expected { failures.push(format!( "[{id}] databricks_v2_wire_route: got {:?}, expected {:?}", @@ -214,9 +220,9 @@ mod shared_corpus_tests { } } - // Check normalization_policy if present - if let Some(expected_policy) = &expect.normalization_policy { - let expected = parse_normalization_policy(expected_policy); + // normalization_policy + { + let expected = parse_normalization_policy(&expect.normalization_policy); if result.normalization_policy != expected { failures.push(format!( "[{id}] normalization_policy: got {:?}, expected {:?}", @@ -225,14 +231,14 @@ mod shared_corpus_tests { } } - // Check registry_label if present (JSON string or null) - if let Some(expected_rl) = &expect.registry_label { - let expected_opt: Option<&str> = match expected_rl { + // registry_label (string or null) + { + let expected_opt: Option<&str> = match &expect.registry_label { serde_json::Value::Null => None, serde_json::Value::String(s) => Some(s.as_str()), - other => panic!( - "unexpected registry_label value in corpus vector {id}: {other:?}" - ), + other => { + panic!("unexpected registry_label value in corpus vector {id}: {other:?}") + } }; if result.registry_label != expected_opt { failures.push(format!( @@ -241,7 +247,6 @@ mod shared_corpus_tests { )); } } - } if !failures.is_empty() { diff --git a/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs b/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs index e3730fe43..31945a69e 100644 --- a/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs +++ b/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs @@ -703,3 +703,40 @@ test("openai-compat alias handles mixed case + whitespace: ' OpenAI-Compat ' c assert.deepEqual([...messy.validValues], [...canonical.validValues]); assert.equal(messy.defaultValue, canonical.defaultValue); }); + + +// --------------------------------------------------------------------------- +// PROVIDER_FALLBACKS prototype-key safety regression +// --------------------------------------------------------------------------- + +test("prototype-key safety: provider='constructor' returns a complete record (Map prevents prototype leak)", () => { + const result = getProviderEffortConfig("constructor", "some-model"); + // Must not crash and must return a non-empty validValues (complete record) + assert.ok(Array.isArray(result.validValues) && result.validValues.length > 0, + "constructor provider must return a complete record (not an empty/broken result from prototype chain)"); +}); + +test("prototype-key safety: provider='__proto__' returns a complete record", () => { + const result = getProviderEffortConfig("__proto__", "some-model"); + assert.ok(Array.isArray(result.validValues) && result.validValues.length > 0, + "__proto__ provider must return a complete record"); +}); + +test("prototype-key safety: provider='hasOwnProperty' returns a complete record", () => { + const result = getProviderEffortConfig("hasOwnProperty", "some-model"); + assert.ok(Array.isArray(result.validValues) && result.validValues.length > 0, + "hasOwnProperty provider must return a complete record"); +}); + +test("prototype-key safety: provider='toString' returns a complete record", () => { + const result = getProviderEffortConfig("toString", "some-model"); + assert.ok(Array.isArray(result.validValues) && result.validValues.length > 0, + "toString provider must return a complete record"); +}); + +test("prototype-key safety: validValues.includes() does not throw for prototype-key provider", () => { + // This exercises the crash path: EffortSelectField calls validValues.includes() + const result = getProviderEffortConfig("constructor", "some-model"); + assert.doesNotThrow(() => result.validValues.includes("low"), + "validValues.includes() must not throw for prototype-key providers"); +}); diff --git a/desktop/src/features/agents/ui/modelCapabilities.ts b/desktop/src/features/agents/ui/modelCapabilities.ts index b4831166d..db982618f 100644 --- a/desktop/src/features/agents/ui/modelCapabilities.ts +++ b/desktop/src/features/agents/ui/modelCapabilities.ts @@ -122,6 +122,25 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean { } } +/** + * gpt-version-segment match: token is an exact segment AND (token itself starts with a digit, + * OR the next segment after it starts with a digit). Prevents "gpt-neox-20b" from matching + * "gpt" because "neox" starts with a letter, while "gpt-5.5" matches because next seg "5" is digit. + */ +function gptVersionSegmentMatchesGenerated(m: string, token: string): boolean { + const segs = m.split(/[^a-z0-9]+/); + for (let i = 0; i < segs.length; i++) { + if (segs[i] === token) { + // Dashless numeric form (e.g. "gpt5") or token itself starts with digit + if (token.length > 3 || /^\d/.test(token)) return true; + // For "gpt": require the next segment to start with a digit + const nextSeg = segs[i + 1]; + if (nextSeg !== undefined && /^\d/.test(nextSeg)) return true; + } + } + return false; +} + // --------------------------------------------------------------------------- // Exact records — provider-qualified, pre-prefix-stripping // --------------------------------------------------------------------------- @@ -189,8 +208,8 @@ const EXACT_RECORDS = new Map([ // Provider fallbacks // --------------------------------------------------------------------------- -const PROVIDER_FALLBACKS: Record = { - "anthropic": { +const PROVIDER_FALLBACKS = new Map([ + ["anthropic", { blank: { registryLabel: null, thinkingMode: "adaptive", @@ -207,8 +226,8 @@ const PROVIDER_FALLBACKS: Record VALID_EFFORTS.indexOf(e)); + for (let i = 1; i < indices.length; i++) { + if (indices[i] <= indices[i - 1]) { + throw new Error( + `${label}: must follow canonical order [${VALID_EFFORTS.join(", ")}]; got [${efforts.join(", ")}]`, + ); + } + } +} + function validateFallbackRecord(rec, label) { assertEnum(rec.databricks_v2_wire_route, VALID_DBV2_ROUTES, `${label}.databricks_v2_wire_route`); assertEnum(rec.thinking_mode, VALID_THINKING_MODES, `${label}.thinking_mode`); - assertNonEmpty(rec.supported_efforts, `${label}.supported_efforts`); - for (const e of rec.supported_efforts) { - assertEnum(e, VALID_EFFORTS, `${label}.supported_efforts[]`); - } + assertEfforts(rec.supported_efforts, `${label}.supported_efforts`); if (rec.default_effort !== null) { assertEnum(rec.default_effort, VALID_EFFORTS, `${label}.default_effort`); if (!rec.supported_efforts.includes(rec.default_effort)) { @@ -334,10 +354,11 @@ for (const rule of manifest.family_rules) { seenRuleIds.add(rule.id); assertEnum(rule.match_kind, VALID_MATCH_KINDS, `rule ${rule.id} match_kind`); assertEnum(rule.thinking_mode, VALID_THINKING_MODES, `rule ${rule.id} thinking_mode`); - assertNonEmpty(rule.supported_efforts, `rule ${rule.id} supported_efforts`); - for (const e of rule.supported_efforts) { - assertEnum(e, VALID_EFFORTS, `rule ${rule.id} supported_efforts[]`); + // match_priority must be a non-negative integer (interpolated into Rust comments and TS code) + if (!Number.isInteger(rule.match_priority) || rule.match_priority < 0) { + throw new Error(`rule ${rule.id}: match_priority must be a non-negative integer, got ${JSON.stringify(rule.match_priority)}`); } + assertEfforts(rule.supported_efforts, `rule ${rule.id} supported_efforts`); if (rule.default_effort !== null) { assertEnum(rule.default_effort, VALID_EFFORTS, `rule ${rule.id} default_effort`); if (!rule.supported_efforts.includes(rule.default_effort)) { @@ -368,39 +389,25 @@ for (const [provider, fb] of Object.entries(manifest.provider_fallbacks)) { } // Validate exact_records — full invariant checks on every override axis +// Keys are normalized to lowercase before duplicate detection to match build-time key lowercasing. const seenExactKeys = new Set(); for (const rec of manifest.exact_records ?? []) { if (!rec.provider || !rec.raw_model_id) throw new Error("exact_record missing provider or raw_model_id"); - const key = `${rec.provider}::${rec.raw_model_id}`; + // Normalize to lowercase — both emitters lowercase provider+raw_model_id at build time + const key = `${rec.provider.toLowerCase()}::${rec.raw_model_id.toLowerCase()}`; if (seenExactKeys.has(key)) throw new Error(`duplicate exact_record key: ${key}`); seenExactKeys.add(key); - // Validate match_priority: must be a non-negative integer if present (injected into comments). + // Validate match_priority: must be a non-negative integer if present (interpolated into source). if (rec.match_priority !== undefined) { if (!Number.isInteger(rec.match_priority) || rec.match_priority < 0) throw new Error(`exact_record ${key}: match_priority must be a non-negative integer`); } - // Validate supported_efforts_override if present. + // Validate supported_efforts_override if present (shared validator: non-empty, no dupes, canonical order). if (rec.supported_efforts_override !== undefined) { - assertNonEmpty(rec.supported_efforts_override, `exact_record ${key} supported_efforts_override`); - const seenEfforts = new Set(); - for (const e of rec.supported_efforts_override) { - assertEnum(e, VALID_EFFORTS, `exact_record ${key} supported_efforts_override[]`); - if (seenEfforts.has(e)) - throw new Error(`exact_record ${key}: duplicate effort "${e}" in supported_efforts_override`); - seenEfforts.add(e); - } - // Validate ordering matches canonical effort order (Rust clamp assumes sorted). - const canonicalIndices = rec.supported_efforts_override.map((e) => VALID_EFFORTS.indexOf(e)); - for (let i = 1; i < canonicalIndices.length; i++) { - if (canonicalIndices[i] <= canonicalIndices[i - 1]) { - throw new Error( - `exact_record ${key}: supported_efforts_override must follow canonical order [${VALID_EFFORTS.join(", ")}]; got [${rec.supported_efforts_override.join(", ")}]`, - ); - } - } + assertEfforts(rec.supported_efforts_override, `exact_record ${key} supported_efforts_override`); // Validate default_effort is in the override if present. if (rec.default_effort !== undefined && rec.default_effort !== null) { assertEnum(rec.default_effort, VALID_EFFORTS, `exact_record ${key} default_effort`); @@ -431,23 +438,53 @@ for (const rec of manifest.exact_records ?? []) { } } +// Post-inheritance materialized validation for exact_records. +// An exact_record with no supported_efforts_override inherits the family default — validate +// that the final materialized default_effort is within the final materialized supported_efforts. +for (const rec of manifest.exact_records ?? []) { + const key = `${rec.provider.toLowerCase()}::${rec.raw_model_id.toLowerCase()}`; + const materializedResult = resolve(rec.provider, rec.raw_model_id); + const matEfforts = materializedResult.supported_efforts; + const matDefault = materializedResult.default_effort; + if (matDefault !== undefined && matDefault !== null) { + if (!matEfforts.includes(matDefault)) { + throw new Error( + `exact_record ${key}: materialized default_effort "${matDefault}" not in materialized supported_efforts [${matEfforts.join(", ")}] (check family default inheritance)`, + ); + } + } +} + // --------------------------------------------------------------------------- // Resolution engine (mirrors plan resolver contract) // --------------------------------------------------------------------------- /** * Strip catalog prefix to get the normalized alias for family-rule matching. - * Finds the first occurrence of a known family token and returns from there. - * e.g. "goose-claude-fable-5" → "claude-fable-5" - * "databricks-gpt-5.5" → "gpt-5.5" - * "claude-opus-4-7" → "claude-opus-4-7" (no prefix) + * Finds the first boundary-aligned occurrence of a known family token and returns from there. + * Boundary-aligned: the token must start at position 0 or be preceded by a non-alphanumeric char. + * This prevents "customgpt-5-5-endpoint" from stripping to "gpt-5-5-endpoint" via "gpt-" inside + * the "customgpt-" prefix. + * e.g. "goose-claude-fable-5" → "claude-fable-5" (boundary: preceded by "-") + * "databricks-gpt-5.5" → "gpt-5.5" (boundary: preceded by "-") + * "claude-opus-4-7" → "claude-opus-4-7" (no prefix, already at boundary) + * "customgpt-5-5-ep" → "customgpt-5-5-ep" (no boundary match: 'g' preceded by 'm') */ function stripCatalogPrefix(model) { const lower = model.toLowerCase(); let firstIdx = Infinity; for (const tok of manifest.family_tokens) { - const idx = lower.indexOf(tok); - if (idx !== -1 && idx < firstIdx) firstIdx = idx; + let start = 0; + while (true) { + const idx = lower.indexOf(tok, start); + if (idx === -1) break; + // Boundary check: position 0 or preceded by a non-alphanumeric character + if (idx === 0 || !/[a-z0-9]/.test(lower[idx - 1])) { + if (idx < firstIdx) firstIdx = idx; + break; + } + start = idx + 1; + } } return firstIdx === Infinity ? model : model.slice(firstIdx); } @@ -516,6 +553,30 @@ function gpt5BaseMatches(model, token) { } } +/** + * gpt-version-segment match: the token appears as an exact segment AND the next segment + * (the one immediately after the token) starts with a digit. + * This matches "gpt" in "gpt-5.5" (next seg "5") and "gpt" in "gpt-5-4-mini" (next seg "5") + * but NOT "gpt" in "gpt-neox-20b" (next seg "neox" starts with a letter). + * Also matches the dashless exact-segment alias (e.g. "gpt5") directly via segment equality. + */ +function gptVersionSegmentMatches(model, token) { + const lower = model.toLowerCase(); + const tok = token.toLowerCase(); + const segs = lower.split(/[^a-z0-9]+/); + for (let i = 0; i < segs.length; i++) { + if (segs[i] === tok) { + // Exact segment match (e.g. "gpt5" matches "gpt5-custom") + if (tok.length > 3 || /^\d/.test(tok)) return true; // dashless numeric form like "gpt5" + // For "gpt": require the next segment to start with a digit + const nextSeg = segs[i + 1]; + if (nextSeg !== undefined && /^\d/.test(nextSeg)) return true; + // No valid next segment — this "gpt" segment alone does not match + } + } + return false; +} + /** * Test if a family rule matches the given (normalized) model string for a provider. */ @@ -540,6 +601,8 @@ function ruleMatchesModel(rule, normalizedModel, provider) { const segs = lower.split(/[^a-z0-9]+/); return allTokens.some((t) => segs.some((s) => s.startsWith(t.toLowerCase()))); } + case "gpt-version-segment": + return allTokens.some((t) => gptVersionSegmentMatches(lower, t)); default: throw new Error(`unknown match_kind: ${rule.match_kind}`); } @@ -876,21 +939,45 @@ ${emitRustCapabilityResult(clean, " ")} // --------------------------------------------------------------------------- /// Strip any catalog-naming prefix to get the normalized model alias for family matching. -/// Finds the first occurrence of a known family token (claude-, gpt-) and returns from there. +/// Finds the first boundary-aligned occurrence of a known family token and returns from there. +/// Boundary-aligned: the token must start at position 0 or be preceded by a non-alphanumeric char. +/// This prevents "customgpt-5-5-endpoint" from stripping to "gpt-5-5-endpoint" via "gpt-" inside +/// the "customgpt-" prefix. /// /// Examples: -/// "goose-claude-fable-5" → "claude-fable-5" -/// "databricks-gpt-5.5" → "gpt-5.5" -/// "team-x-claude-opus-4-7" → "claude-opus-4-7" -/// "claude-opus-4-7" → "claude-opus-4-7" (no prefix) +/// "goose-claude-fable-5" → "claude-fable-5" (boundary: preceded by "-") +/// "databricks-gpt-5.5" → "gpt-5.5" (boundary: preceded by "-") +/// "team-x-claude-opus-4-7" → "claude-opus-4-7" (boundary: preceded by "-") +/// "claude-opus-4-7" → "claude-opus-4-7" (no prefix, already boundary) +/// "customgpt-5-5-ep" → "customgpt-5-5-ep" (no boundary match for "gpt-") /// "llama-3" → "llama-3" (no family token) pub fn strip_catalog_prefix(model: &str) -> &str { const FAMILY_TOKENS: &[&str] = &[${manifest.family_tokens.map((t) => `"${t}"`).join(", ")}]; let lower = model.to_ascii_lowercase(); - let first_idx = FAMILY_TOKENS - .iter() - .filter_map(|tok| lower.find(tok)) - .min(); + let mut first_idx: Option = None; + for tok in FAMILY_TOKENS { + let tok_bytes = tok.as_bytes(); + let lower_bytes = lower.as_bytes(); + let mut start = 0usize; + loop { + match lower_bytes[start..].windows(tok_bytes.len()).position(|w| w == tok_bytes) { + None => break, + Some(rel) => { + let idx = start + rel; + // Boundary check: position 0 or preceded by a non-alphanumeric byte + let at_boundary = idx == 0 || { + let prev = lower_bytes[idx - 1]; + !prev.is_ascii_alphanumeric() + }; + if at_boundary { + first_idx = Some(first_idx.map_or(idx, |f| f.min(idx))); + break; + } + start = idx + 1; + } + } + } + } match first_idx { Some(idx) => &model[idx..], None => model, @@ -1082,6 +1169,8 @@ function buildRustMatchExpr(rule, provider) { ) .join(" || "); } + case "gpt-version-segment": + return allTokens.map((t) => `gpt_version_segment_matches_rs(lower, "${t.toLowerCase()}")`).join(" || "); default: throw new Error(`unknown match_kind: ${rule.match_kind}`); } @@ -1157,8 +1246,7 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { let dash_rest = &suffix[1..]; // Reject -<1-3 digits> followed by non-alphanumeric or end (mirrors TS /^\\d{1,3}(?:[^a-z\\d]|$)/i). let first_non_digit = dash_rest.find(|c: char| !c.is_ascii_digit()).unwrap_or(dash_rest.len()); - let is_short_version = first_non_digit >= 1 - && first_non_digit <= 3 + let is_short_version = (1..=3).contains(&first_non_digit) && (first_non_digit == dash_rest.len() || !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric()); if is_short_version { @@ -1171,6 +1259,32 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { } } +// --------------------------------------------------------------------------- +// gpt-version-segment helper (used by generated family resolver) +// --------------------------------------------------------------------------- + +/// gpt-version-segment match: token is an exact segment AND (token itself starts with a digit, +/// OR the next segment after it starts with a digit). Prevents "gpt-neox-20b" from matching +/// "gpt" because "neox" starts with a letter, while "gpt-5.5" matches because next seg "5" is digit. +fn gpt_version_segment_matches_rs(model: &str, token: &str) -> bool { + let segs: Vec<&str> = model.split(|c: char| !c.is_ascii_alphanumeric()).collect(); + for (i, seg) in segs.iter().enumerate() { + if *seg == token { + // Dashless numeric form (e.g. "gpt5"): token length > 3 or starts with digit + if token.len() > 3 || token.as_bytes().first().is_some_and(|b| b.is_ascii_digit()) { + return true; + } + // For short alpha tokens like "gpt": require the next segment to start with a digit + if let Some(next_seg) = segs.get(i + 1) { + if next_seg.as_bytes().first().is_some_and(|b| b.is_ascii_digit()) { + return true; + } + } + } + } + false +} + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -1286,6 +1400,8 @@ function buildTsMatchExpr(rule, provider) { `lower.split(/[^a-z0-9]+/).some(s => s.startsWith("${t.toLowerCase()}"))`, ) .join(" || "); + case "gpt-version-segment": + return allTokens.map((t) => `gptVersionSegmentMatchesGenerated(lower, "${t.toLowerCase()}")`).join(" || "); default: throw new Error(`unknown match_kind: ${rule.match_kind}`); } @@ -1298,10 +1414,10 @@ const tsProviderFallbacks = providerFallbackKeys const concClean = { ...fb.concrete_unknown, registry_label: null }; delete blankClean._provenance; delete concClean._provenance; - return ` "${provider}": { + return ` ["${provider}", { blank: ${emitTsCapabilityResult(blankClean, " ")}, concreteUnknown: ${emitTsCapabilityResult(concClean, " ")}, - },`; + }],`; }) .join("\n"); @@ -1411,6 +1527,25 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean { } } +/** + * gpt-version-segment match: token is an exact segment AND (token itself starts with a digit, + * OR the next segment after it starts with a digit). Prevents "gpt-neox-20b" from matching + * "gpt" because "neox" starts with a letter, while "gpt-5.5" matches because next seg "5" is digit. + */ +function gptVersionSegmentMatchesGenerated(m: string, token: string): boolean { + const segs = m.split(/[^a-z0-9]+/); + for (let i = 0; i < segs.length; i++) { + if (segs[i] === token) { + // Dashless numeric form (e.g. "gpt5") or token itself starts with digit + if (token.length > 3 || /^\\d/.test(token)) return true; + // For "gpt": require the next segment to start with a digit + const nextSeg = segs[i + 1]; + if (nextSeg !== undefined && /^\\d/.test(nextSeg)) return true; + } + } + return false; +} + // --------------------------------------------------------------------------- // Exact records — provider-qualified, pre-prefix-stripping // --------------------------------------------------------------------------- @@ -1428,9 +1563,9 @@ ${tsExactEntries // Provider fallbacks // --------------------------------------------------------------------------- -const PROVIDER_FALLBACKS: Record = { +const PROVIDER_FALLBACKS = new Map([ ${tsProviderFallbacks} -}; +]); const DEFAULT_FALLBACK = { blank: ${emitTsCapabilityResult(defaultFbBlank, " ")}, @@ -1438,15 +1573,25 @@ const DEFAULT_FALLBACK = { }; // --------------------------------------------------------------------------- -// Strip catalog prefix — finds first family token occurrence +// Strip catalog prefix — boundary-aware, finds first boundary-aligned family token // --------------------------------------------------------------------------- export function stripCatalogPrefix(model: string): string { const FAMILY_TOKENS = [${manifest.family_tokens.map((t) => `"${t}"`).join(", ")}] as const; + const lower = model.toLowerCase(); let firstIdx = Infinity; for (const tok of FAMILY_TOKENS) { - const idx = model.toLowerCase().indexOf(tok); - if (idx !== -1 && idx < firstIdx) firstIdx = idx; + let start = 0; + while (true) { + const idx = lower.indexOf(tok, start); + if (idx === -1) break; + // Boundary check: position 0 or preceded by a non-alphanumeric character + if (idx === 0 || !/[a-z0-9]/.test(lower[idx - 1])) { + if (idx < firstIdx) firstIdx = idx; + break; + } + start = idx + 1; + } } return firstIdx === Infinity ? model : model.slice(firstIdx); } @@ -1492,7 +1637,7 @@ export function resolveModelCapabilities( // Step 3: provider fallback const isBlank = rawModelId.trim() === ""; - const fb = PROVIDER_FALLBACKS[provider] ?? DEFAULT_FALLBACK; + const fb = PROVIDER_FALLBACKS.get(provider) ?? DEFAULT_FALLBACK; return isBlank ? { ...fb.blank, registryLabel: null } : { ...fb.concreteUnknown, registryLabel: null }; } `; diff --git a/scripts/model-capabilities.json b/scripts/model-capabilities.json index d5787eb1d..d3279dab1 100644 --- a/scripts/model-capabilities.json +++ b/scripts/model-capabilities.json @@ -436,8 +436,8 @@ }, { "id": "dbv2-gpt-code-names-segment", - "_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj false-positives. Note: gpt-neox is a residual collision — ‘gpt-neox’ segments to [‘gpt’,‘neox’], so ‘gpt’ matches and it routes openai-responses. This is pinned as known behavior in the normative corpus, not an endorsement.", - "match_kind": "segment", + "_comment": "DBv2-only rule: models whose normalized alias has 'gpt' as a segment followed by a numeric segment (e.g. databricks-gpt-5.5 \u2192 'gpt' seg + '5' seg) route via OpenAI Responses. 'gpt5' dashless alias catches gpt5-custom forms. match_kind=gpt-version-segment: token must be an exact segment AND the next segment must start with a digit \u2014 prevents gptoss/gptj (no 'gpt' segment) AND gpt-neox-20b ('gpt' segment but next segment 'neox' is not numeric). Priority 6 > dbv2-claude's 5.", + "match_kind": "gpt-version-segment", "match_value": "gpt", "providers": [ "databricks_v2" diff --git a/scripts/normative-corpus.json b/scripts/normative-corpus.json index 580bb9a0a..6cbde28ea 100644 --- a/scripts/normative-corpus.json +++ b/scripts/normative-corpus.json @@ -15,7 +15,9 @@ "high" ], "default_effort": null, - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -30,7 +32,9 @@ "high" ], "default_effort": null, - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.5" } }, { @@ -47,7 +51,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.7" } }, { @@ -64,7 +70,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.8" } }, { @@ -81,7 +89,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Sonnet 5" } }, { @@ -98,7 +108,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Fable 5" } }, { @@ -115,7 +127,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Mythos 5" } }, { @@ -131,7 +145,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.6" } }, { @@ -147,7 +163,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Sonnet 4.6" } }, { @@ -163,7 +181,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Mythos Preview" } }, { @@ -183,7 +203,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -200,7 +222,9 @@ "max" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -216,7 +240,9 @@ "high" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5 Pro" } }, { @@ -234,7 +260,9 @@ "max" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6" } }, { @@ -252,7 +280,9 @@ "max" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6" } }, { @@ -269,7 +299,9 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.5" } }, { @@ -286,7 +318,9 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4" } }, { @@ -302,7 +336,9 @@ "high" ], "default_effort": "none", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.1" } }, { @@ -318,11 +354,13 @@ "high" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5" } }, { - "_group": "OpenAI adversarial \u2014 gpt5 boundary-aware matching (ported from config.rs tests)" + "_group": "OpenAI adversarial — gpt5 boundary-aware matching (ported from config.rs tests)" }, { "id": "openai-gpt5-1106-should-not-match-base", @@ -330,12 +368,17 @@ "raw_model_id": "gpt-5-1106", "_note": "gpt-5-1106: '-1106' is a 4-digit date segment, NOT a short version (gpt5-base rejects only 1-3 digit suffixes). Must match base table [minimal,low,medium,high], NOT fall through to unknown.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "minimal", "low", "medium", "high" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5" } }, { @@ -344,12 +387,17 @@ "raw_model_id": "gpt-5-4o", "_note": "gpt-5-4o: '4o' after '-' is NOT a short numeric suffix (it contains a letter). Must match gpt5-base. Crucially, must NOT match gpt-5.4 (the '4' is followed by 'o', not boundary char).", "expect": { + "thinking_mode": "none", "supported_efforts": [ "minimal", "low", "medium", "high" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5" } }, { @@ -358,18 +406,23 @@ "raw_model_id": "gpt-5-pro", "_note": "gpt-5-pro should hit gpt5-pro rule (priority 20), NOT gpt-5 base.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "high" ], - "default_effort": "high" + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5 Pro" } }, { "id": "openai-multi-digit-version-gpt5-10", "provider": "openai", "raw_model_id": "gpt-5-10", - "_note": "gpt-5-10 \u2014 two-digit suffix prevents gpt5-base match. Falls through to unknown.", + "_note": "gpt-5-10 — two-digit suffix prevents gpt5-base match. Falls through to unknown.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -377,39 +430,52 @@ "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "openai-gpt5-date-suffix", "provider": "openai", "raw_model_id": "gpt-5-20260101", - "_note": "gpt-5-20260101 \u2014 long numeric suffix after base: '20260101' is 8 digits, beyond 1-3 digit reject, should hit gpt5-base.", + "_note": "gpt-5-20260101 — long numeric suffix after base: '20260101' is 8 digits, beyond 1-3 digit reject, should hit gpt5-base.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "minimal", "low", "medium", "high" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5" } }, { - "_group": "DatabricksV2 \u2014 segment-based routing (ported from llm.rs tests)" + "_group": "DatabricksV2 — segment-based routing (ported from llm.rs tests)" }, { "id": "dbv2-gpt5-route-openai-responses", "provider": "databricks_v2", "raw_model_id": "gpt-5.5", "expect": { - "databricks_v2_wire_route": "openai-responses", + "thinking_mode": "none", "supported_efforts": [ "none", "low", "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.5" } }, { @@ -417,7 +483,6 @@ "provider": "databricks_v2", "raw_model_id": "claude-opus-4-7", "expect": { - "databricks_v2_wire_route": "anthropic-messages", "thinking_mode": "adaptive", "supported_efforts": [ "low", @@ -425,26 +490,39 @@ "high", "xhigh", "max" - ] + ], + "default_effort": "high", + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.7" } }, { "id": "dbv2-claude-prefix-stripped", "provider": "databricks_v2", "raw_model_id": "databricks-claude-opus-4-7", - "_note": "databricks- prefix stripped \u2192 claude-opus-4-7 \u2192 Anthropic route", + "_note": "databricks- prefix stripped → claude-opus-4-7 → Anthropic route", "expect": { + "thinking_mode": "adaptive", + "supported_efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "high", "databricks_v2_wire_route": "anthropic-messages", - "thinking_mode": "adaptive" + "normalization_policy": "none", + "registry_label": "Claude Opus 4.7" } }, { "id": "dbv2-goose-claude-prefix-stripped", "provider": "databricks_v2", "raw_model_id": "goose-claude-fable-5", - "_note": "goose- prefix stripped \u2192 claude-fable-5 \u2192 Anthropic adaptive+xhigh", + "_note": "goose- prefix stripped → claude-fable-5 → Anthropic adaptive+xhigh", "expect": { - "databricks_v2_wire_route": "anthropic-messages", "thinking_mode": "adaptive", "supported_efforts": [ "low", @@ -452,16 +530,19 @@ "high", "xhigh", "max" - ] + ], + "default_effort": "high", + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": "Claude Fable 5" } }, { "id": "dbv2-team-prefix-stripped", "provider": "databricks_v2", "raw_model_id": "team-x-claude-opus-4-7", - "_note": "team-x- prefix stripped \u2192 claude-opus-4-7 \u2192 Anthropic route", + "_note": "team-x- prefix stripped → claude-opus-4-7 → Anthropic route", "expect": { - "databricks_v2_wire_route": "anthropic-messages", "thinking_mode": "adaptive", "supported_efforts": [ "low", @@ -469,25 +550,53 @@ "high", "xhigh", "max" - ] + ], + "default_effort": "high", + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": "Claude Opus 4.7" } }, { "id": "dbv2-consolidated-llama-not-sol", "provider": "databricks_v2", "raw_model_id": "consolidated-llama", - "_note": "segment test: 'sol' is a SUBSTRING of 'consolidated' \u2014 must NOT match DATABRICKS_V2_OPENAI_CODE_NAMES 'sol'. Falls through to mlflow-chat.", + "_note": "segment test: 'sol' is a SUBSTRING of 'consolidated' — must NOT match DATABRICKS_V2_OPENAI_CODE_NAMES 'sol'. Falls through to mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "dbv2-terraform-coder-not-terra", "provider": "databricks_v2", "raw_model_id": "terraform-coder", - "_note": "segment test: 'terra' is a prefix of 'terraform' \u2014 must NOT match 'terra' code name. Falls through to mlflow-chat.", + "_note": "segment test: 'terra' is a prefix of 'terraform' — must NOT match 'terra' code name. Falls through to mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -496,7 +605,19 @@ "raw_model_id": "corpus-reranker", "_note": "segment test: 'opus' is NOT a segment of corpus-reranker (segments: corpus, reranker). mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -505,7 +626,19 @@ "raw_model_id": "octopus-model", "_note": "segment test: 'opus' is not a segment of octopus-model (segments: octopus, model). mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -514,11 +647,22 @@ "raw_model_id": "goose-opus-5", "_note": "'opus' IS a named segment of goose-opus-5 (segments: goose, opus, 5). Routes Anthropic. Key test: agrees with llm.rs but disagreed with old config.rs.", "expect": { - "databricks_v2_wire_route": "anthropic-messages" + "thinking_mode": "omit-fields", + "supported_efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "high", + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null } }, { - "_group": "P2-A resolver-contract vectors (plan v4 \u00a7Resolver contract)" + "_group": "P2-A resolver-contract vectors (plan v4 §Resolver contract)" }, { "id": "resolver-exact-raw-id-hit", @@ -526,26 +670,36 @@ "raw_model_id": "databricks-gpt-5-4-mini", "_note": "Exact record exists. Must return exact Databricks override: low|medium|high (not family's none+xhigh).", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4 Mini" } }, { "id": "resolver-prefixed-alias-misses-exact", "provider": "databricks_v2", "raw_model_id": "team-x-databricks-gpt-5-4-mini", - "_note": "Prefixed alias of an exact ID. Raw exact lookup MUST miss (key is team-x-..., not databricks-...). Falls to family rules (gpt5-4 family \u2192 none+xhigh).", + "_note": "Prefixed alias of an exact ID. Raw exact lookup MUST miss (key is team-x-..., not databricks-...). Falls to family rules (gpt5-4 family → none+xhigh).", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", "low", "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4" } }, { @@ -554,36 +708,55 @@ "raw_model_id": "databricks-gpt-5-4-mini", "_note": "Same raw ID but different provider. Exact record is databricks_v2-scoped; must miss. Falls to openai family rules.", "expect": { - "databricks_v2_wire_route": "not-applicable" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4" } }, { "id": "resolver-exact-efforts-plus-family-route", "provider": "databricks_v2", "raw_model_id": "databricks-gpt-5-6-sol", - "_note": "Exact record with efforts from models.dev (low|medium|high|max \u2014 provider-advertised, no none/xhigh). Route materialized from gpt5-6 family rule (openai-responses). Must return both, complete.", + "_note": "Exact record with efforts from models.dev (low|medium|high|max — provider-advertised, no none/xhigh). Route materialized from gpt5-6 family rule (openai-responses). Must return both, complete.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high", "max" ], - "databricks_v2_wire_route": "openai-responses" + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6 Sol" } }, { "id": "dbv2-gpt5-5-exact-override", "provider": "databricks_v2", "raw_model_id": "databricks-gpt-5-5", - "_note": "Exact record adopts models.dev advertised set [low,medium,high]. Family rule (gpt5-5) has none+xhigh \u2014 provider-advertised wins per plan F1.", + "_note": "Exact record adopts models.dev advertised set [low,medium,high]. Family rule (gpt5-5) has none+xhigh — provider-advertised wins per plan F1.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" ], - "databricks_v2_wire_route": "openai-responses" + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.5" } }, { @@ -595,7 +768,7 @@ "raw_model_id": "", "_note": "DBv2 blank: route-unknown, all 7 efforts, default medium.", "expect": { - "databricks_v2_wire_route": "route-unknown", + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -605,7 +778,10 @@ "xhigh", "max" ], - "default_effort": "medium" + "default_effort": "medium", + "databricks_v2_wire_route": "route-unknown", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -614,7 +790,7 @@ "raw_model_id": "some-unknown-model-xyz", "_note": "DBv2 concrete-unknown: mlflow-chat, all-except-max (6 efforts).", "expect": { - "databricks_v2_wire_route": "mlflow-chat", + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -622,7 +798,11 @@ "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -631,7 +811,7 @@ "raw_model_id": "", "_note": "OpenAI blank: not-applicable route, all-except-max, medium default.", "expect": { - "databricks_v2_wire_route": "not-applicable", + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -640,7 +820,10 @@ "high", "xhigh" ], - "default_effort": "medium" + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -649,7 +832,7 @@ "raw_model_id": "gpt-4o", "_note": "OpenAI concrete unknown (unverified family): not-applicable route, all-except-max, medium default.", "expect": { - "databricks_v2_wire_route": "not-applicable", + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -658,7 +841,10 @@ "high", "xhigh" ], - "default_effort": "medium" + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -675,7 +861,10 @@ "xhigh", "max" ], - "default_effort": "high" + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -684,7 +873,18 @@ "raw_model_id": "claude-ultra-9000", "_note": "Anthropic concrete-unknown: omit-fields (never guess request shape).", "expect": { - "thinking_mode": "omit-fields" + "thinking_mode": "omit-fields", + "supported_efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -694,22 +894,25 @@ "id": "databricks-gpt5-pro-effort", "provider": "databricks", "raw_model_id": "databricks-gpt-5-pro", - "_note": "Legacy databricks GPT-5 Pro: same effort set as openai/gpt-5-pro \u2014 only [high], default high. Wire route not-applicable.", + "_note": "Legacy databricks GPT-5 Pro: same effort set as openai/gpt-5-pro — only [high], default high. Wire route not-applicable.", "expect": { - "databricks_v2_wire_route": "not-applicable", + "thinking_mode": "none", "supported_efforts": [ "high" ], - "default_effort": "high" + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5 Pro" } }, { "id": "databricks-gpt5-6-effort", "provider": "databricks", "raw_model_id": "databricks-gpt-5.6", - "_note": "Legacy databricks GPT-5.6: same effort set as openai/gpt-5.6 \u2014 [none,low,medium,high,xhigh,max], default medium. Wire route not-applicable.", + "_note": "Legacy databricks GPT-5.6: same effort set as openai/gpt-5.6 — [none,low,medium,high,xhigh,max], default medium. Wire route not-applicable.", "expect": { - "databricks_v2_wire_route": "not-applicable", + "thinking_mode": "none", "supported_efforts": [ "none", "low", @@ -718,28 +921,34 @@ "xhigh", "max" ], - "default_effort": "medium" + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6" } }, { "id": "databricks-gpt5-1-effort", "provider": "databricks", "raw_model_id": "databricks-gpt-5.1", - "_note": "Legacy databricks GPT-5.1: same effort set as openai/gpt-5.1 \u2014 [none,low,medium,high], default none. Wire route not-applicable.", + "_note": "Legacy databricks GPT-5.1: same effort set as openai/gpt-5.1 — [none,low,medium,high], default none. Wire route not-applicable.", "expect": { - "databricks_v2_wire_route": "not-applicable", + "thinking_mode": "none", "supported_efforts": [ "none", "low", "medium", "high" ], - "default_effort": "none" + "default_effort": "none", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.1" } }, { "_group": "openai-compat alias canonicalization (Thufir P3 corrective action 1)", - "_note": "Rust normalizes openai-compat \u2192 Provider::OpenAi before reaching normalize_effort_for_provider. TS PROVIDER_ALIASES must match so the UI effort table equals the Rust request behavior. Interpreters must canonicalize openai-compat \u2192 openai before resolving; the expected values are identical to the corresponding openai vectors." + "_note": "Rust normalizes openai-compat → Provider::OpenAi before reaching normalize_effort_for_provider. TS PROVIDER_ALIASES must match so the UI effort table equals the Rust request behavior. Interpreters must canonicalize openai-compat → openai before resolving; the expected values are identical to the corresponding openai vectors." }, { "id": "openai-compat-gpt-5-pro", @@ -752,7 +961,9 @@ "high" ], "default_effort": "high", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5 Pro" } }, { @@ -770,14 +981,16 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.5" } }, { "id": "openai-compat-empty-model", "provider": "openai-compat", "raw_model_id": "", - "_note": "openai-compat with blank model: resolves identically to openai unknown \u2014 all-except-max, default medium.", + "_note": "openai-compat with blank model: resolves identically to openai unknown — all-except-max, default medium.", "expect": { "thinking_mode": "none", "supported_efforts": [ @@ -789,7 +1002,9 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable" + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -800,8 +1015,9 @@ "id": "openai-gpt5-10-preview-reject-base", "provider": "openai", "raw_model_id": "gpt-5-10-preview", - "_note": "CRITICAL divergence fix: -10- is a 2-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "_note": "CRITICAL divergence fix: -10- is a 2-digit suffix → gpt5-base rejects. Falls to openai concrete-unknown.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -809,15 +1025,20 @@ "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "openai-gpt5-2-mini-reject-base", "provider": "openai", "raw_model_id": "gpt-5-2-mini", - "_note": "CRITICAL divergence fix: -2- is a 1-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "_note": "CRITICAL divergence fix: -2- is a 1-digit suffix → gpt5-base rejects. Falls to openai concrete-unknown.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -825,15 +1046,20 @@ "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "openai-gpt5-9-dot-1-reject-base", "provider": "openai", "raw_model_id": "gpt-5-9.1", - "_note": "CRITICAL divergence fix: -9 followed by '.' is a 1-digit suffix + non-alnum \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "_note": "CRITICAL divergence fix: -9 followed by '.' is a 1-digit suffix + non-alnum → gpt5-base rejects. Falls to openai concrete-unknown.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", "minimal", @@ -841,22 +1067,32 @@ "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "openai-customgpt-5-5-no-token-match", "provider": "openai", "raw_model_id": "customgpt-5-5-endpoint", - "_note": "Left-boundary fix context: customgpt-5-5-endpoint strips to gpt-5-5-endpoint which DOES match gpt5-5 family. This is correct. Update: expect gpt5-5 efforts.", + "_note": "boundary-aware stripCatalogPrefix fix: customgpt-5-5-endpoint has no boundary-aligned gpt- token (g in customgpt is preceded by m), so the alias is NOT stripped. Falls to openai concrete-unknown fallback.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "none", + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -869,7 +1105,19 @@ "raw_model_id": "gptoss-model", "_note": "Collision-negative: 'gptoss' is a segment starting with 'gpt' but NOT an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { @@ -878,43 +1126,100 @@ "raw_model_id": "gptj-6b", "_note": "Collision-negative: 'gptj' is not an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.", "expect": { - "databricks_v2_wire_route": "mlflow-chat" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "dbv2-customgpt-not-responses", "provider": "databricks_v2", "raw_model_id": "customgpt-5-5-endpoint", - "_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint → gpt5-5 family → openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.", + "_note": "customgpt-5-5-endpoint: boundary-aware strip finds no boundary-aligned gpt- (preceded by m in customgpt). Normalized alias is customgpt-5-5-endpoint itself, segments=[customgpt,5,5,endpoint], no gpt segment → mlflow-chat. This is the corrected behavior.", "expect": { - "databricks_v2_wire_route": "openai-responses" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { - "id": "dbv2-gpt-neox-residual-collision", + "id": "dbv2-gpt-neox-mlflow", "provider": "databricks_v2", "raw_model_id": "gpt-neox-20b", - "_note": "Residual-collision pin (not an endorsement): 'gpt-neox-20b' splits to segments ['gpt','neox','20b'], so 'gpt' exactly matches and routes openai-responses. gptoss/gptj are fixed; gpt-neox is a known residual of exact-segment matching. Pinned here so any future fix surfaces as a test change.", + "_note": "gpt-version-segment fix: gpt-neox-20b strips to itself (boundary-aligned at start), segments=[gpt,neox,20b]. gpt-version-segment requires next segment after gpt to be numeric; neox starts with n → no match → falls to mlflow-chat. This is corrected behavior.", "expect": { - "databricks_v2_wire_route": "openai-responses" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } }, { "id": "dbv2-gpt5-segment-positive", "provider": "databricks_v2", "raw_model_id": "databricks-gpt5-custom", - "_note": "DBv2 gpt segment rule positive: normalized 'gpt5-custom' \u2192 segment 'gpt5' IS an exact match in match_aliases. Routes openai-responses.", + "_note": "DBv2 gpt segment rule positive: normalized 'gpt5-custom' → segment 'gpt5' IS an exact match in match_aliases. Routes openai-responses.", "expect": { - "databricks_v2_wire_route": "openai-responses" + "thinking_mode": "none", + "supported_efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5" } }, { "id": "dbv2-dual-marker-gpt-wins-openai", "provider": "databricks_v2", "raw_model_id": "gpt-opus-5", - "_note": "Dual-marker: normalized 'gpt-opus-5' \u2192 segments include 'gpt' AND 'opus'. Priority 6 (gpt) > 5 (claude): OpenAI wins. Must route openai-responses.", + "_note": "Dual-marker: normalized 'gpt-opus-5' → segments include 'gpt' AND 'opus'. Priority 6 (gpt) > 5 (claude): OpenAI wins. Must route openai-responses.", "expect": { - "databricks_v2_wire_route": "openai-responses" + "thinking_mode": "omit-fields", + "supported_efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "high", + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null } }, { @@ -934,7 +1239,10 @@ "xhigh", "max" ], - "default_effort": "high" + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": "Claude Opus 5" } }, { @@ -943,13 +1251,17 @@ "raw_model_id": "databricks-gpt-5-6-sol", "_note": "Sol exact record: normalization_policy comes from gpt5-6 family rule (openai-standard). Also checks supported_efforts override.", "expect": { - "normalization_policy": "openai-standard", + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high", "max" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6 Sol" } }, { @@ -958,12 +1270,16 @@ "raw_model_id": "databricks-gpt-5-6-luna", "_note": "Luna exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" ], - "databricks_v2_wire_route": "openai-responses" + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6 Luna" } }, { @@ -972,12 +1288,16 @@ "raw_model_id": "databricks-gpt-5-6-terra", "_note": "Terra exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" ], - "databricks_v2_wire_route": "openai-responses" + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.6 Terra" } }, { @@ -986,22 +1306,38 @@ "raw_model_id": "databricks-gpt-5-4-nano", "_note": "Exact record: gpt-5-4-nano, efforts [low,medium,high], registry_label 'GPT-5.4 Nano'.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" ], - "registry_label": "GPT-5.4 Nano", - "databricks_v2_wire_route": "openai-responses" + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4 Nano" } }, { "id": "openrouter-concrete-unknown-fallback", "provider": "openrouter", "raw_model_id": "some-model-xyz", - "_note": "openrouter concrete-unknown \u2192 _default fallback (wire route not-applicable).", + "_note": "openrouter concrete-unknown → _default fallback (wire route not-applicable).", "expect": { - "databricks_v2_wire_route": "not-applicable" + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null } }, { @@ -1010,10 +1346,14 @@ "raw_model_id": "gpt-5-pro", "_note": "Casing: uppercase provider 'OpenAI' normalized to 'openai'. Must resolve same as openai/gpt-5-pro.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "high" ], - "default_effort": "high" + "default_effort": "high", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5 Pro" } }, { @@ -1022,11 +1362,196 @@ "raw_model_id": "DATABRICKS-GPT-5-4-NANO", "_note": "Casing: uppercase raw_model_id. Case-insensitive exact lookup must hit the lowercase record.", "expect": { + "thinking_mode": "none", "supported_efforts": [ "low", "medium", "high" - ] + ], + "default_effort": "medium", + "databricks_v2_wire_route": "openai-responses", + "normalization_policy": "openai-standard", + "registry_label": "GPT-5.4 Nano" + } + }, + { + "_group": "Prototype-key safety vectors", + "_note": "Verify that PROVIDER_FALLBACKS Map prevents prototype-chain pollution." + }, + { + "id": "prototype-key-constructor-blank", + "provider": "constructor", + "raw_model_id": "", + "_note": "Prototype-key safety: \"constructor\" as provider must not resolve through Object prototype chain. Must return a complete record identical to default fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "prototype-key-constructor-some-model", + "provider": "constructor", + "raw_model_id": "some-model", + "_note": "Prototype-key safety: \"constructor\" as provider must not resolve through Object prototype chain. Must return a complete record identical to default fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "prototype-key-proto__-blank", + "provider": "__proto__", + "raw_model_id": "", + "_note": "Prototype-key safety: \"__proto__\" as provider must not resolve through Object prototype chain. Must return a complete record identical to default fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "prototype-key-proto__-some-model", + "provider": "__proto__", + "raw_model_id": "some-model", + "_note": "Prototype-key safety: \"__proto__\" as provider must not resolve through Object prototype chain. Must return a complete record identical to default fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "_group": "Boundary-aware strip prefix negative vectors", + "_note": "customgpt/sgpt/mygpt have no boundary-aligned family token, so no prefix is stripped." + }, + { + "id": "openai-sgpt-5-5-no-strip", + "provider": "openai", + "raw_model_id": "sgpt-5-5", + "_note": "boundary-aware strip: \"sgpt-5-5\" has gpt- at a non-boundary position (preceded by alphanumeric). No strip occurs. Falls to openai concrete-unknown fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null + } + }, + { + "id": "dbv2-sgpt-5-5-mlflow", + "provider": "databricks_v2", + "raw_model_id": "sgpt-5-5", + "_note": "boundary-aware strip: \"sgpt-5-5\" has gpt- at a non-boundary position (preceded by alphanumeric). No strip occurs. Falls to mlflow-chat.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null + } + }, + { + "id": "openai-mygpt-5-no-strip", + "provider": "openai", + "raw_model_id": "mygpt-5", + "_note": "boundary-aware strip: \"mygpt-5\" has gpt- at a non-boundary position (preceded by alphanumeric). No strip occurs. Falls to openai concrete-unknown fallback.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "not-applicable", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null + } + }, + { + "id": "dbv2-mygpt-5-mlflow", + "provider": "databricks_v2", + "raw_model_id": "mygpt-5", + "_note": "boundary-aware strip: \"mygpt-5\" has gpt- at a non-boundary position (preceded by alphanumeric). No strip occurs. Falls to mlflow-chat.", + "expect": { + "thinking_mode": "none", + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "default_effort": "medium", + "databricks_v2_wire_route": "mlflow-chat", + "normalization_policy": "openai-clamp-max-to-xhigh", + "registry_label": null } } ] diff --git a/scripts/run-corpus.mjs b/scripts/run-corpus.mjs index 1037a8f30..6fcb00643 100644 --- a/scripts/run-corpus.mjs +++ b/scripts/run-corpus.mjs @@ -48,11 +48,29 @@ function canonicalizeProvider(provider) { let passed = 0; let failed = 0; +const REQUIRED_AXES = [ + "thinking_mode", + "supported_efforts", + "default_effort", + "databricks_v2_wire_route", + "normalization_policy", + "registry_label", +]; + for (const entry of corpus) { // Skip group header entries if (entry._group) continue; if (!entry.expect) continue; + // Require all 6 axes on every executable vector — sparse vectors hide divergences. + const missingAxes = REQUIRED_AXES.filter((ax) => !(ax in entry.expect)); + if (missingAxes.length > 0) { + failed++; + console.error(` FAIL ${entry.id}: sparse vector — missing axes: ${missingAxes.join(", ")}`); + console.error(` All six axes are required: ${REQUIRED_AXES.join(", ")}`); + continue; + } + // resolveModelCapabilities returns camelCase keys (registryLabel, thinkingMode, etc.) const result = resolveModelCapabilities(canonicalizeProvider(entry.provider), entry.raw_model_id); const expect = entry.expect; diff --git a/scripts/test-manifest-validator.mjs b/scripts/test-manifest-validator.mjs index fafab5051..8b28bf8a3 100644 --- a/scripts/test-manifest-validator.mjs +++ b/scripts/test-manifest-validator.mjs @@ -538,4 +538,129 @@ test("schema-negative: exact_record invalid normalization_policy is rejected", ( ); }); +// --------------------------------------------------------------------------- +// Rule: family_rule match_priority must be a non-negative integer +// --------------------------------------------------------------------------- +test("schema-negative: family_rule match_priority string (injection vector) is rejected", () => { + assertRejects( + "family_rule string match_priority", + mutate((m) => { + // Inject a string that contains a compile_error! macro — must be rejected before emission + m.family_rules[0].match_priority = 'compile_error!("THUFIR_INJECTED")'; + }), + "match_priority", + ); +}); + +test("schema-negative: family_rule match_priority negative integer is rejected", () => { + assertRejects( + "family_rule negative match_priority", + mutate((m) => { + m.family_rules[0].match_priority = -1; + }), + "match_priority", + ); +}); + +test("schema-negative: family_rule match_priority float is rejected", () => { + assertRejects( + "family_rule float match_priority", + mutate((m) => { + m.family_rules[0].match_priority = 1.5; + }), + "match_priority", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record uppercase duplicate key is rejected (after lowercasing) +// --------------------------------------------------------------------------- +test("schema-negative: exact_record uppercase duplicate key is rejected", () => { + assertRejects( + "exact_record uppercase duplicate key", + mutate((m) => { + // Add an uppercase copy of an existing exact record key + const existing = m.exact_records[0]; + m.exact_records.push({ + ...existing, + provider: existing.provider.toUpperCase(), + raw_model_id: existing.raw_model_id.toUpperCase(), + }); + }), + "duplicate exact_record key", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record inherited default_effort not in inherited supported_efforts +// --------------------------------------------------------------------------- +test("schema-negative: exact_record inherited default_effort outside materialized supported_efforts is rejected", () => { + assertRejects( + "exact_record inherited default out of materialized efforts", + mutate((m) => { + // Override supported_efforts_override to a single value that excludes the family default. + // For any exact record that inherits family default_effort, override efforts to exclude it. + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + // Family default for the gpt5-4 rule is "none". Override to only ["low"] to force mismatch. + rec.supported_efforts_override = ["low"]; + // No explicit default_effort — inherits "none" from family, but "none" is not in ["low"] + }), + "materialized default_effort", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: family_rule supported_efforts must have no duplicates +// --------------------------------------------------------------------------- +test("schema-negative: family_rule supported_efforts with duplicate is rejected", () => { + assertRejects( + "family_rule duplicate effort", + mutate((m) => { + m.family_rules[0].supported_efforts = ["low", "low", "medium"]; + }), + "duplicate effort", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: family_rule supported_efforts must follow canonical order +// --------------------------------------------------------------------------- +test("schema-negative: family_rule supported_efforts out of canonical order is rejected", () => { + assertRejects( + "family_rule efforts out of order", + mutate((m) => { + m.family_rules[0].supported_efforts = ["high", "low", "medium"]; + }), + "canonical order", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: provider_fallback supported_efforts must have no duplicates +// --------------------------------------------------------------------------- +test("schema-negative: provider_fallback supported_efforts with duplicate is rejected", () => { + assertRejects( + "provider_fallback duplicate effort", + mutate((m) => { + const provider = Object.keys(m.provider_fallbacks)[0]; + m.provider_fallbacks[provider].blank.supported_efforts = ["low", "low", "medium"]; + }), + "duplicate effort", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: provider_fallback supported_efforts must follow canonical order +// --------------------------------------------------------------------------- +test("schema-negative: provider_fallback supported_efforts out of canonical order is rejected", () => { + assertRejects( + "provider_fallback efforts out of order", + mutate((m) => { + const provider = Object.keys(m.provider_fallbacks)[0]; + m.provider_fallbacks[provider].blank.supported_efforts = ["high", "low", "medium"]; + }), + "canonical order", + ); +}); + console.log("\nSchema-negative validator tests complete.");