diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a8f077806..5bea7a45f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -146,7 +146,7 @@ jobs: - name: Set up Node.js uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 with: - node-version: '22' + node-version: '24' package-manager-cache: false - name: Regenerate artifacts diff --git a/crates/buzz-agent/src/generated_model_capabilities.rs b/crates/buzz-agent/src/generated_model_capabilities.rs index 219c22e35..7afcf2e4f 100644 --- a/crates/buzz-agent/src/generated_model_capabilities.rs +++ b/crates/buzz-agent/src/generated_model_capabilities.rs @@ -94,8 +94,11 @@ pub struct CapabilityResult { /// Returns the exact capability record for a provider-qualified raw model ID, /// if one exists in the manifest. This is checked BEFORE any prefix stripping. +/// Matching is case-insensitive — both inputs are lowercased before comparison. pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option { - match (provider, raw_model_id) { + let prov_lc = provider.to_lowercase(); + let id_lc = raw_model_id.to_lowercase(); + match (prov_lc.as_str(), id_lc.as_str()) { ("databricks_v2", "databricks-gpt-5-4-mini") => { // provenance: exact(databricks_v2::databricks-gpt-5-4-mini) // registry_label: exact_record @@ -204,6 +207,48 @@ pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option { + // provenance: exact(databricks_v2::databricks-gpt-5-6-luna) + // registry_label: exact_record + // supported_efforts: exact_record + // databricks_v2_wire_route: family:openai-gpt5-6@15 + // thinking_mode: family:openai-gpt5-6@15 + // normalization_policy: family:openai-gpt5-6@15 + // default_effort: family:openai-gpt5-6@15 + Some(CapabilityResult { + registry_label: Some("GPT-5.6 Luna"), + thinking_mode: ThinkingMode::None, + supported_efforts: Cow::Borrowed(&[ + ThinkingEffort::Low, + ThinkingEffort::Medium, + ThinkingEffort::High, + ]), + default_effort: Some(ThinkingEffort::Medium), + databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses, + normalization_policy: NormalizationPolicy::OpenAiStandard, + }) + } + ("databricks_v2", "databricks-gpt-5-6-terra") => { + // provenance: exact(databricks_v2::databricks-gpt-5-6-terra) + // registry_label: exact_record + // supported_efforts: exact_record + // databricks_v2_wire_route: family:openai-gpt5-6@15 + // thinking_mode: family:openai-gpt5-6@15 + // normalization_policy: family:openai-gpt5-6@15 + // default_effort: family:openai-gpt5-6@15 + Some(CapabilityResult { + registry_label: Some("GPT-5.6 Terra"), + thinking_mode: ThinkingMode::None, + supported_efforts: Cow::Borrowed(&[ + ThinkingEffort::Low, + ThinkingEffort::Medium, + ThinkingEffort::High, + ]), + default_effort: Some(ThinkingEffort::Medium), + databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses, + normalization_policy: NormalizationPolicy::OpenAiStandard, + }) + } _ => None, } } @@ -957,6 +1002,31 @@ pub fn lookup_by_family_rules(provider: &str, normalized: &str) -> Option Option bool { let lower = model; @@ -1346,6 +1396,15 @@ fn gpt5_token_matches_rs(model: &str, token: &str) -> bool { Some(rel_idx) => { let abs_idx = start + rel_idx; let after_idx = abs_idx + tok_lower.len(); + // Left-boundary check: must start at string start or after '-' or '.'. + let left_ok = abs_idx == 0 || { + let prev = lower.as_bytes()[abs_idx - 1]; + prev == b'-' || prev == b'.' + }; + if !left_ok { + start = after_idx; + continue; + } let after_char = lower[after_idx..].chars().next(); match after_char { None | Some('-') => return true, @@ -1367,6 +1426,15 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { Some(rel_idx) => { let abs_idx = start + rel_idx; let after_idx = abs_idx + tok_lower.len(); + // Left-boundary check: must start at string start or after '-' or '.'. + let left_ok = abs_idx == 0 || { + let prev = lower.as_bytes()[abs_idx - 1]; + prev == b'-' || prev == b'.' + }; + if !left_ok { + start = after_idx; + continue; + } let suffix = &lower[after_idx..]; if suffix.is_empty() { return true; @@ -1376,15 +1444,14 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { continue; } let dash_rest = &suffix[1..]; - // Reject -<1-3 digits> that look like version numbers. - let is_short_version = - dash_rest.chars().take(4).enumerate().all(|(i, c)| { - if i < 3 { - c.is_ascii_digit() - } else { - !c.is_ascii_alphanumeric() - } - }) && dash_rest.chars().next().is_some_and(|c| c.is_ascii_digit()); + // Reject -<1-3 digits> followed by non-alphanumeric or end (mirrors TS /^\d{1,3}(?:[^a-z\d]|$)/i). + let first_non_digit = dash_rest + .find(|c: char| !c.is_ascii_digit()) + .unwrap_or(dash_rest.len()); + let is_short_version = first_non_digit >= 1 + && first_non_digit <= 3 + && (first_non_digit == dash_rest.len() + || !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric()); if is_short_version { start = after_idx; continue; diff --git a/crates/buzz-agent/src/generated_model_capabilities_tests.rs b/crates/buzz-agent/src/generated_model_capabilities_tests.rs index 4f956af8d..efa6f6153 100644 --- a/crates/buzz-agent/src/generated_model_capabilities_tests.rs +++ b/crates/buzz-agent/src/generated_model_capabilities_tests.rs @@ -1,6 +1,7 @@ //! Tests for generated model capabilities — normative corpus + handwritten supplements. //! //! This module is conditionally compiled as #[cfg(test)] from generated_model_capabilities.rs. +//! It is hand-maintained (not regenerated) and lives outside the regen-diff gate. //! //! Three layers: //! 1. Shared normative corpus executed against the Rust interpreter — every vector in @@ -139,8 +140,10 @@ mod shared_corpus_tests { // Canonicalize provider aliases before resolving — mirrors the // production path where Rust normalizes "openai-compat" → Provider::OpenAi // (config.rs) and the TS canonicalizeProvider() resolves "databricks-v2" - // → "databricks_v2" before generated lookups. - let canonical_provider = match provider { + // → "databricks_v2" before generated lookups. Lowercase first so corpus + // vectors like provider="OpenAI" test case-insensitive normalization. + let provider_lc = provider.to_lowercase(); + let canonical_provider = match provider_lc.as_str() { "openai-compat" => "openai", "databricks-v2" => "databricks_v2", other => other, diff --git a/desktop/src/features/agents/lib/formatAgentModelLabel.ts b/desktop/src/features/agents/lib/formatAgentModelLabel.ts index c6e2cffd5..a14e74764 100644 --- a/desktop/src/features/agents/lib/formatAgentModelLabel.ts +++ b/desktop/src/features/agents/lib/formatAgentModelLabel.ts @@ -10,11 +10,14 @@ import { * - "openai-compat" → "openai": Rust already accepts "openai-compat" as Provider::OpenAi * (crates/buzz-agent/src/config.rs); TS must canonicalize identically so the UI shows * the same effort table that the Rust request path will apply. + * + * Map is used (not a plain object) to avoid prototype-chain collisions + * ("constructor", "__proto__", etc.) silently resolving to a built-in value. */ -const PROVIDER_ALIASES: Readonly> = { - "databricks-v2": "databricks_v2", - "openai-compat": "openai", -}; +const PROVIDER_ALIASES = new Map([ + ["databricks-v2", "databricks_v2"], + ["openai-compat", "openai"], +]); /** * Normalizes a provider id to the canonical form expected by the generated @@ -25,7 +28,7 @@ const PROVIDER_ALIASES: Readonly> = { */ export function canonicalizeProvider(provider: string): string { const normalized = provider.trim().toLowerCase(); - return PROVIDER_ALIASES[normalized] ?? normalized; + return PROVIDER_ALIASES.get(normalized) ?? normalized; } /** diff --git a/desktop/src/features/agents/ui/buzzAgentConfig.ts b/desktop/src/features/agents/ui/buzzAgentConfig.ts index 8638af47c..b8de87c58 100644 --- a/desktop/src/features/agents/ui/buzzAgentConfig.ts +++ b/desktop/src/features/agents/ui/buzzAgentConfig.ts @@ -75,7 +75,7 @@ export function getProviderEffortConfig( ): ProviderEffortConfig { const cap = resolveModelCapabilities( canonicalizeProvider(providerId), - model ?? "", + (model ?? "").trim(), ); return { validValues: cap.supportedEfforts, diff --git a/desktop/src/features/agents/ui/modelCapabilities.ts b/desktop/src/features/agents/ui/modelCapabilities.ts index f17819d03..b4831166d 100644 --- a/desktop/src/features/agents/ui/modelCapabilities.ts +++ b/desktop/src/features/agents/ui/modelCapabilities.ts @@ -87,12 +87,19 @@ export const DATABRICKS_MODEL_NAMES: Map = new Map([ // gpt5 boundary-aware token helpers // --------------------------------------------------------------------------- +function hasLeftBoundaryGenerated(m: string, idx: number): boolean { + if (idx === 0) return true; + const prev = m[idx - 1]; + return prev === "-" || prev === "."; +} + function gpt5TokenMatchesGenerated(m: string, token: string): boolean { let start = 0; while (true) { const idx = m.indexOf(token, start); if (idx === -1) return false; const afterIdx = idx + token.length; + if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; } const afterChar = afterIdx < m.length ? m[afterIdx] : ""; if (afterChar === "" || afterChar === "-") return true; start = afterIdx; @@ -105,6 +112,7 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean { const idx = m.indexOf(token, start); if (idx === -1) return false; const afterIdx = idx + token.length; + if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; } const suffix = m.slice(afterIdx); if (suffix === "") return true; if (!suffix.startsWith("-")) { start = afterIdx; continue; } @@ -159,6 +167,22 @@ const EXACT_RECORDS = new Map([ databricksV2WireRoute: "anthropic-messages", normalizationPolicy: "none", }], + ["databricks_v2::databricks-gpt-5-6-luna", { + registryLabel: "GPT-5.6 Luna", + thinkingMode: "none", + supportedEfforts: ["low", "medium", "high"] as const, + defaultEffort: "medium", + databricksV2WireRoute: "openai-responses", + normalizationPolicy: "openai-standard", + }], + ["databricks_v2::databricks-gpt-5-6-terra", { + registryLabel: "GPT-5.6 Terra", + thinkingMode: "none", + supportedEfforts: ["low", "medium", "high"] as const, + defaultEffort: "medium", + databricksV2WireRoute: "openai-responses", + normalizationPolicy: "openai-standard", + }], ]); // --------------------------------------------------------------------------- @@ -737,6 +761,17 @@ function lookupByFamilyRules(provider: string, normalized: string): CapabilityRe normalizationPolicy: "openai-standard", }; } + // rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 6 + if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("gpt") || lower.split(/[^a-z0-9]+/).includes("gpt5"))) { + return { + registryLabel: null, + thinkingMode: "none", + supportedEfforts: ["none", "minimal", "low", "medium", "high", "xhigh"] as const, + defaultEffort: "medium", + databricksV2WireRoute: "openai-responses", + normalizationPolicy: "openai-clamp-max-to-xhigh", + }; + } // rule: dbv2-claude-code-names-segment, provider: databricks_v2, priority: 5 if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("claude") || lower.split(/[^a-z0-9]+/).includes("opus") || lower.split(/[^a-z0-9]+/).includes("sonnet") || lower.split(/[^a-z0-9]+/).includes("haiku") || lower.split(/[^a-z0-9]+/).includes("mythos") || lower.split(/[^a-z0-9]+/).includes("fable"))) { return { @@ -748,17 +783,6 @@ function lookupByFamilyRules(provider: string, normalized: string): CapabilityRe normalizationPolicy: "none", }; } - // rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 5 - if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).some(s => s.startsWith("gpt")))) { - return { - registryLabel: null, - thinkingMode: "none", - supportedEfforts: ["none", "minimal", "low", "medium", "high", "xhigh"] as const, - defaultEffort: "medium", - databricksV2WireRoute: "openai-responses", - normalizationPolicy: "openai-clamp-max-to-xhigh", - }; - } // rule: dbv2-sol-luna-terra-segment, provider: databricks_v2, priority: 5 if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("sol") || lower.split(/[^a-z0-9]+/).includes("luna") || lower.split(/[^a-z0-9]+/).includes("terra"))) { return { @@ -792,8 +816,8 @@ export function resolveModelCapabilities( provider: string, rawModelId: string, ): CapabilityResult { - // Step 1: raw exact lookup - const exactKey = `${provider}::${rawModelId}`; + // Step 1: raw exact lookup (case-insensitive — keys lowercased at build time) + const exactKey = `${provider.toLowerCase()}::${rawModelId.toLowerCase()}`; const exact = EXACT_RECORDS.get(exactKey); if (exact) return exact; diff --git a/scripts/generate-databricks-model-names.py b/scripts/generate-databricks-model-names.py index 43f0e5c2e..9863de5f8 100755 --- a/scripts/generate-databricks-model-names.py +++ b/scripts/generate-databricks-model-names.py @@ -29,7 +29,7 @@ TS_OUT = REPO_ROOT / "desktop/src/features/agents/lib/databricksModelNames.ts" # Allowed characters in endpoint IDs and curated names. SAFE_ID_RE = re.compile(r"^[a-z0-9][a-z0-9.\-]*$") -SAFE_NAME_RE = re.compile(r"^[^\x00-\x1f\"\\<>&]*$") +SAFE_NAME_RE = re.compile(r"^[^\x00-\x1f\"\\<>&]+$") def fetch(url: str) -> bytes: diff --git a/scripts/generate-model-capabilities.mjs b/scripts/generate-model-capabilities.mjs index 1ac36cba0..394a12964 100644 --- a/scripts/generate-model-capabilities.mjs +++ b/scripts/generate-model-capabilities.mjs @@ -367,7 +367,7 @@ for (const [provider, fb] of Object.entries(manifest.provider_fallbacks)) { } } -// Validate exact_records — check for duplicate (provider, raw_model_id) keys +// Validate exact_records — full invariant checks on every override axis const seenExactKeys = new Set(); for (const rec of manifest.exact_records ?? []) { if (!rec.provider || !rec.raw_model_id) @@ -375,6 +375,60 @@ for (const rec of manifest.exact_records ?? []) { const key = `${rec.provider}::${rec.raw_model_id}`; if (seenExactKeys.has(key)) throw new Error(`duplicate exact_record key: ${key}`); seenExactKeys.add(key); + + // Validate match_priority: must be a non-negative integer if present (injected into comments). + if (rec.match_priority !== undefined) { + if (!Number.isInteger(rec.match_priority) || rec.match_priority < 0) + throw new Error(`exact_record ${key}: match_priority must be a non-negative integer`); + } + + // Validate supported_efforts_override if present. + if (rec.supported_efforts_override !== undefined) { + assertNonEmpty(rec.supported_efforts_override, `exact_record ${key} supported_efforts_override`); + const seenEfforts = new Set(); + for (const e of rec.supported_efforts_override) { + assertEnum(e, VALID_EFFORTS, `exact_record ${key} supported_efforts_override[]`); + if (seenEfforts.has(e)) + throw new Error(`exact_record ${key}: duplicate effort "${e}" in supported_efforts_override`); + seenEfforts.add(e); + } + // Validate ordering matches canonical effort order (Rust clamp assumes sorted). + const canonicalIndices = rec.supported_efforts_override.map((e) => VALID_EFFORTS.indexOf(e)); + for (let i = 1; i < canonicalIndices.length; i++) { + if (canonicalIndices[i] <= canonicalIndices[i - 1]) { + throw new Error( + `exact_record ${key}: supported_efforts_override must follow canonical order [${VALID_EFFORTS.join(", ")}]; got [${rec.supported_efforts_override.join(", ")}]`, + ); + } + } + // Validate default_effort is in the override if present. + if (rec.default_effort !== undefined && rec.default_effort !== null) { + assertEnum(rec.default_effort, VALID_EFFORTS, `exact_record ${key} default_effort`); + if (!rec.supported_efforts_override.includes(rec.default_effort)) { + throw new Error( + `exact_record ${key}: default_effort "${rec.default_effort}" not in supported_efforts_override [${rec.supported_efforts_override.join(", ")}]`, + ); + } + } + } else if (rec.default_effort !== undefined && rec.default_effort !== null) { + // default_effort override without supported_efforts_override — still validate enum. + assertEnum(rec.default_effort, VALID_EFFORTS, `exact_record ${key} default_effort`); + } + + // Validate thinking_mode override if present. + if (rec.thinking_mode !== undefined) { + assertEnum(rec.thinking_mode, VALID_THINKING_MODES, `exact_record ${key} thinking_mode`); + } + + // Validate databricks_v2_wire_route override if present. + if (rec.databricks_v2_wire_route !== undefined) { + assertEnum(rec.databricks_v2_wire_route, VALID_DBV2_ROUTES, `exact_record ${key} databricks_v2_wire_route`); + } + + // Validate normalization_policy override if present. + if (rec.normalization_policy !== undefined) { + assertEnum(rec.normalization_policy, VALID_NORM_POLICIES, `exact_record ${key} normalization_policy`); + } } // --------------------------------------------------------------------------- @@ -398,17 +452,33 @@ function stripCatalogPrefix(model) { return firstIdx === Infinity ? model : model.slice(firstIdx); } +/** + * Returns true if the character at position idx-1 in str is a valid left boundary: + * start-of-string, "-", or ".". Prevents substring matches like "customgpt" matching "gpt". + */ +function hasLeftBoundary(str, idx) { + if (idx === 0) return true; + const prev = str[idx - 1]; + return prev === "-" || prev === "."; +} + /** * gpt5-token match: model contains token at a word boundary (end-of-string or "-"). * Does NOT match if followed by a digit or letter. + * Left-boundary checked: token must start at beginning of string or after "-" or ".". */ function gpt5TokenMatches(model, token) { const lower = model.toLowerCase(); + const tok = token.toLowerCase(); let start = 0; while (true) { - const idx = lower.indexOf(token.toLowerCase(), start); + const idx = lower.indexOf(tok, start); if (idx === -1) return false; - const afterIdx = idx + token.length; + const afterIdx = idx + tok.length; + if (!hasLeftBoundary(lower, idx)) { + start = afterIdx; + continue; + } const afterChar = afterIdx < lower.length ? lower[afterIdx] : ""; if (afterChar === "" || afterChar === "-") return true; start = afterIdx; @@ -417,14 +487,20 @@ function gpt5TokenMatches(model, token) { /** * gpt5-base match: like gpt5-token but also rejects short -<1-3 digit> suffixes. + * Left-boundary checked: token must start at beginning of string or after "-" or ".". */ function gpt5BaseMatches(model, token) { const lower = model.toLowerCase(); + const tok = token.toLowerCase(); let start = 0; while (true) { - const idx = lower.indexOf(token.toLowerCase(), start); + const idx = lower.indexOf(tok, start); if (idx === -1) return false; - const afterIdx = idx + token.length; + const afterIdx = idx + tok.length; + if (!hasLeftBoundary(lower, idx)) { + start = afterIdx; + continue; + } const suffix = lower.slice(afterIdx); if (suffix === "") return true; if (!suffix.startsWith("-")) { @@ -776,11 +852,14 @@ pub struct CapabilityResult { /// Returns the exact capability record for a provider-qualified raw model ID, /// if one exists in the manifest. This is checked BEFORE any prefix stripping. +/// Matching is case-insensitive — both inputs are lowercased before comparison. pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option { - match (provider, raw_model_id) { + let prov_lc = provider.to_lowercase(); + let id_lc = raw_model_id.to_lowercase(); + match (prov_lc.as_str(), id_lc.as_str()) { ${exactMapEntries .map(({ rec, clean, provNote }) => { - return ` ("${rec.provider}", "${rec.raw_model_id}") => { + return ` ("${rec.provider.toLowerCase()}", "${rec.raw_model_id.toLowerCase()}") => { ${provNote} Some( ${emitRustCapabilityResult(clean, " ")} @@ -1014,7 +1093,9 @@ const rustGpt5Helpers = ` // gpt5 boundary-aware token helpers (used by generated family resolver) // --------------------------------------------------------------------------- -/// Returns true if \`model\` contains \`token\` at a word boundary (end-of-string or "-"). +/// Returns true if \`model\` contains \`token\` at left+right word boundaries. +/// Left boundary: start-of-string or preceded by '-' or '.'. +/// Right boundary: end-of-string or followed by '-'. /// Does not match if followed immediately by a digit or letter. fn gpt5_token_matches_rs(model: &str, token: &str) -> bool { let lower = model; @@ -1026,6 +1107,15 @@ fn gpt5_token_matches_rs(model: &str, token: &str) -> bool { Some(rel_idx) => { let abs_idx = start + rel_idx; let after_idx = abs_idx + tok_lower.len(); + // Left-boundary check: must start at string start or after '-' or '.'. + let left_ok = abs_idx == 0 || { + let prev = lower.as_bytes()[abs_idx - 1]; + prev == b'-' || prev == b'.' + }; + if !left_ok { + start = after_idx; + continue; + } let after_char = lower[after_idx..].chars().next(); match after_char { None | Some('-') => return true, @@ -1047,6 +1137,15 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { Some(rel_idx) => { let abs_idx = start + rel_idx; let after_idx = abs_idx + tok_lower.len(); + // Left-boundary check: must start at string start or after '-' or '.'. + let left_ok = abs_idx == 0 || { + let prev = lower.as_bytes()[abs_idx - 1]; + prev == b'-' || prev == b'.' + }; + if !left_ok { + start = after_idx; + continue; + } let suffix = &lower[after_idx..]; if suffix.is_empty() { return true; @@ -1056,16 +1155,12 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool { continue; } let dash_rest = &suffix[1..]; - // Reject -<1-3 digits> that look like version numbers. - let is_short_version = dash_rest - .chars() - .take(4) - .enumerate() - .all(|(i, c)| { - if i < 3 { c.is_ascii_digit() } - else { !c.is_ascii_alphanumeric() } - }) - && dash_rest.chars().next().is_some_and(|c| c.is_ascii_digit()); + // Reject -<1-3 digits> followed by non-alphanumeric or end (mirrors TS /^\\d{1,3}(?:[^a-z\\d]|$)/i). + let first_non_digit = dash_rest.find(|c: char| !c.is_ascii_digit()).unwrap_or(dash_rest.len()); + let is_short_version = first_non_digit >= 1 + && first_non_digit <= 3 + && (first_non_digit == dash_rest.len() + || !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric()); if is_short_version { start = after_idx; continue; @@ -1281,12 +1376,19 @@ ${registryLabelsArr // gpt5 boundary-aware token helpers // --------------------------------------------------------------------------- +function hasLeftBoundaryGenerated(m: string, idx: number): boolean { + if (idx === 0) return true; + const prev = m[idx - 1]; + return prev === "-" || prev === "."; +} + function gpt5TokenMatchesGenerated(m: string, token: string): boolean { let start = 0; while (true) { const idx = m.indexOf(token, start); if (idx === -1) return false; const afterIdx = idx + token.length; + if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; } const afterChar = afterIdx < m.length ? m[afterIdx] : ""; if (afterChar === "" || afterChar === "-") return true; start = afterIdx; @@ -1299,6 +1401,7 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean { const idx = m.indexOf(token, start); if (idx === -1) return false; const afterIdx = idx + token.length; + if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; } const suffix = m.slice(afterIdx); if (suffix === "") return true; if (!suffix.startsWith("-")) { start = afterIdx; continue; } @@ -1315,7 +1418,8 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean { const EXACT_RECORDS = new Map([ ${tsExactEntries .map(({ rec, clean }) => { - return ` ["${rec.provider}::${rec.raw_model_id}", ${emitTsCapabilityResult(clean, " ")}],`; + // Keys are lowercased at build time; resolveModelCapabilities lowercases at lookup time. + return ` ["${rec.provider.toLowerCase()}::${rec.raw_model_id.toLowerCase()}", ${emitTsCapabilityResult(clean, " ")}],`; }) .join("\n")} ]); @@ -1376,8 +1480,8 @@ export function resolveModelCapabilities( provider: string, rawModelId: string, ): CapabilityResult { - // Step 1: raw exact lookup - const exactKey = \`\${provider}::\${rawModelId}\`; + // Step 1: raw exact lookup (case-insensitive — keys lowercased at build time) + const exactKey = \`\${provider.toLowerCase()}::\${rawModelId.toLowerCase()}\`; const exact = EXACT_RECORDS.get(exactKey); if (exact) return exact; @@ -1448,5 +1552,5 @@ if (CHECK_MODE && checkFailed) { process.exit(1); } if (!CHECK_MODE) { - console.log("Done. Generated 3 files."); + console.log("Done. Generated 2 files."); } diff --git a/scripts/model-capabilities.json b/scripts/model-capabilities.json index baac0a5d7..7ab43ebb9 100644 --- a/scripts/model-capabilities.json +++ b/scripts/model-capabilities.json @@ -1,5 +1,4 @@ { - "$schema": "./model-capabilities-schema.json", "_comment": "Hand-curated model capability manifest. Edit here; run scripts/generate-model-capabilities.mjs to regenerate artifacts.", "_generated_by": "scripts/generate-model-capabilities.mjs", "_sources": { @@ -437,13 +436,13 @@ }, { "id": "dbv2-gpt-code-names-segment", - "_comment": "DBv2-only rule: endpoint names containing a GPT segment prefix (gpt*) route via OpenAI Responses. Handles 'gpt', 'gpt5', 'gpt-5' segments. Priority < individual gpt5 family rules so explicit families take precedence.", - "match_kind": "segment-prefix", + "_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj/gpt-neox false-positives.", + "match_kind": "segment", "match_value": "gpt", "providers": [ "databricks_v2" ], - "match_priority": 5, + "match_priority": 6, "thinking_mode": "none", "supported_efforts": [ "none", @@ -455,7 +454,10 @@ ], "default_effort": "medium", "databricks_v2_wire_route": "openai-responses", - "normalization_policy": "openai-clamp-max-to-xhigh" + "normalization_policy": "openai-clamp-max-to-xhigh", + "match_aliases": [ + "gpt5" + ] }, { "id": "dbv2-sol-luna-terra-segment", @@ -678,6 +680,34 @@ "_reconciliation": "no-effort-divergence", "_reconciliation_note": "models.dev advertises reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]. This is a different capability axis (extended thinking token budget), not an effort-level selector. No effort divergence to reconcile \u2014 efforts for this model come from the anthropic family rule (anthropic-adaptive-xhigh-opus-4-7).", "_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-claude-opus-4-7\"].reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]" + }, + { + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-6-luna", + "registry_label": "GPT-5.6 Luna", + "supported_efforts_override": [ + "low", + "medium", + "high" + ], + "source": "models.dev reasoning_options: low|medium|high", + "_reconciliation": "adopt", + "_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; luna endpoint does not expose none, xhigh, or max. Provider-advertised wins.", + "_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-luna\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]" + }, + { + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-6-terra", + "registry_label": "GPT-5.6 Terra", + "supported_efforts_override": [ + "low", + "medium", + "high" + ], + "source": "models.dev reasoning_options: low|medium|high", + "_reconciliation": "adopt", + "_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; terra endpoint does not expose none, xhigh, or max. Provider-advertised wins.", + "_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-terra\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]" } ], "provider_fallbacks": { diff --git a/scripts/normative-corpus.json b/scripts/normative-corpus.json index cf44b01c1..e30eed151 100644 --- a/scripts/normative-corpus.json +++ b/scripts/normative-corpus.json @@ -791,5 +791,233 @@ "default_effort": "medium", "databricks_v2_wire_route": "not-applicable" } + }, + { + "_group": "gpt-5-base guard divergence fix (CRITICAL)", + "_note": "These vectors cover the 1-2 digit version suffix window where Rust and TS diverged. Must reject from gpt5-base, fall to concrete-unknown." + }, + { + "id": "openai-gpt5-10-preview-reject-base", + "provider": "openai", + "raw_model_id": "gpt-5-10-preview", + "_note": "CRITICAL divergence fix: -10- is a 2-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "expect": { + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + { + "id": "openai-gpt5-2-mini-reject-base", + "provider": "openai", + "raw_model_id": "gpt-5-2-mini", + "_note": "CRITICAL divergence fix: -2- is a 1-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "expect": { + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + { + "id": "openai-gpt5-9-dot-1-reject-base", + "provider": "openai", + "raw_model_id": "gpt-5-9.1", + "_note": "CRITICAL divergence fix: -9 followed by '.' is a 1-digit suffix + non-alnum \u2192 gpt5-base rejects. Falls to openai concrete-unknown.", + "expect": { + "supported_efforts": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + { + "id": "openai-customgpt-5-5-no-token-match", + "provider": "openai", + "raw_model_id": "customgpt-5-5-endpoint", + "_note": "Left-boundary fix context: customgpt-5-5-endpoint strips to gpt-5-5-endpoint which DOES match gpt5-5 family. This is correct. Update: expect gpt5-5 efforts.", + "expect": { + "supported_efforts": [ + "none", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + { + "_group": "DBv2 gpt segment rule boundary vectors", + "_note": "Verify segment match (not segment-prefix): 'gpt' must be an exact segment, not just a prefix of a segment." + }, + { + "id": "dbv2-gptoss-not-responses", + "provider": "databricks_v2", + "raw_model_id": "gptoss-model", + "_note": "Collision-negative: 'gptoss' is a segment starting with 'gpt' but NOT an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.", + "expect": { + "databricks_v2_wire_route": "mlflow-chat" + } + }, + { + "id": "dbv2-gptj-6b-not-responses", + "provider": "databricks_v2", + "raw_model_id": "gptj-6b", + "_note": "Collision-negative: 'gptj' is not an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.", + "expect": { + "databricks_v2_wire_route": "mlflow-chat" + } + }, + { + "id": "dbv2-customgpt-not-responses", + "provider": "databricks_v2", + "raw_model_id": "customgpt-5-5-endpoint", + "_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint \u2192 gpt5-5 family \u2192 openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.", + "expect": { + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "dbv2-gpt5-segment-positive", + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt5-custom", + "_note": "DBv2 gpt segment rule positive: normalized 'gpt5-custom' \u2192 segment 'gpt5' IS an exact match in match_aliases. Routes openai-responses.", + "expect": { + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "dbv2-dual-marker-gpt-wins-openai", + "provider": "databricks_v2", + "raw_model_id": "gpt-opus-5", + "_note": "Dual-marker: normalized 'gpt-opus-5' \u2192 segments include 'gpt' AND 'opus'. Priority 6 (gpt) > 5 (claude): OpenAI wins. Must route openai-responses.", + "expect": { + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "_group": "Missing positive vectors" + }, + { + "id": "anthropic-opus-5-adaptive-xhigh", + "provider": "anthropic", + "raw_model_id": "claude-opus-5-20270101", + "_note": "Opus-5 prefix rule: adaptive, supports xhigh+max. Verifies the rule is wired.", + "expect": { + "thinking_mode": "adaptive", + "supported_efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "default_effort": "high" + } + }, + { + "id": "dbv2-sol-normalization-policy", + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-6-sol", + "_note": "Sol exact record: normalization_policy comes from gpt5-6 family rule (openai-standard). Also checks supported_efforts override.", + "expect": { + "normalization_policy": "openai-standard", + "supported_efforts": [ + "low", + "medium", + "high", + "max" + ] + } + }, + { + "id": "dbv2-luna-segment-route", + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-6-luna", + "_note": "Luna exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).", + "expect": { + "supported_efforts": [ + "low", + "medium", + "high" + ], + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "dbv2-terra-segment-route", + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-6-terra", + "_note": "Terra exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).", + "expect": { + "supported_efforts": [ + "low", + "medium", + "high" + ], + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "dbv2-gpt5-4-nano-exact", + "provider": "databricks_v2", + "raw_model_id": "databricks-gpt-5-4-nano", + "_note": "Exact record: gpt-5-4-nano, efforts [low,medium,high], registry_label 'GPT-5.4 Nano'.", + "expect": { + "supported_efforts": [ + "low", + "medium", + "high" + ], + "registry_label": "GPT-5.4 Nano", + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "openrouter-concrete-unknown-fallback", + "provider": "openrouter", + "raw_model_id": "some-model-xyz", + "_note": "openrouter concrete-unknown \u2192 _default fallback (wire route not-applicable).", + "expect": { + "databricks_v2_wire_route": "not-applicable" + } + }, + { + "id": "openai-gpt5-pro-uppercase-provider", + "provider": "OpenAI", + "raw_model_id": "gpt-5-pro", + "_note": "Casing: uppercase provider 'OpenAI' normalized to 'openai'. Must resolve same as openai/gpt-5-pro.", + "expect": { + "supported_efforts": [ + "high" + ], + "default_effort": "high" + } + }, + { + "id": "dbv2-exact-record-uppercase-model", + "provider": "databricks_v2", + "raw_model_id": "DATABRICKS-GPT-5-4-NANO", + "_note": "Casing: uppercase raw_model_id. Case-insensitive exact lookup must hit the lowercase record.", + "expect": { + "supported_efforts": [ + "low", + "medium", + "high" + ] + } } ] diff --git a/scripts/run-corpus.mjs b/scripts/run-corpus.mjs index bcec37cb8..1037a8f30 100644 --- a/scripts/run-corpus.mjs +++ b/scripts/run-corpus.mjs @@ -33,14 +33,14 @@ const corpus = JSON.parse( // ----- Provider alias canonicalization ----- // Mirrors production canonicalizeProvider() in desktop/src/features/agents/lib/formatAgentModelLabel.ts. // Applied before every generated lookup so alias vectors (e.g. "openai-compat") pass both interpreters. -const PROVIDER_ALIASES = { - "databricks-v2": "databricks_v2", - "openai-compat": "openai", -}; +const PROVIDER_ALIASES = new Map([ + ["databricks-v2", "databricks_v2"], + ["openai-compat", "openai"], +]); function canonicalizeProvider(provider) { const normalized = (provider ?? "").trim().toLowerCase(); - return PROVIDER_ALIASES[normalized] ?? normalized; + return PROVIDER_ALIASES.get(normalized) ?? normalized; } // ----- Run corpus ----- diff --git a/scripts/test-manifest-validator.mjs b/scripts/test-manifest-validator.mjs index cbe596a76..fafab5051 100644 --- a/scripts/test-manifest-validator.mjs +++ b/scripts/test-manifest-validator.mjs @@ -410,4 +410,132 @@ test("schema-negative: exact_record registry_label with unsafe chars is rejected ); }); +// --------------------------------------------------------------------------- +// Rule: exact_record supported_efforts_override must be non-empty +// --------------------------------------------------------------------------- +test("schema-negative: exact_record empty supported_efforts_override is rejected", () => { + assertRejects( + "exact_record empty supported_efforts_override", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.supported_efforts_override = []; + }), + "supported_efforts_override", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record supported_efforts_override with invalid enum value +// --------------------------------------------------------------------------- +test("schema-negative: exact_record supported_efforts_override with bogus enum is rejected", () => { + assertRejects( + "exact_record bogus effort enum", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.supported_efforts_override = ["ultra-high"]; + }), + "supported_efforts_override", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record supported_efforts_override with duplicate effort +// --------------------------------------------------------------------------- +test("schema-negative: exact_record supported_efforts_override with duplicate effort is rejected", () => { + assertRejects( + "exact_record duplicate effort", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.supported_efforts_override = ["low", "low"]; + }), + "duplicate", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record supported_efforts_override must follow canonical order +// --------------------------------------------------------------------------- +test("schema-negative: exact_record supported_efforts_override out of canonical order is rejected", () => { + assertRejects( + "exact_record efforts out of order", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + // Reverse order — [high, medium, low] is not canonical [low, medium, high] + rec.supported_efforts_override = ["high", "medium", "low"]; + }), + "canonical order", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record default_effort not in supported_efforts_override +// --------------------------------------------------------------------------- +test("schema-negative: exact_record default_effort not in supported_efforts_override is rejected", () => { + assertRejects( + "exact_record default_effort outside override", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.supported_efforts_override = ["low", "medium"]; + rec.default_effort = "high"; // not in override + }), + "default_effort", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record match_priority must be a non-negative integer +// --------------------------------------------------------------------------- +test("schema-negative: exact_record match_priority non-integer is rejected", () => { + assertRejects( + "exact_record match_priority non-integer", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.match_priority = "five"; + }), + "match_priority", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record thinking_mode must be a valid enum +// --------------------------------------------------------------------------- +test("schema-negative: exact_record invalid thinking_mode is rejected", () => { + assertRejects( + "exact_record invalid thinking_mode", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.thinking_mode = "turbo-thinking"; + }), + "thinking_mode", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record databricks_v2_wire_route must be a valid enum +// --------------------------------------------------------------------------- +test("schema-negative: exact_record invalid databricks_v2_wire_route is rejected", () => { + assertRejects( + "exact_record invalid wire_route", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.databricks_v2_wire_route = "http-sse"; + }), + "databricks_v2_wire_route", + ); +}); + +// --------------------------------------------------------------------------- +// Rule: exact_record normalization_policy must be a valid enum +// --------------------------------------------------------------------------- +test("schema-negative: exact_record invalid normalization_policy is rejected", () => { + assertRejects( + "exact_record invalid normalization_policy", + mutate((m) => { + const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini"); + rec.normalization_policy = "pass-through-all"; + }), + "normalization_policy", + ); +}); + console.log("\nSchema-negative validator tests complete.");