fix(models): address Kalvin-review: gpt-segment/left-boundary/luna-terra/validator

CRITICAL: Fix Rust gpt5-base short-version guard to mirror TS regex semantics.
The old 4-char window check diverged from TS /^\d{1,3}(?:[^a-z\d]|$)/i for
inputs like gpt-5-10-preview (10 digits). Replaced with find(non-digit) approach
that exactly matches the TS digit-run semantics. Add left-boundary guards (start
or preceded by '-'/'.') to both gpt5_token_matches_rs and gpt5_base_matches_rs
to prevent sgpt-model-class false-positives.

IMPORTANT: Add exact_records validation to generator — enum checks for all
optional override axes, non-empty override check, canonical-order enforcement,
default_effort-in-override check, match_priority non-negative-int check
(injected into source comments, treated as injection surface). Add 9 schema-
negative tests in test-manifest-validator.mjs.

IMPORTANT: Restrict DBv2 gpt segment rule from starts_with("gpt") to exact
segment match ("gpt" or "gpt5"). Raise priority 5->6 to restore old dual-marker
OpenAI-before-Claude contract. Add left-boundary guards to token helpers. Add
collision-negative corpus vectors: gptoss-model, gptj-6b, gpt-neox,
customgpt-5-5-endpoint.

MINOR: Restore .trim() on model string before resolveModelCapabilities call in
buzzAgentConfig.ts. Make exact-record lookup case-insensitive (lowercase keys at
build + lookup). Extend run-corpus.mjs to compare all 6 axes including
normalization_policy and registry_label. Add positive corpus vectors: opus-5
rule, dbv2 gpt-segment, sol/luna/terra, gpt-5-4-nano exact record, openrouter/
unknown/legacy-databricks fallbacks, case-insensitive lookup vectors. Fix Rust
corpus test harness provider lowercasing to handle vectors like provider=OpenAI.

HYGIENE: Fix "Generated 3 files" -> "Generated 2 files" message. Remove dead
$schema pointer from manifest. Add module doc to generated_model_capabilities_tests.rs
noting hand-maintained status. Switch PROVIDER_ALIASES from plain object to Map
in formatAgentModelLabel.ts and run-corpus.mjs. Align CI node-version to 24.
Tighten python SAFE_NAME_RE from * to + to reject empty display names.

luna/terra: models.dev confirms databricks-gpt-5-6-luna and -terra advertise
[low,medium,high] (differs from sol's [low,medium,high,max]). Added two exact
records with supported_efforts=[low,medium,high], default_effort=medium.

Co-authored-by: Will Pfleger <pfleger.will@gmail.com>
Signed-off-by: Will Pfleger <pfleger.will@gmail.com>
This commit is contained in:
npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7
2026-08-04 13:15:11 -04:00
co-authored by Will Pfleger
parent c898da7149
commit 8ba3814854
12 changed files with 675 additions and 88 deletions
+1 -1
View File
@@ -146,7 +146,7 @@ jobs:
- name: Set up Node.js
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
with:
node-version: '22'
node-version: '24'
package-manager-cache: false
- name: Regenerate artifacts
@@ -94,8 +94,11 @@ pub struct CapabilityResult {
/// Returns the exact capability record for a provider-qualified raw model ID,
/// if one exists in the manifest. This is checked BEFORE any prefix stripping.
/// Matching is case-insensitive — both inputs are lowercased before comparison.
pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option<CapabilityResult> {
match (provider, raw_model_id) {
let prov_lc = provider.to_lowercase();
let id_lc = raw_model_id.to_lowercase();
match (prov_lc.as_str(), id_lc.as_str()) {
("databricks_v2", "databricks-gpt-5-4-mini") => {
// provenance: exact(databricks_v2::databricks-gpt-5-4-mini)
// registry_label: exact_record
@@ -204,6 +207,48 @@ pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option<CapabilityResu
normalization_policy: NormalizationPolicy::None,
})
}
("databricks_v2", "databricks-gpt-5-6-luna") => {
// provenance: exact(databricks_v2::databricks-gpt-5-6-luna)
// registry_label: exact_record
// supported_efforts: exact_record
// databricks_v2_wire_route: family:openai-gpt5-6@15
// thinking_mode: family:openai-gpt5-6@15
// normalization_policy: family:openai-gpt5-6@15
// default_effort: family:openai-gpt5-6@15
Some(CapabilityResult {
registry_label: Some("GPT-5.6 Luna"),
thinking_mode: ThinkingMode::None,
supported_efforts: Cow::Borrowed(&[
ThinkingEffort::Low,
ThinkingEffort::Medium,
ThinkingEffort::High,
]),
default_effort: Some(ThinkingEffort::Medium),
databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses,
normalization_policy: NormalizationPolicy::OpenAiStandard,
})
}
("databricks_v2", "databricks-gpt-5-6-terra") => {
// provenance: exact(databricks_v2::databricks-gpt-5-6-terra)
// registry_label: exact_record
// supported_efforts: exact_record
// databricks_v2_wire_route: family:openai-gpt5-6@15
// thinking_mode: family:openai-gpt5-6@15
// normalization_policy: family:openai-gpt5-6@15
// default_effort: family:openai-gpt5-6@15
Some(CapabilityResult {
registry_label: Some("GPT-5.6 Terra"),
thinking_mode: ThinkingMode::None,
supported_efforts: Cow::Borrowed(&[
ThinkingEffort::Low,
ThinkingEffort::Medium,
ThinkingEffort::High,
]),
default_effort: Some(ThinkingEffort::Medium),
databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses,
normalization_policy: NormalizationPolicy::OpenAiStandard,
})
}
_ => None,
}
}
@@ -957,6 +1002,31 @@ pub fn lookup_by_family_rules(provider: &str, normalized: &str) -> Option<Capabi
normalization_policy: NormalizationPolicy::OpenAiStandard,
});
}
// rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 6
if provider == "databricks_v2"
&& (lower
.split(|c: char| !c.is_ascii_alphanumeric())
.any(|s| s == "gpt")
|| lower
.split(|c: char| !c.is_ascii_alphanumeric())
.any(|s| s == "gpt5"))
{
return Some(CapabilityResult {
registry_label: None,
thinking_mode: ThinkingMode::None,
supported_efforts: Cow::Borrowed(&[
ThinkingEffort::None,
ThinkingEffort::Minimal,
ThinkingEffort::Low,
ThinkingEffort::Medium,
ThinkingEffort::High,
ThinkingEffort::XHigh,
]),
default_effort: Some(ThinkingEffort::Medium),
databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses,
normalization_policy: NormalizationPolicy::OpenAiClampMaxToXHigh,
});
}
// rule: dbv2-claude-code-names-segment, provider: databricks_v2, priority: 5
if provider == "databricks_v2"
&& (lower
@@ -993,28 +1063,6 @@ pub fn lookup_by_family_rules(provider: &str, normalized: &str) -> Option<Capabi
normalization_policy: NormalizationPolicy::None,
});
}
// rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 5
if provider == "databricks_v2"
&& (lower
.split(|c: char| !c.is_ascii_alphanumeric())
.any(|s| s.starts_with("gpt")))
{
return Some(CapabilityResult {
registry_label: None,
thinking_mode: ThinkingMode::None,
supported_efforts: Cow::Borrowed(&[
ThinkingEffort::None,
ThinkingEffort::Minimal,
ThinkingEffort::Low,
ThinkingEffort::Medium,
ThinkingEffort::High,
ThinkingEffort::XHigh,
]),
default_effort: Some(ThinkingEffort::Medium),
databricks_v2_wire_route: DatabricksV2Route::OpenAiResponses,
normalization_policy: NormalizationPolicy::OpenAiClampMaxToXHigh,
});
}
// rule: dbv2-sol-luna-terra-segment, provider: databricks_v2, priority: 5
if provider == "databricks_v2"
&& (lower
@@ -1334,7 +1382,9 @@ pub const DATABRICKS_MODEL_NAMES: &[(&str, &str)] = &[
// gpt5 boundary-aware token helpers (used by generated family resolver)
// ---------------------------------------------------------------------------
/// Returns true if `model` contains `token` at a word boundary (end-of-string or "-").
/// Returns true if `model` contains `token` at left+right word boundaries.
/// Left boundary: start-of-string or preceded by '-' or '.'.
/// Right boundary: end-of-string or followed by '-'.
/// Does not match if followed immediately by a digit or letter.
fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
let lower = model;
@@ -1346,6 +1396,15 @@ fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
Some(rel_idx) => {
let abs_idx = start + rel_idx;
let after_idx = abs_idx + tok_lower.len();
// Left-boundary check: must start at string start or after '-' or '.'.
let left_ok = abs_idx == 0 || {
let prev = lower.as_bytes()[abs_idx - 1];
prev == b'-' || prev == b'.'
};
if !left_ok {
start = after_idx;
continue;
}
let after_char = lower[after_idx..].chars().next();
match after_char {
None | Some('-') => return true,
@@ -1367,6 +1426,15 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool {
Some(rel_idx) => {
let abs_idx = start + rel_idx;
let after_idx = abs_idx + tok_lower.len();
// Left-boundary check: must start at string start or after '-' or '.'.
let left_ok = abs_idx == 0 || {
let prev = lower.as_bytes()[abs_idx - 1];
prev == b'-' || prev == b'.'
};
if !left_ok {
start = after_idx;
continue;
}
let suffix = &lower[after_idx..];
if suffix.is_empty() {
return true;
@@ -1376,15 +1444,14 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool {
continue;
}
let dash_rest = &suffix[1..];
// Reject -<1-3 digits> that look like version numbers.
let is_short_version =
dash_rest.chars().take(4).enumerate().all(|(i, c)| {
if i < 3 {
c.is_ascii_digit()
} else {
!c.is_ascii_alphanumeric()
}
}) && dash_rest.chars().next().is_some_and(|c| c.is_ascii_digit());
// Reject -<1-3 digits> followed by non-alphanumeric or end (mirrors TS /^\d{1,3}(?:[^a-z\d]|$)/i).
let first_non_digit = dash_rest
.find(|c: char| !c.is_ascii_digit())
.unwrap_or(dash_rest.len());
let is_short_version = first_non_digit >= 1
&& first_non_digit <= 3
&& (first_non_digit == dash_rest.len()
|| !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric());
if is_short_version {
start = after_idx;
continue;
@@ -1,6 +1,7 @@
//! Tests for generated model capabilities — normative corpus + handwritten supplements.
//!
//! This module is conditionally compiled as #[cfg(test)] from generated_model_capabilities.rs.
//! It is hand-maintained (not regenerated) and lives outside the regen-diff gate.
//!
//! Three layers:
//! 1. Shared normative corpus executed against the Rust interpreter — every vector in
@@ -139,8 +140,10 @@ mod shared_corpus_tests {
// Canonicalize provider aliases before resolving — mirrors the
// production path where Rust normalizes "openai-compat" → Provider::OpenAi
// (config.rs) and the TS canonicalizeProvider() resolves "databricks-v2"
// → "databricks_v2" before generated lookups.
let canonical_provider = match provider {
// → "databricks_v2" before generated lookups. Lowercase first so corpus
// vectors like provider="OpenAI" test case-insensitive normalization.
let provider_lc = provider.to_lowercase();
let canonical_provider = match provider_lc.as_str() {
"openai-compat" => "openai",
"databricks-v2" => "databricks_v2",
other => other,
@@ -10,11 +10,14 @@ import {
* - "openai-compat" → "openai": Rust already accepts "openai-compat" as Provider::OpenAi
* (crates/buzz-agent/src/config.rs); TS must canonicalize identically so the UI shows
* the same effort table that the Rust request path will apply.
*
* Map is used (not a plain object) to avoid prototype-chain collisions
* ("constructor", "__proto__", etc.) silently resolving to a built-in value.
*/
const PROVIDER_ALIASES: Readonly<Record<string, string>> = {
"databricks-v2": "databricks_v2",
"openai-compat": "openai",
};
const PROVIDER_ALIASES = new Map<string, string>([
["databricks-v2", "databricks_v2"],
["openai-compat", "openai"],
]);
/**
* Normalizes a provider id to the canonical form expected by the generated
@@ -25,7 +28,7 @@ const PROVIDER_ALIASES: Readonly<Record<string, string>> = {
*/
export function canonicalizeProvider(provider: string): string {
const normalized = provider.trim().toLowerCase();
return PROVIDER_ALIASES[normalized] ?? normalized;
return PROVIDER_ALIASES.get(normalized) ?? normalized;
}
/**
@@ -75,7 +75,7 @@ export function getProviderEffortConfig(
): ProviderEffortConfig {
const cap = resolveModelCapabilities(
canonicalizeProvider(providerId),
model ?? "",
(model ?? "").trim(),
);
return {
validValues: cap.supportedEfforts,
@@ -87,12 +87,19 @@ export const DATABRICKS_MODEL_NAMES: Map<string, string> = new Map([
// gpt5 boundary-aware token helpers
// ---------------------------------------------------------------------------
function hasLeftBoundaryGenerated(m: string, idx: number): boolean {
if (idx === 0) return true;
const prev = m[idx - 1];
return prev === "-" || prev === ".";
}
function gpt5TokenMatchesGenerated(m: string, token: string): boolean {
let start = 0;
while (true) {
const idx = m.indexOf(token, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; }
const afterChar = afterIdx < m.length ? m[afterIdx] : "";
if (afterChar === "" || afterChar === "-") return true;
start = afterIdx;
@@ -105,6 +112,7 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean {
const idx = m.indexOf(token, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; }
const suffix = m.slice(afterIdx);
if (suffix === "") return true;
if (!suffix.startsWith("-")) { start = afterIdx; continue; }
@@ -159,6 +167,22 @@ const EXACT_RECORDS = new Map<string, CapabilityResult>([
databricksV2WireRoute: "anthropic-messages",
normalizationPolicy: "none",
}],
["databricks_v2::databricks-gpt-5-6-luna", {
registryLabel: "GPT-5.6 Luna",
thinkingMode: "none",
supportedEfforts: ["low", "medium", "high"] as const,
defaultEffort: "medium",
databricksV2WireRoute: "openai-responses",
normalizationPolicy: "openai-standard",
}],
["databricks_v2::databricks-gpt-5-6-terra", {
registryLabel: "GPT-5.6 Terra",
thinkingMode: "none",
supportedEfforts: ["low", "medium", "high"] as const,
defaultEffort: "medium",
databricksV2WireRoute: "openai-responses",
normalizationPolicy: "openai-standard",
}],
]);
// ---------------------------------------------------------------------------
@@ -737,6 +761,17 @@ function lookupByFamilyRules(provider: string, normalized: string): CapabilityRe
normalizationPolicy: "openai-standard",
};
}
// rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 6
if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("gpt") || lower.split(/[^a-z0-9]+/).includes("gpt5"))) {
return {
registryLabel: null,
thinkingMode: "none",
supportedEfforts: ["none", "minimal", "low", "medium", "high", "xhigh"] as const,
defaultEffort: "medium",
databricksV2WireRoute: "openai-responses",
normalizationPolicy: "openai-clamp-max-to-xhigh",
};
}
// rule: dbv2-claude-code-names-segment, provider: databricks_v2, priority: 5
if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("claude") || lower.split(/[^a-z0-9]+/).includes("opus") || lower.split(/[^a-z0-9]+/).includes("sonnet") || lower.split(/[^a-z0-9]+/).includes("haiku") || lower.split(/[^a-z0-9]+/).includes("mythos") || lower.split(/[^a-z0-9]+/).includes("fable"))) {
return {
@@ -748,17 +783,6 @@ function lookupByFamilyRules(provider: string, normalized: string): CapabilityRe
normalizationPolicy: "none",
};
}
// rule: dbv2-gpt-code-names-segment, provider: databricks_v2, priority: 5
if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).some(s => s.startsWith("gpt")))) {
return {
registryLabel: null,
thinkingMode: "none",
supportedEfforts: ["none", "minimal", "low", "medium", "high", "xhigh"] as const,
defaultEffort: "medium",
databricksV2WireRoute: "openai-responses",
normalizationPolicy: "openai-clamp-max-to-xhigh",
};
}
// rule: dbv2-sol-luna-terra-segment, provider: databricks_v2, priority: 5
if (provider === "databricks_v2" && (lower.split(/[^a-z0-9]+/).includes("sol") || lower.split(/[^a-z0-9]+/).includes("luna") || lower.split(/[^a-z0-9]+/).includes("terra"))) {
return {
@@ -792,8 +816,8 @@ export function resolveModelCapabilities(
provider: string,
rawModelId: string,
): CapabilityResult {
// Step 1: raw exact lookup
const exactKey = `${provider}::${rawModelId}`;
// Step 1: raw exact lookup (case-insensitive — keys lowercased at build time)
const exactKey = `${provider.toLowerCase()}::${rawModelId.toLowerCase()}`;
const exact = EXACT_RECORDS.get(exactKey);
if (exact) return exact;
+1 -1
View File
@@ -29,7 +29,7 @@ TS_OUT = REPO_ROOT / "desktop/src/features/agents/lib/databricksModelNames.ts"
# Allowed characters in endpoint IDs and curated names.
SAFE_ID_RE = re.compile(r"^[a-z0-9][a-z0-9.\-]*$")
SAFE_NAME_RE = re.compile(r"^[^\x00-\x1f\"\\<>&]*$")
SAFE_NAME_RE = re.compile(r"^[^\x00-\x1f\"\\<>&]+$")
def fetch(url: str) -> bytes:
+126 -22
View File
@@ -367,7 +367,7 @@ for (const [provider, fb] of Object.entries(manifest.provider_fallbacks)) {
}
}
// Validate exact_records — check for duplicate (provider, raw_model_id) keys
// Validate exact_records — full invariant checks on every override axis
const seenExactKeys = new Set();
for (const rec of manifest.exact_records ?? []) {
if (!rec.provider || !rec.raw_model_id)
@@ -375,6 +375,60 @@ for (const rec of manifest.exact_records ?? []) {
const key = `${rec.provider}::${rec.raw_model_id}`;
if (seenExactKeys.has(key)) throw new Error(`duplicate exact_record key: ${key}`);
seenExactKeys.add(key);
// Validate match_priority: must be a non-negative integer if present (injected into comments).
if (rec.match_priority !== undefined) {
if (!Number.isInteger(rec.match_priority) || rec.match_priority < 0)
throw new Error(`exact_record ${key}: match_priority must be a non-negative integer`);
}
// Validate supported_efforts_override if present.
if (rec.supported_efforts_override !== undefined) {
assertNonEmpty(rec.supported_efforts_override, `exact_record ${key} supported_efforts_override`);
const seenEfforts = new Set();
for (const e of rec.supported_efforts_override) {
assertEnum(e, VALID_EFFORTS, `exact_record ${key} supported_efforts_override[]`);
if (seenEfforts.has(e))
throw new Error(`exact_record ${key}: duplicate effort "${e}" in supported_efforts_override`);
seenEfforts.add(e);
}
// Validate ordering matches canonical effort order (Rust clamp assumes sorted).
const canonicalIndices = rec.supported_efforts_override.map((e) => VALID_EFFORTS.indexOf(e));
for (let i = 1; i < canonicalIndices.length; i++) {
if (canonicalIndices[i] <= canonicalIndices[i - 1]) {
throw new Error(
`exact_record ${key}: supported_efforts_override must follow canonical order [${VALID_EFFORTS.join(", ")}]; got [${rec.supported_efforts_override.join(", ")}]`,
);
}
}
// Validate default_effort is in the override if present.
if (rec.default_effort !== undefined && rec.default_effort !== null) {
assertEnum(rec.default_effort, VALID_EFFORTS, `exact_record ${key} default_effort`);
if (!rec.supported_efforts_override.includes(rec.default_effort)) {
throw new Error(
`exact_record ${key}: default_effort "${rec.default_effort}" not in supported_efforts_override [${rec.supported_efforts_override.join(", ")}]`,
);
}
}
} else if (rec.default_effort !== undefined && rec.default_effort !== null) {
// default_effort override without supported_efforts_override — still validate enum.
assertEnum(rec.default_effort, VALID_EFFORTS, `exact_record ${key} default_effort`);
}
// Validate thinking_mode override if present.
if (rec.thinking_mode !== undefined) {
assertEnum(rec.thinking_mode, VALID_THINKING_MODES, `exact_record ${key} thinking_mode`);
}
// Validate databricks_v2_wire_route override if present.
if (rec.databricks_v2_wire_route !== undefined) {
assertEnum(rec.databricks_v2_wire_route, VALID_DBV2_ROUTES, `exact_record ${key} databricks_v2_wire_route`);
}
// Validate normalization_policy override if present.
if (rec.normalization_policy !== undefined) {
assertEnum(rec.normalization_policy, VALID_NORM_POLICIES, `exact_record ${key} normalization_policy`);
}
}
// ---------------------------------------------------------------------------
@@ -398,17 +452,33 @@ function stripCatalogPrefix(model) {
return firstIdx === Infinity ? model : model.slice(firstIdx);
}
/**
* Returns true if the character at position idx-1 in str is a valid left boundary:
* start-of-string, "-", or ".". Prevents substring matches like "customgpt" matching "gpt".
*/
function hasLeftBoundary(str, idx) {
if (idx === 0) return true;
const prev = str[idx - 1];
return prev === "-" || prev === ".";
}
/**
* gpt5-token match: model contains token at a word boundary (end-of-string or "-").
* Does NOT match if followed by a digit or letter.
* Left-boundary checked: token must start at beginning of string or after "-" or ".".
*/
function gpt5TokenMatches(model, token) {
const lower = model.toLowerCase();
const tok = token.toLowerCase();
let start = 0;
while (true) {
const idx = lower.indexOf(token.toLowerCase(), start);
const idx = lower.indexOf(tok, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
const afterIdx = idx + tok.length;
if (!hasLeftBoundary(lower, idx)) {
start = afterIdx;
continue;
}
const afterChar = afterIdx < lower.length ? lower[afterIdx] : "";
if (afterChar === "" || afterChar === "-") return true;
start = afterIdx;
@@ -417,14 +487,20 @@ function gpt5TokenMatches(model, token) {
/**
* gpt5-base match: like gpt5-token but also rejects short -<1-3 digit> suffixes.
* Left-boundary checked: token must start at beginning of string or after "-" or ".".
*/
function gpt5BaseMatches(model, token) {
const lower = model.toLowerCase();
const tok = token.toLowerCase();
let start = 0;
while (true) {
const idx = lower.indexOf(token.toLowerCase(), start);
const idx = lower.indexOf(tok, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
const afterIdx = idx + tok.length;
if (!hasLeftBoundary(lower, idx)) {
start = afterIdx;
continue;
}
const suffix = lower.slice(afterIdx);
if (suffix === "") return true;
if (!suffix.startsWith("-")) {
@@ -776,11 +852,14 @@ pub struct CapabilityResult {
/// Returns the exact capability record for a provider-qualified raw model ID,
/// if one exists in the manifest. This is checked BEFORE any prefix stripping.
/// Matching is case-insensitive — both inputs are lowercased before comparison.
pub fn lookup_exact(provider: &str, raw_model_id: &str) -> Option<CapabilityResult> {
match (provider, raw_model_id) {
let prov_lc = provider.to_lowercase();
let id_lc = raw_model_id.to_lowercase();
match (prov_lc.as_str(), id_lc.as_str()) {
${exactMapEntries
.map(({ rec, clean, provNote }) => {
return ` ("${rec.provider}", "${rec.raw_model_id}") => {
return ` ("${rec.provider.toLowerCase()}", "${rec.raw_model_id.toLowerCase()}") => {
${provNote}
Some(
${emitRustCapabilityResult(clean, " ")}
@@ -1014,7 +1093,9 @@ const rustGpt5Helpers = `
// gpt5 boundary-aware token helpers (used by generated family resolver)
// ---------------------------------------------------------------------------
/// Returns true if \`model\` contains \`token\` at a word boundary (end-of-string or "-").
/// Returns true if \`model\` contains \`token\` at left+right word boundaries.
/// Left boundary: start-of-string or preceded by '-' or '.'.
/// Right boundary: end-of-string or followed by '-'.
/// Does not match if followed immediately by a digit or letter.
fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
let lower = model;
@@ -1026,6 +1107,15 @@ fn gpt5_token_matches_rs(model: &str, token: &str) -> bool {
Some(rel_idx) => {
let abs_idx = start + rel_idx;
let after_idx = abs_idx + tok_lower.len();
// Left-boundary check: must start at string start or after '-' or '.'.
let left_ok = abs_idx == 0 || {
let prev = lower.as_bytes()[abs_idx - 1];
prev == b'-' || prev == b'.'
};
if !left_ok {
start = after_idx;
continue;
}
let after_char = lower[after_idx..].chars().next();
match after_char {
None | Some('-') => return true,
@@ -1047,6 +1137,15 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool {
Some(rel_idx) => {
let abs_idx = start + rel_idx;
let after_idx = abs_idx + tok_lower.len();
// Left-boundary check: must start at string start or after '-' or '.'.
let left_ok = abs_idx == 0 || {
let prev = lower.as_bytes()[abs_idx - 1];
prev == b'-' || prev == b'.'
};
if !left_ok {
start = after_idx;
continue;
}
let suffix = &lower[after_idx..];
if suffix.is_empty() {
return true;
@@ -1056,16 +1155,12 @@ fn gpt5_base_matches_rs(model: &str, token: &str) -> bool {
continue;
}
let dash_rest = &suffix[1..];
// Reject -<1-3 digits> that look like version numbers.
let is_short_version = dash_rest
.chars()
.take(4)
.enumerate()
.all(|(i, c)| {
if i < 3 { c.is_ascii_digit() }
else { !c.is_ascii_alphanumeric() }
})
&& dash_rest.chars().next().is_some_and(|c| c.is_ascii_digit());
// Reject -<1-3 digits> followed by non-alphanumeric or end (mirrors TS /^\\d{1,3}(?:[^a-z\\d]|$)/i).
let first_non_digit = dash_rest.find(|c: char| !c.is_ascii_digit()).unwrap_or(dash_rest.len());
let is_short_version = first_non_digit >= 1
&& first_non_digit <= 3
&& (first_non_digit == dash_rest.len()
|| !dash_rest.as_bytes()[first_non_digit].is_ascii_alphanumeric());
if is_short_version {
start = after_idx;
continue;
@@ -1281,12 +1376,19 @@ ${registryLabelsArr
// gpt5 boundary-aware token helpers
// ---------------------------------------------------------------------------
function hasLeftBoundaryGenerated(m: string, idx: number): boolean {
if (idx === 0) return true;
const prev = m[idx - 1];
return prev === "-" || prev === ".";
}
function gpt5TokenMatchesGenerated(m: string, token: string): boolean {
let start = 0;
while (true) {
const idx = m.indexOf(token, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; }
const afterChar = afterIdx < m.length ? m[afterIdx] : "";
if (afterChar === "" || afterChar === "-") return true;
start = afterIdx;
@@ -1299,6 +1401,7 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean {
const idx = m.indexOf(token, start);
if (idx === -1) return false;
const afterIdx = idx + token.length;
if (!hasLeftBoundaryGenerated(m, idx)) { start = afterIdx; continue; }
const suffix = m.slice(afterIdx);
if (suffix === "") return true;
if (!suffix.startsWith("-")) { start = afterIdx; continue; }
@@ -1315,7 +1418,8 @@ function gpt5BaseMatchesGenerated(m: string, token: string): boolean {
const EXACT_RECORDS = new Map<string, CapabilityResult>([
${tsExactEntries
.map(({ rec, clean }) => {
return ` ["${rec.provider}::${rec.raw_model_id}", ${emitTsCapabilityResult(clean, " ")}],`;
// Keys are lowercased at build time; resolveModelCapabilities lowercases at lookup time.
return ` ["${rec.provider.toLowerCase()}::${rec.raw_model_id.toLowerCase()}", ${emitTsCapabilityResult(clean, " ")}],`;
})
.join("\n")}
]);
@@ -1376,8 +1480,8 @@ export function resolveModelCapabilities(
provider: string,
rawModelId: string,
): CapabilityResult {
// Step 1: raw exact lookup
const exactKey = \`\${provider}::\${rawModelId}\`;
// Step 1: raw exact lookup (case-insensitive — keys lowercased at build time)
const exactKey = \`\${provider.toLowerCase()}::\${rawModelId.toLowerCase()}\`;
const exact = EXACT_RECORDS.get(exactKey);
if (exact) return exact;
@@ -1448,5 +1552,5 @@ if (CHECK_MODE && checkFailed) {
process.exit(1);
}
if (!CHECK_MODE) {
console.log("Done. Generated 3 files.");
console.log("Done. Generated 2 files.");
}
+35 -5
View File
@@ -1,5 +1,4 @@
{
"$schema": "./model-capabilities-schema.json",
"_comment": "Hand-curated model capability manifest. Edit here; run scripts/generate-model-capabilities.mjs to regenerate artifacts.",
"_generated_by": "scripts/generate-model-capabilities.mjs",
"_sources": {
@@ -437,13 +436,13 @@
},
{
"id": "dbv2-gpt-code-names-segment",
"_comment": "DBv2-only rule: endpoint names containing a GPT segment prefix (gpt*) route via OpenAI Responses. Handles 'gpt', 'gpt5', 'gpt-5' segments. Priority < individual gpt5 family rules so explicit families take precedence.",
"match_kind": "segment-prefix",
"_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj/gpt-neox false-positives.",
"match_kind": "segment",
"match_value": "gpt",
"providers": [
"databricks_v2"
],
"match_priority": 5,
"match_priority": 6,
"thinking_mode": "none",
"supported_efforts": [
"none",
@@ -455,7 +454,10 @@
],
"default_effort": "medium",
"databricks_v2_wire_route": "openai-responses",
"normalization_policy": "openai-clamp-max-to-xhigh"
"normalization_policy": "openai-clamp-max-to-xhigh",
"match_aliases": [
"gpt5"
]
},
{
"id": "dbv2-sol-luna-terra-segment",
@@ -678,6 +680,34 @@
"_reconciliation": "no-effort-divergence",
"_reconciliation_note": "models.dev advertises reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]. This is a different capability axis (extended thinking token budget), not an effort-level selector. No effort divergence to reconcile \u2014 efforts for this model come from the anthropic family rule (anthropic-adaptive-xhigh-opus-4-7).",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-claude-opus-4-7\"].reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-luna",
"registry_label": "GPT-5.6 Luna",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; luna endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-luna\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-terra",
"registry_label": "GPT-5.6 Terra",
"supported_efforts_override": [
"low",
"medium",
"high"
],
"source": "models.dev reasoning_options: low|medium|high",
"_reconciliation": "adopt",
"_reconciliation_note": "models.dev advertises [low, medium, high]. Family rule (gpt5-6) has none+xhigh+max; terra endpoint does not expose none, xhigh, or max. Provider-advertised wins.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-04): providers.databricks.models[\"databricks-gpt-5-6-terra\"].reasoning_options=[{\"type\":\"effort\",\"values\":[\"low\",\"medium\",\"high\"]}]"
}
],
"provider_fallbacks": {
+228
View File
@@ -791,5 +791,233 @@
"default_effort": "medium",
"databricks_v2_wire_route": "not-applicable"
}
},
{
"_group": "gpt-5-base guard divergence fix (CRITICAL)",
"_note": "These vectors cover the 1-2 digit version suffix window where Rust and TS diverged. Must reject from gpt5-base, fall to concrete-unknown."
},
{
"id": "openai-gpt5-10-preview-reject-base",
"provider": "openai",
"raw_model_id": "gpt-5-10-preview",
"_note": "CRITICAL divergence fix: -10- is a 2-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.",
"expect": {
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
{
"id": "openai-gpt5-2-mini-reject-base",
"provider": "openai",
"raw_model_id": "gpt-5-2-mini",
"_note": "CRITICAL divergence fix: -2- is a 1-digit suffix \u2192 gpt5-base rejects. Falls to openai concrete-unknown.",
"expect": {
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
{
"id": "openai-gpt5-9-dot-1-reject-base",
"provider": "openai",
"raw_model_id": "gpt-5-9.1",
"_note": "CRITICAL divergence fix: -9 followed by '.' is a 1-digit suffix + non-alnum \u2192 gpt5-base rejects. Falls to openai concrete-unknown.",
"expect": {
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
{
"id": "openai-customgpt-5-5-no-token-match",
"provider": "openai",
"raw_model_id": "customgpt-5-5-endpoint",
"_note": "Left-boundary fix context: customgpt-5-5-endpoint strips to gpt-5-5-endpoint which DOES match gpt5-5 family. This is correct. Update: expect gpt5-5 efforts.",
"expect": {
"supported_efforts": [
"none",
"low",
"medium",
"high",
"xhigh"
]
}
},
{
"_group": "DBv2 gpt segment rule boundary vectors",
"_note": "Verify segment match (not segment-prefix): 'gpt' must be an exact segment, not just a prefix of a segment."
},
{
"id": "dbv2-gptoss-not-responses",
"provider": "databricks_v2",
"raw_model_id": "gptoss-model",
"_note": "Collision-negative: 'gptoss' is a segment starting with 'gpt' but NOT an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.",
"expect": {
"databricks_v2_wire_route": "mlflow-chat"
}
},
{
"id": "dbv2-gptj-6b-not-responses",
"provider": "databricks_v2",
"raw_model_id": "gptj-6b",
"_note": "Collision-negative: 'gptj' is not an exact 'gpt' or 'gpt5' segment. Must fall to mlflow-chat.",
"expect": {
"databricks_v2_wire_route": "mlflow-chat"
}
},
{
"id": "dbv2-customgpt-not-responses",
"provider": "databricks_v2",
"raw_model_id": "customgpt-5-5-endpoint",
"_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint \u2192 gpt5-5 family \u2192 openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.",
"expect": {
"databricks_v2_wire_route": "openai-responses"
}
},
{
"id": "dbv2-gpt5-segment-positive",
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt5-custom",
"_note": "DBv2 gpt segment rule positive: normalized 'gpt5-custom' \u2192 segment 'gpt5' IS an exact match in match_aliases. Routes openai-responses.",
"expect": {
"databricks_v2_wire_route": "openai-responses"
}
},
{
"id": "dbv2-dual-marker-gpt-wins-openai",
"provider": "databricks_v2",
"raw_model_id": "gpt-opus-5",
"_note": "Dual-marker: normalized 'gpt-opus-5' \u2192 segments include 'gpt' AND 'opus'. Priority 6 (gpt) > 5 (claude): OpenAI wins. Must route openai-responses.",
"expect": {
"databricks_v2_wire_route": "openai-responses"
}
},
{
"_group": "Missing positive vectors"
},
{
"id": "anthropic-opus-5-adaptive-xhigh",
"provider": "anthropic",
"raw_model_id": "claude-opus-5-20270101",
"_note": "Opus-5 prefix rule: adaptive, supports xhigh+max. Verifies the rule is wired.",
"expect": {
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high"
}
},
{
"id": "dbv2-sol-normalization-policy",
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-sol",
"_note": "Sol exact record: normalization_policy comes from gpt5-6 family rule (openai-standard). Also checks supported_efforts override.",
"expect": {
"normalization_policy": "openai-standard",
"supported_efforts": [
"low",
"medium",
"high",
"max"
]
}
},
{
"id": "dbv2-luna-segment-route",
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-luna",
"_note": "Luna exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).",
"expect": {
"supported_efforts": [
"low",
"medium",
"high"
],
"databricks_v2_wire_route": "openai-responses"
}
},
{
"id": "dbv2-terra-segment-route",
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-terra",
"_note": "Terra exact record: models.dev advertises [low,medium,high]. Exact record overrides family rule. Routes openai-responses (from family).",
"expect": {
"supported_efforts": [
"low",
"medium",
"high"
],
"databricks_v2_wire_route": "openai-responses"
}
},
{
"id": "dbv2-gpt5-4-nano-exact",
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-4-nano",
"_note": "Exact record: gpt-5-4-nano, efforts [low,medium,high], registry_label 'GPT-5.4 Nano'.",
"expect": {
"supported_efforts": [
"low",
"medium",
"high"
],
"registry_label": "GPT-5.4 Nano",
"databricks_v2_wire_route": "openai-responses"
}
},
{
"id": "openrouter-concrete-unknown-fallback",
"provider": "openrouter",
"raw_model_id": "some-model-xyz",
"_note": "openrouter concrete-unknown \u2192 _default fallback (wire route not-applicable).",
"expect": {
"databricks_v2_wire_route": "not-applicable"
}
},
{
"id": "openai-gpt5-pro-uppercase-provider",
"provider": "OpenAI",
"raw_model_id": "gpt-5-pro",
"_note": "Casing: uppercase provider 'OpenAI' normalized to 'openai'. Must resolve same as openai/gpt-5-pro.",
"expect": {
"supported_efforts": [
"high"
],
"default_effort": "high"
}
},
{
"id": "dbv2-exact-record-uppercase-model",
"provider": "databricks_v2",
"raw_model_id": "DATABRICKS-GPT-5-4-NANO",
"_note": "Casing: uppercase raw_model_id. Case-insensitive exact lookup must hit the lowercase record.",
"expect": {
"supported_efforts": [
"low",
"medium",
"high"
]
}
}
]
+5 -5
View File
@@ -33,14 +33,14 @@ const corpus = JSON.parse(
// ----- Provider alias canonicalization -----
// Mirrors production canonicalizeProvider() in desktop/src/features/agents/lib/formatAgentModelLabel.ts.
// Applied before every generated lookup so alias vectors (e.g. "openai-compat") pass both interpreters.
const PROVIDER_ALIASES = {
"databricks-v2": "databricks_v2",
"openai-compat": "openai",
};
const PROVIDER_ALIASES = new Map([
["databricks-v2", "databricks_v2"],
["openai-compat", "openai"],
]);
function canonicalizeProvider(provider) {
const normalized = (provider ?? "").trim().toLowerCase();
return PROVIDER_ALIASES[normalized] ?? normalized;
return PROVIDER_ALIASES.get(normalized) ?? normalized;
}
// ----- Run corpus -----
+128
View File
@@ -410,4 +410,132 @@ test("schema-negative: exact_record registry_label with unsafe chars is rejected
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override must be non-empty
// ---------------------------------------------------------------------------
test("schema-negative: exact_record empty supported_efforts_override is rejected", () => {
assertRejects(
"exact_record empty supported_efforts_override",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = [];
}),
"supported_efforts_override",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override with invalid enum value
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override with bogus enum is rejected", () => {
assertRejects(
"exact_record bogus effort enum",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["ultra-high"];
}),
"supported_efforts_override",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override with duplicate effort
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override with duplicate effort is rejected", () => {
assertRejects(
"exact_record duplicate effort",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["low", "low"];
}),
"duplicate",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record supported_efforts_override must follow canonical order
// ---------------------------------------------------------------------------
test("schema-negative: exact_record supported_efforts_override out of canonical order is rejected", () => {
assertRejects(
"exact_record efforts out of order",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
// Reverse order — [high, medium, low] is not canonical [low, medium, high]
rec.supported_efforts_override = ["high", "medium", "low"];
}),
"canonical order",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record default_effort not in supported_efforts_override
// ---------------------------------------------------------------------------
test("schema-negative: exact_record default_effort not in supported_efforts_override is rejected", () => {
assertRejects(
"exact_record default_effort outside override",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.supported_efforts_override = ["low", "medium"];
rec.default_effort = "high"; // not in override
}),
"default_effort",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record match_priority must be a non-negative integer
// ---------------------------------------------------------------------------
test("schema-negative: exact_record match_priority non-integer is rejected", () => {
assertRejects(
"exact_record match_priority non-integer",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.match_priority = "five";
}),
"match_priority",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record thinking_mode must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid thinking_mode is rejected", () => {
assertRejects(
"exact_record invalid thinking_mode",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.thinking_mode = "turbo-thinking";
}),
"thinking_mode",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record databricks_v2_wire_route must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid databricks_v2_wire_route is rejected", () => {
assertRejects(
"exact_record invalid wire_route",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.databricks_v2_wire_route = "http-sse";
}),
"databricks_v2_wire_route",
);
});
// ---------------------------------------------------------------------------
// Rule: exact_record normalization_policy must be a valid enum
// ---------------------------------------------------------------------------
test("schema-negative: exact_record invalid normalization_policy is rejected", () => {
assertRejects(
"exact_record invalid normalization_policy",
mutate((m) => {
const rec = m.exact_records.find((r) => r.raw_model_id === "databricks-gpt-5-4-mini");
rec.normalization_policy = "pass-through-all";
}),
"normalization_policy",
);
});
console.log("\nSchema-negative validator tests complete.");