From 144ef4d9d0aa2ffdbbad9856bc366f23bb0ef370 Mon Sep 17 00:00:00 2001 From: npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7 Date: Wed, 29 Jul 2026 15:03:47 -0400 Subject: [PATCH] feat(catalog): replace Databricks model-label tokenizer with models.dev registry lookup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The frontend heuristic in #3586 title-cased every word and converted adjacent digit pairs to decimals for any string starting with 'databricks-'. This invented false labels for custom workspace endpoints: 'databricks-team-2025-01' → 'Team 2025.01', 'databricks-finance-2025-01-30' → 'Finance 2025.01 30'. Replace with a lookup-and-pass-through approach drawn from the same design goose uses in production: - Generator script (scripts/generate-databricks-model-names.py) fetches https://models.dev/api.json and emits a sorted Rust static slice of (id, name) pairs under crates/buzz-agent/src/databricks_model_names.rs. Refresh by rerunning the script and committing the diff. - catalog.rs consults the table via databricks_model_name(id): known managed endpoints get curated names (e.g. 'databricks-gpt-5-5' → 'GPT-5.5'), unknown/custom endpoints return their raw ID unchanged. Applied to all ModelEntry construction paths: v1/v2 discovery, the empty-list fallback, and discovery_failure_fallback. - Frontend: a matching TS registry (desktop/src/features/agents/lib/ databricksModelNames.ts) covers persisted raw IDs on cards, rows, and popovers that render before discovery data is available. formatAgentModelLabel and formatDefaultModelLabel use the registry; no heuristic string mangling exists anywhere in the TS layer. - AGENTS.md documents the three-tier label precedence: API/runtime name first, table-backed fallback second, raw ID last. Supersedes #3586 (kennylopez-model-display-labels). Co-authored-by: Will Pfleger Signed-off-by: Will Pfleger --- crates/buzz-agent/src/catalog.rs | 90 +++++++++++++++++-- .../buzz-agent/src/databricks_model_names.rs | 45 ++++++++++ crates/buzz-agent/src/lib.rs | 1 + desktop/src/features/agents/AGENTS.md | 7 ++ .../agents/lib/agentCardModelLabel.test.mjs | 56 ++++++++++++ .../agents/lib/agentCardModelLabel.ts | 5 +- .../agents/lib/databricksModelNames.test.mjs | 50 +++++++++++ .../agents/lib/databricksModelNames.ts | 54 +++++++++++ .../agents/lib/formatAgentModelLabel.ts | 9 +- .../features/agents/ui/ManagedAgentRow.tsx | 3 +- .../profile/ui/UserProfilePopover.tsx | 3 +- scripts/generate-databricks-model-names.py | 72 +++++++++++++++ 12 files changed, 385 insertions(+), 10 deletions(-) create mode 100644 crates/buzz-agent/src/databricks_model_names.rs create mode 100644 desktop/src/features/agents/lib/databricksModelNames.test.mjs create mode 100644 desktop/src/features/agents/lib/databricksModelNames.ts create mode 100755 scripts/generate-databricks-model-names.py diff --git a/crates/buzz-agent/src/catalog.rs b/crates/buzz-agent/src/catalog.rs index aa2a121c9..659cbd76f 100644 --- a/crates/buzz-agent/src/catalog.rs +++ b/crates/buzz-agent/src/catalog.rs @@ -10,6 +10,8 @@ //! - PKCE cache empty / no token: returns `Err(AgentError::LlmAuth)` — the //! caller degrades gracefully; no browser, no hang. +use crate::databricks_model_names::DATABRICKS_MODEL_NAMES; + use reqwest::Client; use crate::{ @@ -18,8 +20,24 @@ use crate::{ types::AgentError, }; +/// Returns the curated display name for a Databricks endpoint ID, or the raw +/// ID when no entry exists in the registry. +/// +/// The registry (`DATABRICKS_MODEL_NAMES`) is generated from +/// [models.dev](https://models.dev/api.json) and covers the ~30 managed +/// Databricks endpoints. Any custom/workspace endpoint not in the table is +/// returned untouched — no heuristic guessing. +pub(crate) fn databricks_model_name(id: &str) -> &str { + DATABRICKS_MODEL_NAMES + .iter() + .find(|(k, _)| *k == id) + .map(|(_, v)| *v) + .unwrap_or(id) +} + /// A discovered model entry: `id` is the picker value, `name` is the display -/// label (same as `id` for Databricks — the API has no separate display name). +/// label (curated from models.dev for known managed endpoints; raw ID for +/// custom/unknown endpoints). #[derive(Debug, Clone, PartialEq, Eq)] pub struct ModelEntry { pub id: String, @@ -55,7 +73,7 @@ pub fn discovery_failure_fallback(provider: Provider, configured_model: &str) -> let configured_model = configured_model.trim(); let configured = ModelEntry { id: configured_model.to_string(), - name: configured_model.to_string(), + name: databricks_model_name(configured_model).to_string(), }; match provider { Provider::DatabricksV2 => { @@ -69,7 +87,7 @@ pub fn discovery_failure_fallback(provider: Provider, configured_model: &str) -> .filter(|id| **id != configured_model) .map(|id| ModelEntry { id: id.to_string(), - name: id.to_string(), + name: databricks_model_name(id).to_string(), }), ); entries @@ -207,7 +225,7 @@ pub(crate) fn parse_v1_endpoints(json: &serde_json::Value) -> Result = result.iter().map(|m| m.id.as_str()).collect(); assert_eq!(ids, DATABRICKS_V2_KNOWN_MODELS.to_vec()); } + + // --------------------------------------------------------------------------- + // Databricks model name registry + // --------------------------------------------------------------------------- + + #[test] + fn databricks_model_name_known_id_returns_curated_name() { + assert_eq!(databricks_model_name("databricks-gpt-5-5"), "GPT-5.5"); + assert_eq!( + databricks_model_name("databricks-claude-opus-4-7"), + "Claude Opus 4.7" + ); + assert_eq!( + databricks_model_name("databricks-gpt-oss-120b"), + "GPT OSS 120B" + ); + } + + #[test] + fn databricks_model_name_unknown_custom_endpoint_returns_raw_id() { + // Custom workspace endpoints must never be guessed — pass through unchanged. + assert_eq!( + databricks_model_name("databricks-team-2025-01"), + "databricks-team-2025-01" + ); + assert_eq!( + databricks_model_name("databricks-finance-2025-01-30"), + "databricks-finance-2025-01-30" + ); + assert_eq!( + databricks_model_name("some-unknown-endpoint"), + "some-unknown-endpoint" + ); + } + + #[test] + fn v2_known_models_fallback_entries_get_curated_names() { + // The DATABRICKS_V2_KNOWN_MODELS constant lists IDs that are in the + // registry, so their fallback entries must carry curated names. + for id in DATABRICKS_V2_KNOWN_MODELS { + let name = databricks_model_name(id); + assert_ne!( + name, *id, + "known model {id} should have a curated name, not raw ID" + ); + } + } + + #[test] + fn v2_discovery_failure_fallback_known_models_have_curated_names() { + let result = discovery_failure_fallback(Provider::DatabricksV2, ""); + for entry in &result { + // All DATABRICKS_V2_KNOWN_MODELS should have curated names now. + assert_ne!( + entry.name, entry.id, + "fallback entry {} should have a curated name, not raw ID", + entry.id + ); + } + } } diff --git a/crates/buzz-agent/src/databricks_model_names.rs b/crates/buzz-agent/src/databricks_model_names.rs new file mode 100644 index 000000000..fc83e9e39 --- /dev/null +++ b/crates/buzz-agent/src/databricks_model_names.rs @@ -0,0 +1,45 @@ +// GENERATED by scripts/generate-databricks-model-names.py +// Source: https://models.dev/api.json -- providers.databricks.models +// Refresh: python3 scripts/generate-databricks-model-names.py \ +// > crates/buzz-agent/src/databricks_model_names.rs +// +// Do not hand-edit -- rerun the script to update. + +/// Curated display names for known Databricks AI Gateway endpoints. +/// +/// Keys are endpoint IDs returned verbatim by the discovery APIs. +/// Values are human-readable display names sourced from models.dev. +/// +/// Unknown endpoint IDs are displayed as their raw ID -- no guessing. +pub(crate) static DATABRICKS_MODEL_NAMES: &[(&str, &str)] = &[ + ("databricks-claude-haiku-4-5", "Claude Haiku 4.5 (latest)"), + ("databricks-claude-opus-4-1", "Claude Opus 4.1 (latest)"), + ("databricks-claude-opus-4-5", "Claude Opus 4.5 (latest)"), + ("databricks-claude-opus-4-6", "Claude Opus 4.6"), + ("databricks-claude-opus-4-7", "Claude Opus 4.7"), + ("databricks-claude-sonnet-4", "Claude Sonnet 4.5"), + ("databricks-claude-sonnet-4-5", "Claude Sonnet 4.5 (latest)"), + ("databricks-claude-sonnet-4-6", "Claude Sonnet 4.6"), + ("databricks-gemini-2-5-flash", "Gemini 2.5 Flash"), + ("databricks-gemini-2-5-pro", "Gemini 2.5 Pro"), + ("databricks-gemini-3-1-flash-lite", "Gemini 3.1 Flash Lite Preview"), + ("databricks-gemini-3-1-pro", "Gemini 3.1 Pro Preview Custom Tools"), + ("databricks-gemini-3-flash", "Gemini 3 Flash Preview"), + ("databricks-gemini-3-pro", "Gemini 3 Pro Preview"), + ("databricks-glm-5-2", "GLM-5.2"), + ("databricks-gpt-5", "GPT-5"), + ("databricks-gpt-5-1", "GPT-5.1"), + ("databricks-gpt-5-2", "GPT-5.2"), + ("databricks-gpt-5-4", "GPT-5.4"), + ("databricks-gpt-5-4-mini", "GPT-5.4 mini"), + ("databricks-gpt-5-4-nano", "GPT-5.4 nano"), + ("databricks-gpt-5-5", "GPT-5.5"), + ("databricks-gpt-5-6-luna", "GPT-5.6 Luna"), + ("databricks-gpt-5-6-sol", "GPT-5.6 Sol"), + ("databricks-gpt-5-6-terra", "GPT-5.6 Terra"), + ("databricks-gpt-5-mini", "GPT-5 Mini"), + ("databricks-gpt-5-nano", "GPT-5 Nano"), + ("databricks-gpt-oss-120b", "GPT OSS 120B"), + ("databricks-gpt-oss-20b", "GPT OSS 20B"), + ("databricks-kimi-k2-7-code", "Kimi K2.7 Code"), +]; diff --git a/crates/buzz-agent/src/lib.rs b/crates/buzz-agent/src/lib.rs index 6745dd0f9..8001a89ac 100644 --- a/crates/buzz-agent/src/lib.rs +++ b/crates/buzz-agent/src/lib.rs @@ -3,6 +3,7 @@ mod agent; pub mod auth; mod builtin; pub mod catalog; +pub(crate) mod databricks_model_names; pub mod config; mod handoff; mod hints; diff --git a/desktop/src/features/agents/AGENTS.md b/desktop/src/features/agents/AGENTS.md index 35ad4a63a..619241794 100644 --- a/desktop/src/features/agents/AGENTS.md +++ b/desktop/src/features/agents/AGENTS.md @@ -143,3 +143,10 @@ matches the code is worse than no rule; a new pattern that isn't written down here will be broken by the next agent that never learns it existed. Reviewers: treat a config-behavior diff without a matching AGENTS.md diff (or an explicit "no rules changed" note) as incomplete. + +## Model display-name precedence + +Labels shown in cards, pickers, and popovers follow a three-tier cascade: +1. **API/runtime `name`** — `AgentModelInfo.name` from discovery (`AgentModelsResponse`). This is the authoritative source; all providers populate it at discover time (`openai_model_display_name`, Anthropic `display_name`, ACP runtime name, Databricks registry lookup). +2. **Table-backed fallback** — for persisted raw Databricks endpoint IDs that render before discovery data is available, `databricksModelName(id)` in `desktop/src/features/agents/lib/databricksModelNames.ts` does a static lookup against the models.dev-seeded registry. Refresh by rerunning `scripts/generate-databricks-model-names.py`. +3. **Raw ID** — any ID not covered by tiers 1 or 2 renders unchanged. No heuristic string mangling. diff --git a/desktop/src/features/agents/lib/agentCardModelLabel.test.mjs b/desktop/src/features/agents/lib/agentCardModelLabel.test.mjs index 696a055d2..d2fd7647b 100644 --- a/desktop/src/features/agents/lib/agentCardModelLabel.test.mjs +++ b/desktop/src/features/agents/lib/agentCardModelLabel.test.mjs @@ -56,3 +56,59 @@ test("resolveAgentCardModelLabel — non-inherited agent with a blank resolved m }); assert.equal(label, "Default model (claude-sonnet)"); }); + +// Databricks registry integration +import { formatAgentModelLabel } from "./formatAgentModelLabel.ts"; + +test("formatAgentModelLabel — known Databricks managed ID returns curated name", () => { + assert.equal(formatAgentModelLabel("databricks-gpt-5-5"), "GPT-5.5"); + assert.equal( + formatAgentModelLabel("databricks-claude-opus-4-7"), + "Claude Opus 4.7", + ); +}); + +test("formatAgentModelLabel — unknown custom Databricks ID returns raw ID unchanged", () => { + assert.equal( + formatAgentModelLabel("databricks-team-2025-01"), + "databricks-team-2025-01", + ); +}); + +test("formatAgentModelLabel — non-Databricks ID returns raw ID unchanged", () => { + assert.equal(formatAgentModelLabel("claude-sonnet-4-7"), "claude-sonnet-4-7"); + assert.equal(formatAgentModelLabel("gpt-4o"), "gpt-4o"); +}); + +test("formatAgentModelLabel — null or empty returns Auto", () => { + assert.equal(formatAgentModelLabel(null), "Auto"); + assert.equal(formatAgentModelLabel(""), "Auto"); + assert.equal(formatAgentModelLabel(" "), "Auto"); +}); + +test("resolveAgentCardModelLabel — known Databricks defaultModel renders curated name in default label", () => { + const label = resolveAgentCardModelLabel({ + agent: undefined, + personaModel: null, + defaultModel: "databricks-gpt-5-5", + }); + assert.equal(label, "Default model (GPT-5.5)"); +}); + +test("resolveAgentCardModelLabel — unknown custom Databricks defaultModel renders raw ID in default label", () => { + const label = resolveAgentCardModelLabel({ + agent: undefined, + personaModel: null, + defaultModel: "databricks-team-2025-01", + }); + assert.equal(label, "Default model (databricks-team-2025-01)"); +}); + +test("resolveAgentCardModelLabel — known Databricks agent model renders curated name", () => { + const label = resolveAgentCardModelLabel({ + agent: { modelSource: "definition", model: "databricks-gpt-oss-120b" }, + personaModel: null, + defaultModel: "something-else", + }); + assert.equal(label, "GPT OSS 120B"); +}); diff --git a/desktop/src/features/agents/lib/agentCardModelLabel.ts b/desktop/src/features/agents/lib/agentCardModelLabel.ts index 4ec06f10c..065940236 100644 --- a/desktop/src/features/agents/lib/agentCardModelLabel.ts +++ b/desktop/src/features/agents/lib/agentCardModelLabel.ts @@ -1,4 +1,5 @@ import { formatAgentModelLabel } from "./formatAgentModelLabel"; +import { databricksModelName } from "./databricksModelNames"; import type { ManagedAgent } from "@/shared/api/types"; /** @@ -40,5 +41,7 @@ export function resolveAgentCardModelLabel(input: { export function formatDefaultModelLabel(defaultModel: string) { const model = defaultModel.trim(); - return model ? `Default model (${model})` : "Default model"; + return model + ? `Default model (${databricksModelName(model)})` + : "Default model"; } diff --git a/desktop/src/features/agents/lib/databricksModelNames.test.mjs b/desktop/src/features/agents/lib/databricksModelNames.test.mjs new file mode 100644 index 000000000..119181c73 --- /dev/null +++ b/desktop/src/features/agents/lib/databricksModelNames.test.mjs @@ -0,0 +1,50 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + databricksModelName, + DATABRICKS_MODEL_NAMES, +} from "./databricksModelNames.ts"; + +test("databricksModelName — known managed endpoint returns curated name", () => { + assert.equal(databricksModelName("databricks-gpt-5-5"), "GPT-5.5"); + assert.equal( + databricksModelName("databricks-claude-opus-4-7"), + "Claude Opus 4.7", + ); + assert.equal(databricksModelName("databricks-gpt-oss-120b"), "GPT OSS 120B"); +}); + +test("databricksModelName — unknown custom endpoint returns raw ID unchanged", () => { + // A workspace-specific or date-stamped endpoint that isn't in the registry + // must never be guessed — return it verbatim. + assert.equal( + databricksModelName("databricks-team-2025-01"), + "databricks-team-2025-01", + ); + assert.equal( + databricksModelName("databricks-finance-2025-01-30"), + "databricks-finance-2025-01-30", + ); + assert.equal( + databricksModelName("some-custom-workspace-model"), + "some-custom-workspace-model", + ); +}); + +test("databricksModelName — empty string returns empty string", () => { + assert.equal(databricksModelName(""), ""); +}); + +test("DATABRICKS_MODEL_NAMES — registry is non-empty and all values are non-empty strings", () => { + const entries = Object.entries(DATABRICKS_MODEL_NAMES); + assert.ok(entries.length > 0, "registry must not be empty"); + for (const [id, name] of entries) { + assert.ok( + id.startsWith("databricks-"), + `ID ${id} must start with 'databricks-'`, + ); + assert.ok(name.length > 0, `name for ${id} must be non-empty`); + assert.notEqual(name, id, `curated name for ${id} must differ from raw ID`); + } +}); diff --git a/desktop/src/features/agents/lib/databricksModelNames.ts b/desktop/src/features/agents/lib/databricksModelNames.ts new file mode 100644 index 000000000..42f443bf9 --- /dev/null +++ b/desktop/src/features/agents/lib/databricksModelNames.ts @@ -0,0 +1,54 @@ +// GENERATED by scripts/generate-databricks-model-names.py +// Source: https://models.dev/api.json -- providers.databricks.models +// Refresh: python3 scripts/generate-databricks-model-names.py (then port to TS) +// +// Do not hand-edit -- rerun the script to update. + +/** + * Curated display names for known Databricks AI Gateway endpoints. + * + * Keys are endpoint IDs returned verbatim by the discovery APIs. + * Values are human-readable display names sourced from models.dev. + * + * Unknown endpoint IDs are displayed as their raw ID -- no guessing. + */ +export const DATABRICKS_MODEL_NAMES: Record = { + "databricks-claude-haiku-4-5": "Claude Haiku 4.5 (latest)", + "databricks-claude-opus-4-1": "Claude Opus 4.1 (latest)", + "databricks-claude-opus-4-5": "Claude Opus 4.5 (latest)", + "databricks-claude-opus-4-6": "Claude Opus 4.6", + "databricks-claude-opus-4-7": "Claude Opus 4.7", + "databricks-claude-sonnet-4": "Claude Sonnet 4.5", + "databricks-claude-sonnet-4-5": "Claude Sonnet 4.5 (latest)", + "databricks-claude-sonnet-4-6": "Claude Sonnet 4.6", + "databricks-gemini-2-5-flash": "Gemini 2.5 Flash", + "databricks-gemini-2-5-pro": "Gemini 2.5 Pro", + "databricks-gemini-3-1-flash-lite": "Gemini 3.1 Flash Lite Preview", + "databricks-gemini-3-1-pro": "Gemini 3.1 Pro Preview Custom Tools", + "databricks-gemini-3-flash": "Gemini 3 Flash Preview", + "databricks-gemini-3-pro": "Gemini 3 Pro Preview", + "databricks-glm-5-2": "GLM-5.2", + "databricks-gpt-5": "GPT-5", + "databricks-gpt-5-1": "GPT-5.1", + "databricks-gpt-5-2": "GPT-5.2", + "databricks-gpt-5-4": "GPT-5.4", + "databricks-gpt-5-4-mini": "GPT-5.4 mini", + "databricks-gpt-5-4-nano": "GPT-5.4 nano", + "databricks-gpt-5-5": "GPT-5.5", + "databricks-gpt-5-6-luna": "GPT-5.6 Luna", + "databricks-gpt-5-6-sol": "GPT-5.6 Sol", + "databricks-gpt-5-6-terra": "GPT-5.6 Terra", + "databricks-gpt-5-mini": "GPT-5 Mini", + "databricks-gpt-5-nano": "GPT-5 Nano", + "databricks-gpt-oss-120b": "GPT OSS 120B", + "databricks-gpt-oss-20b": "GPT OSS 20B", + "databricks-kimi-k2-7-code": "Kimi K2.7 Code", +}; + +/** + * Returns the curated display name for a Databricks endpoint ID, or the raw + * ID when no entry exists in the registry. No heuristic guessing. + */ +export function databricksModelName(id: string): string { + return DATABRICKS_MODEL_NAMES[id] ?? id; +} diff --git a/desktop/src/features/agents/lib/formatAgentModelLabel.ts b/desktop/src/features/agents/lib/formatAgentModelLabel.ts index 6c32d5393..46ad5897e 100644 --- a/desktop/src/features/agents/lib/formatAgentModelLabel.ts +++ b/desktop/src/features/agents/lib/formatAgentModelLabel.ts @@ -1,8 +1,15 @@ +import { databricksModelName } from "./databricksModelNames"; + /** * Returns a human-readable model label for an agent or persona, falling back to * "Auto" when no model is set (empty or whitespace-only). + * + * For known Databricks managed endpoints the registry-curated name is returned + * (e.g. "databricks-gpt-5-5" → "GPT-5.5"). Unknown or custom endpoint IDs are + * returned unchanged — no heuristic string mangling. */ export function formatAgentModelLabel(model: string | null | undefined) { const trimmed = model?.trim(); - return trimmed && trimmed.length > 0 ? trimmed : "Auto"; + if (!trimmed) return "Auto"; + return databricksModelName(trimmed); } diff --git a/desktop/src/features/agents/ui/ManagedAgentRow.tsx b/desktop/src/features/agents/ui/ManagedAgentRow.tsx index 606d2b788..e9897a3e1 100644 --- a/desktop/src/features/agents/ui/ManagedAgentRow.tsx +++ b/desktop/src/features/agents/ui/ManagedAgentRow.tsx @@ -27,6 +27,7 @@ import { friendlyAgentLastError } from "@/features/agents/lib/friendlyAgentLastE import { ManagedAgentLogPanel } from "./ManagedAgentLogPanel"; import { PubKey } from "@/shared/ui/PubKey"; import { SubsectionLabel } from "@/shared/ui/PageHeader"; +import { databricksModelName } from "@/features/agents/lib/databricksModelNames"; export function ManagedAgentRow({ agent, @@ -410,7 +411,7 @@ function RuntimeBlock({ {runtimeSource || agent.model ? (
{runtimeSource ? {runtimeSource} : null} - {agent.model ? {agent.model} : null} + {agent.model ? {databricksModelName(agent.model)} : null}
) : null} diff --git a/desktop/src/features/profile/ui/UserProfilePopover.tsx b/desktop/src/features/profile/ui/UserProfilePopover.tsx index f2739088a..d084d41e0 100644 --- a/desktop/src/features/profile/ui/UserProfilePopover.tsx +++ b/desktop/src/features/profile/ui/UserProfilePopover.tsx @@ -51,6 +51,7 @@ import { BotIdenticon } from "@/features/messages/ui/BotIdenticon"; import { useNow } from "@/shared/lib/useNow"; import { Button } from "@/shared/ui/button"; import { Spinner } from "@/shared/ui/spinner"; +import { databricksModelName } from "@/features/agents/lib/databricksModelNames"; type UserProfilePopoverProps = { children: React.ReactNode; @@ -611,7 +612,7 @@ export function UserProfilePopover({ {runtimeLabel(relayAgent.agentType)} ) : null} {managedAgent?.model ? ( - {managedAgent.model} + {databricksModelName(managedAgent.model)} ) : null} {managedAgent?.acpCommand ? ( ACP: {managedAgent.acpCommand} diff --git a/scripts/generate-databricks-model-names.py b/scripts/generate-databricks-model-names.py new file mode 100755 index 000000000..069259ffa --- /dev/null +++ b/scripts/generate-databricks-model-names.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Generate crates/buzz-agent/src/databricks_model_names.rs from models.dev. + +Usage: + python3 scripts/generate-databricks-model-names.py \ + > crates/buzz-agent/src/databricks_model_names.rs + +Fetches https://models.dev/api.json, extracts providers.databricks.models (the +authoritative curated name registry used by goose and others), and emits a +static Rust slice of (id, display_name) pairs sorted by ID. + +Re-run whenever Databricks ships a new managed endpoint and commit the diff. +""" + +import json +import subprocess +import sys + + +URL = "https://models.dev/api.json" + + +def fetch(url: str) -> bytes: + # urllib blocks with HTTP 403 without a browser User-Agent; use curl when + # available so the script works in hermit environments where curl is pinned. + result = subprocess.run( + ["curl", "-s", "-A", "Mozilla/5.0", url], + capture_output=True, + check=True, + ) + return result.stdout + + +def rust_str(s: str) -> str: + """Emit a Rust string literal with double quotes.""" + escaped = s.replace("\\", "\\\\").replace('"', '\\"') + return f'"{escaped}"' + + +def main() -> None: + raw = fetch(URL) + data = json.loads(raw) + models: dict = data["databricks"]["models"] + entries = sorted( + (k, v["name"] if isinstance(v, dict) else k) for k, v in models.items() + ) + + lines = [ + "// GENERATED by scripts/generate-databricks-model-names.py", + "// Source: https://models.dev/api.json -- providers.databricks.models", + "// Refresh: python3 scripts/generate-databricks-model-names.py \\", + "// > crates/buzz-agent/src/databricks_model_names.rs", + "//", + "// Do not hand-edit -- rerun the script to update.", + "", + "/// Curated display names for known Databricks AI Gateway endpoints.", + "///", + "/// Keys are endpoint IDs returned verbatim by the discovery APIs.", + "/// Values are human-readable display names sourced from models.dev.", + "///", + "/// Unknown endpoint IDs are displayed as their raw ID -- no guessing.", + "pub(crate) static DATABRICKS_MODEL_NAMES: &[(&str, &str)] = &[", + ] + for id_, name in entries: + lines.append(f" ({rust_str(id_)}, {rust_str(name)}),") + lines += ["];", ""] + + sys.stdout.write("\n".join(lines)) + + +if __name__ == "__main__": + main()