From 6301594dc621c033ea10cd3b083b06f32ba9259e Mon Sep 17 00:00:00 2001 From: npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7 Date: Tue, 4 Aug 2026 15:29:14 -0400 Subject: [PATCH] fix(models): close round-2 gaps: 6-axis Rust harness + gpt-neox honesty Co-authored-by: Will Pfleger Signed-off-by: Will Pfleger --- .../src/generated_model_capabilities_tests.rs | 42 ++++++++++++++++++- scripts/model-capabilities.json | 2 +- scripts/normative-corpus.json | 11 ++++- 3 files changed, 52 insertions(+), 3 deletions(-) diff --git a/crates/buzz-agent/src/generated_model_capabilities_tests.rs b/crates/buzz-agent/src/generated_model_capabilities_tests.rs index efa6f6153..12392997c 100644 --- a/crates/buzz-agent/src/generated_model_capabilities_tests.rs +++ b/crates/buzz-agent/src/generated_model_capabilities_tests.rs @@ -23,7 +23,7 @@ mod shared_corpus_tests { use crate::config::ThinkingEffort; use crate::generated_model_capabilities::{ - resolve_model_capabilities, DatabricksV2Route, ThinkingMode, + resolve_model_capabilities, DatabricksV2Route, NormalizationPolicy, ThinkingMode, }; use serde::Deserialize; use std::path::Path; @@ -46,6 +46,8 @@ mod shared_corpus_tests { supported_efforts: Option>, default_effort: Option, // string or null databricks_v2_wire_route: Option, + normalization_policy: Option, + registry_label: Option, // string or null } // --------------------------------------------------------------------------- @@ -87,6 +89,15 @@ mod shared_corpus_tests { } } + fn parse_normalization_policy(s: &str) -> NormalizationPolicy { + match s { + "none" => NormalizationPolicy::None, + "openai-standard" => NormalizationPolicy::OpenAiStandard, + "openai-clamp-max-to-xhigh" => NormalizationPolicy::OpenAiClampMaxToXHigh, + other => panic!("unknown normalization_policy in corpus: {other}"), + } + } + // --------------------------------------------------------------------------- // Corpus loader // --------------------------------------------------------------------------- @@ -202,6 +213,35 @@ mod shared_corpus_tests { )); } } + + // Check normalization_policy if present + if let Some(expected_policy) = &expect.normalization_policy { + let expected = parse_normalization_policy(expected_policy); + if result.normalization_policy != expected { + failures.push(format!( + "[{id}] normalization_policy: got {:?}, expected {:?}", + result.normalization_policy, expected + )); + } + } + + // Check registry_label if present (JSON string or null) + if let Some(expected_rl) = &expect.registry_label { + let expected_opt: Option<&str> = match expected_rl { + serde_json::Value::Null => None, + serde_json::Value::String(s) => Some(s.as_str()), + other => panic!( + "unexpected registry_label value in corpus vector {id}: {other:?}" + ), + }; + if result.registry_label != expected_opt { + failures.push(format!( + "[{id}] registry_label: got {:?}, expected {:?}", + result.registry_label, expected_opt + )); + } + } + } if !failures.is_empty() { diff --git a/scripts/model-capabilities.json b/scripts/model-capabilities.json index 7ab43ebb9..d5787eb1d 100644 --- a/scripts/model-capabilities.json +++ b/scripts/model-capabilities.json @@ -436,7 +436,7 @@ }, { "id": "dbv2-gpt-code-names-segment", - "_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj/gpt-neox false-positives.", + "_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj false-positives. Note: gpt-neox is a residual collision — ‘gpt-neox’ segments to [‘gpt’,‘neox’], so ‘gpt’ matches and it routes openai-responses. This is pinned as known behavior in the normative corpus, not an endorsement.", "match_kind": "segment", "match_value": "gpt", "providers": [ diff --git a/scripts/normative-corpus.json b/scripts/normative-corpus.json index e30eed151..580bb9a0a 100644 --- a/scripts/normative-corpus.json +++ b/scripts/normative-corpus.json @@ -885,7 +885,16 @@ "id": "dbv2-customgpt-not-responses", "provider": "databricks_v2", "raw_model_id": "customgpt-5-5-endpoint", - "_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint \u2192 gpt5-5 family \u2192 openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.", + "_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint → gpt5-5 family → openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.", + "expect": { + "databricks_v2_wire_route": "openai-responses" + } + }, + { + "id": "dbv2-gpt-neox-residual-collision", + "provider": "databricks_v2", + "raw_model_id": "gpt-neox-20b", + "_note": "Residual-collision pin (not an endorsement): 'gpt-neox-20b' splits to segments ['gpt','neox','20b'], so 'gpt' exactly matches and routes openai-responses. gptoss/gptj are fixed; gpt-neox is a known residual of exact-segment matching. Pinned here so any future fix surfaces as a test change.", "expect": { "databricks_v2_wire_route": "openai-responses" }