mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
fix(models): close round-2 gaps: 6-axis Rust harness + gpt-neox honesty
Co-authored-by: Will Pfleger <pfleger.will@gmail.com> Signed-off-by: Will Pfleger <pfleger.will@gmail.com>
This commit is contained in:
co-authored by
Will Pfleger
parent
f423db6f9f
commit
6301594dc6
@@ -23,7 +23,7 @@ mod shared_corpus_tests {
|
||||
|
||||
use crate::config::ThinkingEffort;
|
||||
use crate::generated_model_capabilities::{
|
||||
resolve_model_capabilities, DatabricksV2Route, ThinkingMode,
|
||||
resolve_model_capabilities, DatabricksV2Route, NormalizationPolicy, ThinkingMode,
|
||||
};
|
||||
use serde::Deserialize;
|
||||
use std::path::Path;
|
||||
@@ -46,6 +46,8 @@ mod shared_corpus_tests {
|
||||
supported_efforts: Option<Vec<String>>,
|
||||
default_effort: Option<serde_json::Value>, // string or null
|
||||
databricks_v2_wire_route: Option<String>,
|
||||
normalization_policy: Option<String>,
|
||||
registry_label: Option<serde_json::Value>, // string or null
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -87,6 +89,15 @@ mod shared_corpus_tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_normalization_policy(s: &str) -> NormalizationPolicy {
|
||||
match s {
|
||||
"none" => NormalizationPolicy::None,
|
||||
"openai-standard" => NormalizationPolicy::OpenAiStandard,
|
||||
"openai-clamp-max-to-xhigh" => NormalizationPolicy::OpenAiClampMaxToXHigh,
|
||||
other => panic!("unknown normalization_policy in corpus: {other}"),
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Corpus loader
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -202,6 +213,35 @@ mod shared_corpus_tests {
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Check normalization_policy if present
|
||||
if let Some(expected_policy) = &expect.normalization_policy {
|
||||
let expected = parse_normalization_policy(expected_policy);
|
||||
if result.normalization_policy != expected {
|
||||
failures.push(format!(
|
||||
"[{id}] normalization_policy: got {:?}, expected {:?}",
|
||||
result.normalization_policy, expected
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Check registry_label if present (JSON string or null)
|
||||
if let Some(expected_rl) = &expect.registry_label {
|
||||
let expected_opt: Option<&str> = match expected_rl {
|
||||
serde_json::Value::Null => None,
|
||||
serde_json::Value::String(s) => Some(s.as_str()),
|
||||
other => panic!(
|
||||
"unexpected registry_label value in corpus vector {id}: {other:?}"
|
||||
),
|
||||
};
|
||||
if result.registry_label != expected_opt {
|
||||
failures.push(format!(
|
||||
"[{id}] registry_label: got {:?}, expected {:?}",
|
||||
result.registry_label, expected_opt
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
if !failures.is_empty() {
|
||||
|
||||
@@ -436,7 +436,7 @@
|
||||
},
|
||||
{
|
||||
"id": "dbv2-gpt-code-names-segment",
|
||||
"_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj/gpt-neox false-positives.",
|
||||
"_comment": "DBv2-only rule: endpoint names with exact GPT segment route via OpenAI Responses. Handles segment-exact matches of \"gpt\" or \"gpt5\" (e.g. databricks-gpt-5.5 \u2192 segments include \"gpt\"). Priority > dbv2-claude (6 vs 5) restores old-contract: OpenAI checked before Claude for dual-marker names. segment match (not segment-prefix) prevents gptoss/gptj false-positives. Note: gpt-neox is a residual collision — ‘gpt-neox’ segments to [‘gpt’,‘neox’], so ‘gpt’ matches and it routes openai-responses. This is pinned as known behavior in the normative corpus, not an endorsement.",
|
||||
"match_kind": "segment",
|
||||
"match_value": "gpt",
|
||||
"providers": [
|
||||
|
||||
@@ -885,7 +885,16 @@
|
||||
"id": "dbv2-customgpt-not-responses",
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "customgpt-5-5-endpoint",
|
||||
"_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint \u2192 gpt5-5 family \u2192 openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.",
|
||||
"_note": "customgpt-5-5-endpoint strips to gpt-5-5-endpoint → gpt5-5 family → openai-responses (correct behavior after strip). Segment rule with exact \"gpt\" still correctly excludes gptoss/gptj.",
|
||||
"expect": {
|
||||
"databricks_v2_wire_route": "openai-responses"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "dbv2-gpt-neox-residual-collision",
|
||||
"provider": "databricks_v2",
|
||||
"raw_model_id": "gpt-neox-20b",
|
||||
"_note": "Residual-collision pin (not an endorsement): 'gpt-neox-20b' splits to segments ['gpt','neox','20b'], so 'gpt' exactly matches and routes openai-responses. gptoss/gptj are fixed; gpt-neox is a known residual of exact-segment matching. Pinned here so any future fix surfaces as a test change.",
|
||||
"expect": {
|
||||
"databricks_v2_wire_route": "openai-responses"
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user