diff --git a/benchmarks/harbor-buzz-orchestra/docs/PERSONAS.md b/benchmarks/harbor-buzz-orchestra/docs/PERSONAS.md index dba25663f..c2710e22d 100644 --- a/benchmarks/harbor-buzz-orchestra/docs/PERSONAS.md +++ b/benchmarks/harbor-buzz-orchestra/docs/PERSONAS.md @@ -407,7 +407,7 @@ of every turn and delete most of what the personas currently spend ~600 tokens each rebutting. Worth revisiting if A1n shows `[Base]` is materially hurting, since at that point "production parity" is preserving a known handicap. -**Endpoint names for Opus 5 are not established.** `databricks-live.json` +**Endpoint names for Opus 5 are not established.** `databricks-example.json` currently maps only `databricks-gpt-5-6-sol` and `databricks-gpt-5-6-luna`. Every Tier-2 and Tier-3 condition above assumes an Opus 5 endpoint on the same gateway; A3, B1–B5 and C1–C5 are blocked until it exists and its list prices diff --git a/benchmarks/harbor-buzz-orchestra/manifests/lhtb-or-kimi-lead-2deepseek-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/lhtb-or-kimi-lead-2deepseek-high.yaml index c9ff7e467..27cc464d9 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/lhtb-or-kimi-lead-2deepseek-high.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/lhtb-or-kimi-lead-2deepseek-high.yaml @@ -32,8 +32,9 @@ # # THIS RUNS UNDER THE PATCHED HARBOR. LHTB needs continue_until_timeout honored # (patches/apply_continue_until_timeout.py); stock harbor ignores the flag and -# agents self-report DONE minutes into hour-long budgets. lhtb46.sh refuses to -# start on an unpatched harbor -- do not bypass that check. +# agents self-report DONE minutes into hour-long budgets. Gate the run on +# patches/gatecheck.py -- an unpatched harbor scores this cell far below the +# leaderboard and the run looks merely bad rather than misconfigured. schema_version: "1" condition: lhtb-or-kimi-lead-2deepseek-high roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2luna.yaml index ea4390358..2fc02a3a4 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2luna.yaml @@ -39,7 +39,7 @@ # genuine finding about hierarchy, not a failed rewrite. # # Endpoint names are exact Databricks serving-endpoint names and resolve via -# testbed/endpoints/databricks-live.json. buzz-agent's `databricks_v2` provider +# testbed/endpoints/databricks-example.json. buzz-agent's `databricks_v2` provider # picks the route per model from the endpoint name; see tb-solo-opus.yaml for the # detail. schema_version: "1" diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2terra.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2terra.yaml index 14d530b33..50570a510 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2terra.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-2terra.yaml @@ -7,11 +7,10 @@ # to us through Databricks (`databricks-claude-opus-5`); it is absent from # openai-live.json, and anthropic-live.json carries only Sonnet 4.6 and Haiku # 4.5. Terra is only served direct from OpenAI (`gpt-5.6-terra`); it is absent -# from databricks-live.json. So there is no single endpoint config that can serve -# both seats, and this manifest resolves against a MERGED config -- -# testbed/endpoints/mixed-opus-terra.json -- which needs BOTH DATABRICKS_TOKEN -# and OPENAI_API_KEY exported. Neither sweep.sh nor sweep-openai.sh does that; -# use sweep-mixed.sh. +# from the Databricks catalog. So there is no single-provider endpoint config +# that can serve both seats, and this manifest resolves against a MERGED config +# naming Databricks and OpenAI endpoints in one file -- which needs BOTH +# DATABRICKS_TOKEN and OPENAI_API_KEY exported. Pass it with --endpoint-config. # # What that costs interpretively: # @@ -36,7 +35,7 @@ # solo opus (0.851), that is worth knowing whatever the routes were, and it is # the last untested corner of the delegation hypothesis. # -# Endpoint names resolve via testbed/endpoints/mixed-opus-terra.json. +# Endpoint names resolve via the merged endpoint config described above. schema_version: "1" condition: tb-gt-opus-2terra roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-3luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-3luna.yaml index 5f2b546fe..b35223794 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-3luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-opus-3luna.yaml @@ -30,7 +30,7 @@ # from its 3-agent sibling in a way the score cannot explain, check `free -m` # and container exit-137 before believing the coordination story. # -# Endpoint names resolve via testbed/endpoints/databricks-live.json. +# Endpoint names resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-gt-opus-3luna roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-2luna-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-2luna-high.yaml new file mode 100644 index 000000000..aeb0fdab0 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-2luna-high.yaml @@ -0,0 +1,116 @@ +# G1sh -- goosetown personas, a gpt-5.6-sol lead over one read-only luna scout +# and one luna worker, with every seat pinned to `thinking_effort: high`. +# Single provider, single route, direct from OpenAI. +# +# BUILT FOR LHTB, not Terminal-Bench. Run it with the patched +# `continue_until_timeout` (docs/09) at `--timeout-multiplier 3.0` and `-n 8`. +# +# WHAT THIS ISOLATES. This is tb-gt-sol-2terra-high.yaml with the two worker +# seats swapped from terra to luna, and nothing else touched: same lead model, +# same personas byte-for-byte, same effort on all three seats, same route, same +# container, same clock, same concurrency. Against that cell it is a clean +# WORKER-MODEL delta at the top of the effort ramp. Against tb-solo-sol-high.yaml +# it is the team-vs-solo question with cheap workers instead of mid-priced ones. +# +# The high-effort comparison set this belongs to -- all four at n=8, effort high, +# OpenAI route, so only the roster moves: +# +# tb-solo-sol-high 1x sol (A2h) +# tb-gt-sol-2luna-high sol lead + 2x luna (this cell) +# tb-gt-sol-luna-terra-high sol lead + luna scout + terra worker +# tb-gt-sol-2terra-high sol lead + 2x terra (G1sth) +# +# DO NOT compare this cell against tb-gt-sol-2luna.yaml and call the delta an +# effort effect. That manifest is the G-wave cell and pins no effort, but it also +# ran at n=4 in the LHTB-46 wave (docs/09 §7.3) -- so the pair moves effort AND +# concurrency, and mismatched `-n` compressed between-condition spread by 0.09 in +# the TB wave. Compare only within the four cells listed above. +# +# All three seats are pinned, not just the lead. A lead thinking harder than the +# seats it delegates to is a different condition and needs its own manifest. +# +# Effort support: `gpt-5.6-sol` and `gpt-5.6-luna` both match the `gpt-5.6` +# family token in config.rs, whose supported set includes high. A1h ran luna at +# `high` and A2x ran sol at `xhigh` on this route -- neither value is rejected +# nor silently clamped. +# +# Endpoint names resolve via testbed/endpoints/openai-live.json. +schema_version: "1" +condition: tb-gt-sol-2luna-high +roster: + - id: lead + kind: orchestrator + # See tb-gt-3luna.yaml: `lead`, `scout` and `worker` are all read verbatim + # out of the "Your team" table by the personas. Renaming any of them breaks + # addressing silently -- the @mention resolves to nobody, the send still + # reports success, and the trial stalls to its timeout. + role: lead + count: 1 + endpoint: gpt-5.6-sol + model_revision: gpt-5.6-sol + prompt: + # Byte-identical to the file G0, G1s, G1, G1st, G1sth, G2s and G2 pin, so + # the lead's instructions are not a variable anywhere in this wave. + path: personas/bench/gt/gt-lead.md + sha256: 8a27e833dec0a6d1700c5ea3100f8a81c09e5090ef0a589b9583fac40b24c05c + generation: + thinking_effort: high + + - id: scout + kind: worker + role: scout + count: 1 + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + prompt: + path: personas/bench/gt/gt-scout.md + sha256: 0359a957428dd09e56e57a7fd3fe455d860264d910705f0e4992337dde25e5a5 + generation: + thinking_effort: high + + - id: worker + kind: worker + role: worker + count: 1 + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + prompt: + path: personas/bench/gt/gt-worker.md + sha256: 7a529ae58f2a2ec635fdf65fb43284b30c09d2c9dbf31e70951fe021c3b2b766 + generation: + thinking_effort: high + +prices: + # POST-2026-07-30 SHEET, matching benchmark-runs/tools/rates.py. This differs + # from tb-gt-sol-2luna.yaml and tb-gt-sol-2terra-high.yaml, which both carry + # the launch sheet (luna 1.00/0.10/6.00, terra 2.50/0.25/15.00) because they + # were written before the cut and must not be restated -- editing a completed + # cell's prices changes its condition hash. The consequence is that the + # harness' own cost_usd for THIS cell is on the current sheet while G1sth's is + # on the launch sheet, so do not set the two harness figures side by side. + # Reprice both from receipts through rates.py instead; the high-effort wave has + # per-phase receipts, so that is exact (verified on G1sth: $462.36 launch -> + # $416.49 current, with score and tokens identical). + gpt-5.6-sol: + # Unchanged by the cut, so this row reads the same on either sheet. + input_per_million_usd: 5.0 + cached_input_per_million_usd: 0.5 + output_per_million_usd: 30.0 + cache_read_rate: 0.0 + gpt-5.6-luna: + input_per_million_usd: 0.20 + cached_input_per_million_usd: 0.02 + output_per_million_usd: 1.20 + cache_read_rate: 0.0 +trial_budget: + # Identical to the solo baselines. A team genuinely needs longer than a solo + # agent -- every handoff is a round trip -- but giving it a larger budget would + # confound the comparison it exists to make, and it does not bind anyway: + # Harbor enforces each task's own `[agent] timeout_sec` scaled by 3x. + timeout_seconds: 36000 + +environment: + # Identical in every condition -- see tb-solo-luna.yaml for the full reasoning. + # override_cpus 4 with -n 8 is exactly 1:1 on a 32-vCPU m7a.8xlarge. + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-luna-terra-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-luna-terra-high.yaml new file mode 100644 index 000000000..7910d527e --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-gt-sol-luna-terra-high.yaml @@ -0,0 +1,118 @@ +# G1slth -- goosetown personas, a gpt-5.6-sol lead over a read-only LUNA scout +# and a TERRA worker, every seat pinned to `thinking_effort: high`. Single +# provider, single route, direct from OpenAI. +# +# BUILT FOR LHTB, not Terminal-Bench. Run it with the patched +# `continue_until_timeout` (docs/09) at `--timeout-multiplier 3.0` and `-n 8`. +# +# WHAT THIS ISOLATES. The two homogeneous team cells put the same model in both +# worker seats; this one splits them, and the split is deliberate rather than +# arbitrary. The scout seat is read-only reconnaissance -- it greps, reads and +# summarises, so it is the seat whose output is cheapest to be wrong about and +# the natural place for the cheap model. The worker seat actually edits and +# builds, so it gets the stronger one. If seat-appropriate assignment is worth +# anything, this cell is where it shows up: it should land between the all-luna +# and all-terra cells on cost while tracking the all-terra cell on score. +# +# The high-effort comparison set -- all four at n=8, effort high, OpenAI route, +# same lead, same personas, so only the roster moves: +# +# tb-solo-sol-high 1x sol (A2h) +# tb-gt-sol-2luna-high sol lead + 2x luna +# tb-gt-sol-luna-terra-high sol lead + luna scout + terra worker (this cell) +# tb-gt-sol-2terra-high sol lead + 2x terra (G1sth) +# +# READ THE SCOUT/WORKER ASSIGNMENT BEFORE INTERPRETING ANY DELTA. Against +# tb-gt-sol-2luna-high the worker seat is upgraded luna->terra; against +# tb-gt-sol-2terra-high the scout seat is downgraded terra->luna. Those are two +# different one-seat moves and this cell is the hinge between them, so it can be +# read either way -- but only one seat at a time. It is NOT a "mixed vs +# homogeneous" cell. +# +# DO NOT compare against any unpinned-effort or n=4 cell (docs/09 §7.3 ran at +# n=4); that moves effort and/or concurrency alongside the roster. +# +# Effort support: sol, luna and terra all match the `gpt-5.6` family token in +# config.rs, whose supported set includes high. A2x ran sol at `xhigh`, A1h ran +# luna at `high`, A5h/B3h ran terra at `high` -- no value here is rejected nor +# silently clamped on this route. +# +# Endpoint names resolve via testbed/endpoints/openai-live.json. +schema_version: "1" +condition: tb-gt-sol-luna-terra-high +roster: + - id: lead + kind: orchestrator + # See tb-gt-3luna.yaml: `lead`, `scout` and `worker` are all read verbatim + # out of the "Your team" table by the personas. Renaming any of them breaks + # addressing silently -- the @mention resolves to nobody, the send still + # reports success, and the trial stalls to its timeout. + role: lead + count: 1 + endpoint: gpt-5.6-sol + model_revision: gpt-5.6-sol + prompt: + # Byte-identical to the file every other gt cell pins. + path: personas/bench/gt/gt-lead.md + sha256: 8a27e833dec0a6d1700c5ea3100f8a81c09e5090ef0a589b9583fac40b24c05c + generation: + thinking_effort: high + + - id: scout + kind: worker + role: scout + count: 1 + # The CHEAP model in the read-only seat. This is the whole point of the cell. + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + prompt: + path: personas/bench/gt/gt-scout.md + sha256: 0359a957428dd09e56e57a7fd3fe455d860264d910705f0e4992337dde25e5a5 + generation: + thinking_effort: high + + - id: worker + kind: worker + role: worker + count: 1 + # The STRONGER model in the seat that edits and builds. + endpoint: gpt-5.6-terra + model_revision: gpt-5.6-terra + prompt: + path: personas/bench/gt/gt-worker.md + sha256: 7a529ae58f2a2ec635fdf65fb43284b30c09d2c9dbf31e70951fe021c3b2b766 + generation: + thinking_effort: high + +prices: + # POST-2026-07-30 SHEET, matching benchmark-runs/tools/rates.py. All three rows + # are load-bearing here: this is the only three-model cell in the wave, so the + # headline $/reward depends on the split across seats, not just the totals. + # tb-gt-sol-2terra-high carries the LAUNCH sheet for terra (2.50/0.25/15.00) + # and must not be restated -- reprice from receipts through rates.py to put the + # two on one basis. + gpt-5.6-sol: + # Unchanged by the cut. + input_per_million_usd: 5.0 + cached_input_per_million_usd: 0.5 + output_per_million_usd: 30.0 + cache_read_rate: 0.0 + gpt-5.6-luna: + input_per_million_usd: 0.20 + cached_input_per_million_usd: 0.02 + output_per_million_usd: 1.20 + cache_read_rate: 0.0 + gpt-5.6-terra: + input_per_million_usd: 2.00 + cached_input_per_million_usd: 0.20 + output_per_million_usd: 12.00 + cache_read_rate: 0.0 +trial_budget: + # Identical to the solo baselines -- see tb-gt-sol-2terra-high.yaml. + timeout_seconds: 36000 + +environment: + # Identical in every condition. override_cpus 4 with -n 8 is exactly 1:1 on a + # 32-vCPU m7a.8xlarge. + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-luna-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-luna-high.yaml new file mode 100644 index 000000000..c8bd38208 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-luna-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-buzz-agent-luna-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + harness: buzz-agent + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-luna: + input_per_million_usd: 0.2 + cached_input_per_million_usd: 0.02 + output_per_million_usd: 1.2 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-terra-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-terra-high.yaml new file mode 100644 index 000000000..c3970d989 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-buzz-agent-terra-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-buzz-agent-terra-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-terra + model_revision: gpt-5.6-terra + harness: buzz-agent + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-terra: + input_per_million_usd: 2.0 + cached_input_per_million_usd: 0.2 + output_per_million_usd: 12.0 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-luna-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-luna-high.yaml new file mode 100644 index 000000000..1103617fe --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-luna-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-codex-luna-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + harness: codex + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-luna: + input_per_million_usd: 0.2 + cached_input_per_million_usd: 0.02 + output_per_million_usd: 1.2 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-terra-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-terra-high.yaml new file mode 100644 index 000000000..2aa36ad67 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-codex-terra-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-codex-terra-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-terra + model_revision: gpt-5.6-terra + harness: codex + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-terra: + input_per_million_usd: 2.0 + cached_input_per_million_usd: 0.2 + output_per_million_usd: 12.0 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-luna-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-luna-high.yaml new file mode 100644 index 000000000..f8a6f12f7 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-luna-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-goose-luna-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-luna + model_revision: gpt-5.6-luna + harness: goose + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-luna: + input_per_million_usd: 0.2 + cached_input_per_million_usd: 0.02 + output_per_million_usd: 1.2 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-terra-high.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-terra-high.yaml new file mode 100644 index 000000000..69f446e33 --- /dev/null +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-harness-goose-terra-high.yaml @@ -0,0 +1,26 @@ +schema_version: "1" +condition: tb-harness-goose-terra-high +roster: + - id: solo + kind: orchestrator + role: solo + count: 1 + endpoint: gpt-5.6-terra + model_revision: gpt-5.6-terra + harness: goose + prompt: + path: personas/bench/solo-harness-comparison.md + sha256: 5a3bea67549937b6590a6d5678261cbfa08f3da67cb891dee9e81d408a93730b + generation: + thinking_effort: high +prices: + gpt-5.6-terra: + input_per_million_usd: 2.0 + cached_input_per_million_usd: 0.2 + output_per_million_usd: 12.0 + cache_read_rate: 0.0 +trial_budget: + timeout_seconds: 36000 +environment: + override_cpus: 4 + override_memory_mb: 8192 diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2gemini.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2gemini.yaml index 54df78271..3533bff7a 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2gemini.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2gemini.yaml @@ -21,7 +21,7 @@ # surface. See docs/08-gemini-provider-fixes.md. # # Endpoint names are exact Databricks serving-endpoint names; provider/host/key -# resolve via testbed/endpoints/databricks-live.json. +# resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-peer-2gemini roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2luna.yaml index 9cb583728..1438c6bbf 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-2luna.yaml @@ -19,7 +19,7 @@ # a defect. # # Endpoint names are exact Databricks serving-endpoint names; provider/host/key -# resolve via testbed/endpoints/databricks-live.json. +# resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-peer-2luna roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-sol-opus.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-sol-opus.yaml index 12782a32e..679df38a5 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-sol-opus.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-peer-sol-opus.yaml @@ -28,7 +28,7 @@ # config may name entries from different providers, because benchmark.py's # `write_provisioner_config` resolves each entry's own `api_key_env` and each # entry carries its own `provider` and `env`. It is not needed here -- both -# seats are Databricks, so `testbed/endpoints/databricks-live.json` covers them +# seats are Databricks, so `testbed/endpoints/databricks-example.json` covers them # -- but it means the OpenAI-routed variant is possible without a harness change. # # WHICH MODEL DRIVES, AND WHY IT IS NOT ARBITRARY. Sol drives, Opus navigates. @@ -51,7 +51,7 @@ # untouched. That is bias pointing the wrong way. Do not start this cell while # G1 (`tb-gt-opus-2luna`) is running on another box. # -# Endpoint names resolve via testbed/endpoints/databricks-live.json. +# Endpoint names resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-peer-sol-opus roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-gemini.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-gemini.yaml index e5e9ae24b..754aad733 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-gemini.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-gemini.yaml @@ -17,7 +17,7 @@ # # Endpoint names are exact Databricks serving-endpoint names: the runtime passes # the manifest endpoint name to the gateway as the model. They resolve to -# provider/host/key via testbed/endpoints/databricks-live.json, which is +# provider/host/key via testbed/endpoints/databricks-example.json, which is # deployment config and deliberately outside this manifest. schema_version: "1" condition: tb-solo-gemini diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-luna.yaml index 499cbb46c..0fdd9dd72 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-luna.yaml @@ -7,7 +7,7 @@ # # Endpoint names are exact Databricks serving-endpoint names: the runtime # passes the manifest endpoint name to the gateway as the model. They resolve -# to provider/host/key via testbed/endpoints/databricks-live.json, which is +# to provider/host/key via testbed/endpoints/databricks-example.json, which is # deployment config and deliberately outside this manifest. # # context_window_tokens is 272000, not the model's full window, on purpose. diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus-xhigh.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus-xhigh.yaml index 5797e8199..3bf9a7376 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus-xhigh.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus-xhigh.yaml @@ -29,7 +29,7 @@ # and the cell would report a null that was really a mislabelled `high` run. # # Endpoint names are exact Databricks serving-endpoint names; they resolve to -# provider/host/key via testbed/endpoints/databricks-live.json. +# provider/host/key via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-solo-opus-xhigh roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus.yaml index 414335023..004479079 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-opus.yaml @@ -23,7 +23,7 @@ # # Endpoint names are exact Databricks serving-endpoint names: the runtime passes # the manifest endpoint name to the gateway as the model. They resolve to -# provider/host/key via testbed/endpoints/databricks-live.json, which is +# provider/host/key via testbed/endpoints/databricks-example.json, which is # deployment config and deliberately outside this manifest. schema_version: "1" condition: tb-solo-opus diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-sol.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-sol.yaml index 465a8d0a7..fb1ce4187 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-sol.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-solo-sol.yaml @@ -10,7 +10,7 @@ # # Endpoint names are exact Databricks serving-endpoint names: the runtime passes # the manifest endpoint name to the gateway as the model. They resolve to -# provider/host/key via testbed/endpoints/databricks-live.json, which is +# provider/host/key via testbed/endpoints/databricks-example.json, which is # deployment config and deliberately outside this manifest. schema_version: "1" condition: tb-solo-sol diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-team-3luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-team-3luna.yaml index fc7f59b37..b2744dafa 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-team-3luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-team-3luna.yaml @@ -30,7 +30,7 @@ # anything. # # Endpoint names are exact Databricks serving-endpoint names; provider/host/key -# resolve via testbed/endpoints/databricks-live.json. +# resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-team-3luna roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-team-opus-luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-team-opus-luna.yaml index e8d7442f5..f85d377a6 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-team-opus-luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-team-opus-luna.yaml @@ -26,7 +26,7 @@ # mixed setting, keeping Sol only as a solo baseline (tb-solo-sol). # # Endpoint names are exact Databricks serving-endpoint names; provider/host/key -# resolve via testbed/endpoints/databricks-live.json. +# resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-team-opus-2luna roster: diff --git a/benchmarks/harbor-buzz-orchestra/manifests/tb-triad-3luna.yaml b/benchmarks/harbor-buzz-orchestra/manifests/tb-triad-3luna.yaml index c4443553c..91cdc7557 100644 --- a/benchmarks/harbor-buzz-orchestra/manifests/tb-triad-3luna.yaml +++ b/benchmarks/harbor-buzz-orchestra/manifests/tb-triad-3luna.yaml @@ -30,7 +30,7 @@ # this roster may burn up to 5x the tokens of tb-solo-sol and still win on cost. # # Endpoint names are exact Databricks serving-endpoint names; provider/host/key -# resolve via testbed/endpoints/databricks-live.json. +# resolve via testbed/endpoints/databricks-example.json. schema_version: "1" condition: tb-triad-3luna roster: diff --git a/benchmarks/harbor-buzz-orchestra/patches/lhtb-high.sh b/benchmarks/harbor-buzz-orchestra/patches/lhtb-high.sh deleted file mode 100755 index ebd0c1513..000000000 --- a/benchmarks/harbor-buzz-orchestra/patches/lhtb-high.sh +++ /dev/null @@ -1,73 +0,0 @@ -#!/usr/bin/env bash -# One high-effort LHTB-46 cell, smoke-gated. One cell per box, so the two cells -# of this pair run concurrently on separate hardware instead of serially. -# -# Usage: lhtb-high.sh