mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
## Summary Centralizes model capability knowledge — thinking mode, supported effort levels, wire routes, and human-readable labels — into a single manifest, `scripts/model-capabilities.json`. Rust and TypeScript each get a small interpreter that reads the same manifest, replacing hand-maintained tables scattered across both languages that had already drifted apart. A capability change is now a data edit, not parallel edits to two code paths. Supersedes the codegen approach explored in #3603. A cross-language contract keeps the two interpreters honest: `scripts/normative-corpus.json` is a golden snapshot generated from the Rust resolver (103 vectors covering all six capability axes) and replayed natively in TS. CI fails if either language disagrees with the corpus or the corpus drifts from the resolver. Regenerate with `just regen-model-corpus`. ## Behavior changes - **Effort dropdown for `openai-compat` providers** no longer offers `max`. The request path always clamped `max` to `xhigh` on the wire, so the UI stops offering a value that was silently rewritten. UI-only, wire-identical. - **Databricks v2 routing (wire-visible):** uncurated endpoint names carrying a bare Claude code-name segment (e.g. `goose-opus-5`) now route to the MLflow chat wire instead of Anthropic Messages — they lose Anthropic prompt caching but still succeed on a valid OpenAI-compatible wire. Curated `databricks-claude-*` records and any name starting with `claude` are unchanged. A handful of other uncurated/adversarial name shapes similarly fall back to MLflow chat instead of pattern-matched routes; every curated model resolves identically to before, all axes. - **Curated model labels on the real discovery path.** The Databricks API returns no display name, so discovery emits the raw endpoint id as the model `name` (`{id, name: id}`) on every path. `ModelEntry.name` is now curated at all four construction seams in `buzz-agent` — v2 discovery, v1 parse, the auth-empty default catalog, and the configured-model fallback — via a read-only `databricks_registry_label` lookup over the manifest's `databricks_v2` exact records; `id` stays the raw wire/config value. A known id renders its curated label (`databricks-gpt-5-5` → `GPT-5.5`), an unknown id passes through unchanged, and the default-catalog row reads `GPT-5.5 (default catalog)`. As a defense against older `buzz-agent` binaries and any harness that echoes ids, `resolveModelLabel` treats a discovered name equal to the trimmed id as absent and falls through to the registry tier; a genuinely distinct name (including the suffixed default-catalog label) still wins. ## Cleanup Deletes the duplicated capability tables and their tests: the `config.rs` gpt5 matchers, effort tables, and clamp logic; the legacy segment-based Databricks v2 route classifier in `llm.rs`; and the TS hand tables plus `effortTable.fixture.json`. All are replaced by manifest lookups through the shared resolver — no line of capability data exists in two places. --------- Signed-off-by: Will Pfleger <pfleger.will@gmail.com> Signed-off-by: Duncan <dcfd242e557282d7a1e2cf2e6877522682f1e5c6156dc92ca7d90eaedd3b0f95@buzz.block.builderlab.xyz> Co-authored-by: Duncan <dcfd242e557282d7a1e2cf2e6877522682f1e5c6156dc92ca7d90eaedd3b0f95@buzz.block.builderlab.xyz>
194 lines
6.1 KiB
Bash
Executable File
194 lines
6.1 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# =============================================================================
|
|
# run-tests.sh — Run Buzz test suite
|
|
# =============================================================================
|
|
# Usage:
|
|
# ./scripts/run-tests.sh # run all tests (default)
|
|
# ./scripts/run-tests.sh unit # unit tests only (no infra needed)
|
|
# ./scripts/run-tests.sh integration # integration tests only
|
|
# ./scripts/run-tests.sh all # explicit all
|
|
# =============================================================================
|
|
set -euo pipefail
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"
|
|
MODE="${1:-all}"
|
|
|
|
# Colors
|
|
RED='\033[0;31m'
|
|
GREEN='\033[0;32m'
|
|
YELLOW='\033[1;33m'
|
|
BLUE='\033[0;34m'
|
|
CYAN='\033[0;36m'
|
|
NC='\033[0m'
|
|
|
|
log() { echo -e "${BLUE}[run-tests]${NC} $*"; }
|
|
success(){ echo -e "${GREEN}[run-tests]${NC} $*"; }
|
|
warn() { echo -e "${YELLOW}[run-tests]${NC} $*"; }
|
|
error() { echo -e "${RED}[run-tests]${NC} $*" >&2; }
|
|
section(){ echo -e "\n${CYAN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"; echo -e "${CYAN} $*${NC}"; echo -e "${CYAN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"; }
|
|
|
|
cd "${REPO_ROOT}"
|
|
|
|
# ---- Load .env if present ---------------------------------------------------
|
|
|
|
if [[ -f ".env" ]]; then
|
|
log "Loading .env..."
|
|
set -o allexport
|
|
# shellcheck disable=SC1091
|
|
source .env
|
|
set +o allexport
|
|
else
|
|
# Use defaults matching docker-compose.yml
|
|
export DATABASE_URL="postgres://buzz:buzz_dev@localhost:5432/buzz" # sadscan:disable np.postgres.1
|
|
export PGHOST=localhost
|
|
export PGPORT=5432
|
|
export PGUSER=buzz
|
|
export PGPASSWORD=buzz_dev
|
|
export PGDATABASE=buzz
|
|
export REDIS_URL="redis://localhost:6379"
|
|
fi
|
|
|
|
# ---- Track results ----------------------------------------------------------
|
|
|
|
declare -a PASSED=()
|
|
declare -a FAILED=()
|
|
|
|
run_test_step() {
|
|
local name="$1"
|
|
shift
|
|
log "Running: ${name}"
|
|
if "$@"; then
|
|
success "${name} passed"
|
|
PASSED+=("${name}")
|
|
else
|
|
error "${name} FAILED"
|
|
FAILED+=("${name}")
|
|
fi
|
|
}
|
|
|
|
# ---- Check / start infra (for integration tests) ----------------------------
|
|
|
|
ensure_infra() {
|
|
"${REPO_ROOT}/bin/just" _ensure-migrations
|
|
}
|
|
|
|
# ---- Unit tests (no infra needed) -------------------------------------------
|
|
|
|
run_unit_tests() {
|
|
section "Unit Tests (no infra required)"
|
|
|
|
run_test_step "buzz-core tests" \
|
|
cargo test -p buzz-core --lib -- --nocapture
|
|
|
|
run_test_step "buzz-auth unit tests" \
|
|
cargo test -p buzz-auth --lib -- --nocapture
|
|
|
|
run_test_step "buzz-voice tests" \
|
|
cargo test -p buzz-voice --lib -- --nocapture
|
|
|
|
run_test_step "buzz-cli tests" \
|
|
cargo test -p buzz-cli -- --nocapture
|
|
|
|
# buzz-db migrator/lint unit tests (no infra): guard the embedded-migrator
|
|
# invariant (exactly the consolidated 0001; cutover/backfill stays an operator
|
|
# script, not startup state) and the tenant-scoping lints. The Postgres-backed
|
|
# buzz-db tests are #[ignore]d; nothing here (or in integration mode below,
|
|
# which runs `cargo test -p buzz-db` without --ignored) runs them — they need a
|
|
# separate isolated-DB gate, so --lib keeps this step infra-free.
|
|
run_test_step "buzz-db unit tests" \
|
|
cargo test -p buzz-db --lib -- --nocapture
|
|
|
|
# Multi-tenant conformance gate: independent replay checker + golden
|
|
# fixtures (buzz-conformance). Pure in-process trace replay, no infra.
|
|
run_test_step "buzz-conformance tests" \
|
|
cargo test -p buzz-conformance -- --nocapture
|
|
|
|
run_test_step "buzz-push-gateway tests" \
|
|
cargo test -p buzz-push-gateway -- --nocapture
|
|
|
|
# Kubernetes backend provider: pure decision layers driven by a fake
|
|
# substrate, no cluster. Mirrors the nextest path in `just test-unit` —
|
|
# the two lists must stay in step or the fallback silently covers less.
|
|
run_test_step "buzz-backend-kubernetes tests" \
|
|
cargo test -p buzz-backend-kubernetes -- --nocapture
|
|
|
|
# buzz-agent model-capabilities corpus: the Rust half of the cross-language
|
|
# drift guard. model_capabilities.rs embeds scripts/model-capabilities.json +
|
|
# scripts/normative-corpus.json via include_str! and replays all 103 vectors
|
|
# as pure in-process tests (no infra). Mirrors the nextest path in
|
|
# `just test-unit` — the two lists must stay in step.
|
|
run_test_step "buzz-agent unit tests" \
|
|
cargo test -p buzz-agent --lib -- --nocapture
|
|
}
|
|
|
|
# ---- DB / integration tests (infra required) --------------------------------
|
|
|
|
run_integration_tests() {
|
|
section "Integration Tests (requires running services)"
|
|
|
|
ensure_infra
|
|
|
|
run_test_step "buzz-db tests" \
|
|
cargo test -p buzz-db -- --nocapture
|
|
|
|
if find crates/buzz-auth/tests -maxdepth 1 -name '*.rs' -print -quit 2>/dev/null | grep -q .; then
|
|
run_test_step "buzz-auth integration tests" \
|
|
cargo test -p buzz-auth --test '*' -- --nocapture
|
|
else
|
|
run_test_step "buzz-auth (no integration tests found)" true
|
|
fi
|
|
|
|
run_test_step "workspace integration tests" \
|
|
cargo test --test '*' -- --nocapture 2>/dev/null || \
|
|
run_test_step "workspace integration tests (none found)" true
|
|
}
|
|
|
|
# ---- Main -------------------------------------------------------------------
|
|
|
|
START_TIME=$(date +%s)
|
|
|
|
case "${MODE}" in
|
|
unit)
|
|
run_unit_tests
|
|
;;
|
|
integration)
|
|
run_integration_tests
|
|
;;
|
|
all|*)
|
|
run_unit_tests
|
|
run_integration_tests
|
|
;;
|
|
esac
|
|
|
|
END_TIME=$(date +%s)
|
|
ELAPSED=$((END_TIME - START_TIME))
|
|
|
|
# ---- Summary ----------------------------------------------------------------
|
|
|
|
section "Test Summary"
|
|
echo ""
|
|
echo -e " Duration: ${ELAPSED}s"
|
|
echo ""
|
|
|
|
if [[ ${#PASSED[@]} -gt 0 ]]; then
|
|
echo -e " ${GREEN}Passed (${#PASSED[@]}):${NC}"
|
|
for t in "${PASSED[@]}"; do
|
|
echo -e " ${GREEN}pass${NC} ${t}"
|
|
done
|
|
fi
|
|
|
|
if [[ ${#FAILED[@]} -gt 0 ]]; then
|
|
echo ""
|
|
echo -e " ${RED}Failed (${#FAILED[@]}):${NC}"
|
|
for t in "${FAILED[@]}"; do
|
|
echo -e " ${RED}fail${NC} ${t}"
|
|
done
|
|
echo ""
|
|
exit 1
|
|
fi
|
|
|
|
echo ""
|
|
success "All tests passed!"
|
|
exit 0
|