Files
buzz/scripts/run-tests.sh
Michael Neale 9ca88061fa Merge origin/main into micn/buzz-agent-goose-core
82 commits of main, plus the goose pin moved from bf332b9 to 7c4ba22
(60 commits) because main's changes and goose's API changes overlap.

Conflict resolutions worth knowing about:

* `llm.rs`, `handoff.rs`, `tests/regressions.rs` — main changed files this
  branch deletes. Reviewed each change rather than dropping it silently:
  #5475 raised `BUZZ_AGENT_MAX_OUTPUT_TOKENS` and the truncation-recovery
  allowance in buzz's own request loop, which no longer exists; goose owns
  retries and output limits now, so there is nothing to port.

* `model_capabilities.rs` (#5597) moved to `buzz-model-catalog`, not kept in
  `buzz-agent`. The manifest is read by the desktop model picker, which
  cannot link goose (`libsqlite3-sys` collision), so it has to live in the
  crate the desktop already depends on. `ThinkingEffort` moved with it,
  reduced to the vocabulary the manifest is typed in — the request-path
  mapping went with buzz's HTTP transport. The 103-vector corpus guard
  passes in its new home; `just ci` and `regen-model-corpus` now point at
  it, and `run-tests.sh` runs the new crate's lib tests.

* Curated model labels reach the picker. main added them to the Databricks
  discovery path; on this branch `session/new` enumerates through goose's
  provider API instead, so the label lookup had to be applied there or
  #5597 would have been silently reverted for every provider.

goose API changes this bump required:

* `StateMachine`/`Step`/`Operation` are generic over session and effect
  type, and the loop moved into a new `goose-agent` crate.
* `StateEffect` split: buzz now names `ConversationEffect`, the narrow set,
  rather than goose's `GooseEffect`. `ReplaceConversation` lost its `usage`
  field — buzz always passed `None`, and the driving loop already resets
  the running total, so this is a rename not a behaviour change.

78 unit + 22 integration tests pass, clippy `-D warnings` and fmt clean on
the pinned 1.95.0 toolchain.

Co-authored-by: Michael Neale <michael.neale@gmail.com>
Signed-off-by: Michael Neale <michael.neale@gmail.com>
2026-08-18 11:43:30 +10:00

195 lines
6.2 KiB
Bash
Executable File

#!/usr/bin/env bash
# =============================================================================
# run-tests.sh — Run Buzz test suite
# =============================================================================
# Usage:
# ./scripts/run-tests.sh # run all tests (default)
# ./scripts/run-tests.sh unit # unit tests only (no infra needed)
# ./scripts/run-tests.sh integration # integration tests only
# ./scripts/run-tests.sh all # explicit all
# =============================================================================
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"
MODE="${1:-all}"
# Colors
RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
BLUE='\033[0;34m'
CYAN='\033[0;36m'
NC='\033[0m'
log() { echo -e "${BLUE}[run-tests]${NC} $*"; }
success(){ echo -e "${GREEN}[run-tests]${NC} $*"; }
warn() { echo -e "${YELLOW}[run-tests]${NC} $*"; }
error() { echo -e "${RED}[run-tests]${NC} $*" >&2; }
section(){ echo -e "\n${CYAN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"; echo -e "${CYAN} $*${NC}"; echo -e "${CYAN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"; }
cd "${REPO_ROOT}"
# ---- Load .env if present ---------------------------------------------------
if [[ -f ".env" ]]; then
log "Loading .env..."
set -o allexport
# shellcheck disable=SC1091
source .env
set +o allexport
else
# Use defaults matching docker-compose.yml
export DATABASE_URL="postgres://buzz:buzz_dev@localhost:5432/buzz" # sadscan:disable np.postgres.1
export PGHOST=localhost
export PGPORT=5432
export PGUSER=buzz
export PGPASSWORD=buzz_dev
export PGDATABASE=buzz
export REDIS_URL="redis://localhost:6379"
fi
# ---- Track results ----------------------------------------------------------
declare -a PASSED=()
declare -a FAILED=()
run_test_step() {
local name="$1"
shift
log "Running: ${name}"
if "$@"; then
success "${name} passed"
PASSED+=("${name}")
else
error "${name} FAILED"
FAILED+=("${name}")
fi
}
# ---- Check / start infra (for integration tests) ----------------------------
ensure_infra() {
"${REPO_ROOT}/bin/just" _ensure-migrations
}
# ---- Unit tests (no infra needed) -------------------------------------------
run_unit_tests() {
section "Unit Tests (no infra required)"
run_test_step "buzz-core tests" \
cargo test -p buzz-core --lib -- --nocapture
run_test_step "buzz-auth unit tests" \
cargo test -p buzz-auth --lib -- --nocapture
run_test_step "buzz-voice tests" \
cargo test -p buzz-voice --lib -- --nocapture
run_test_step "buzz-cli tests" \
cargo test -p buzz-cli -- --nocapture
# buzz-db migrator/lint unit tests (no infra): guard the embedded-migrator
# invariant (exactly the consolidated 0001; cutover/backfill stays an operator
# script, not startup state) and the tenant-scoping lints. The Postgres-backed
# buzz-db tests are #[ignore]d; nothing here (or in integration mode below,
# which runs `cargo test -p buzz-db` without --ignored) runs them — they need a
# separate isolated-DB gate, so --lib keeps this step infra-free.
run_test_step "buzz-db unit tests" \
cargo test -p buzz-db --lib -- --nocapture
# Multi-tenant conformance gate: independent replay checker + golden
# fixtures (buzz-conformance). Pure in-process trace replay, no infra.
run_test_step "buzz-conformance tests" \
cargo test -p buzz-conformance -- --nocapture
run_test_step "buzz-push-gateway tests" \
cargo test -p buzz-push-gateway -- --nocapture
# Kubernetes backend provider: pure decision layers driven by a fake
# substrate, no cluster. Mirrors the nextest path in `just test-unit` —
# the two lists must stay in step or the fallback silently covers less.
run_test_step "buzz-backend-kubernetes tests" \
cargo test -p buzz-backend-kubernetes -- --nocapture
# buzz-agent model-capabilities corpus: the Rust half of the cross-language
# drift guard. model_capabilities.rs embeds scripts/model-capabilities.json +
# scripts/normative-corpus.json via include_str! and replays all 103 vectors
# as pure in-process tests (no infra). Mirrors the nextest path in
# `just test-unit` — the two lists must stay in step.
run_test_step "buzz-agent unit tests" \
cargo test -p buzz-agent --lib -- --nocapture
cargo test -p buzz-model-catalog --lib -- --nocapture
}
# ---- DB / integration tests (infra required) --------------------------------
run_integration_tests() {
section "Integration Tests (requires running services)"
ensure_infra
run_test_step "buzz-db tests" \
cargo test -p buzz-db -- --nocapture
if find crates/buzz-auth/tests -maxdepth 1 -name '*.rs' -print -quit 2>/dev/null | grep -q .; then
run_test_step "buzz-auth integration tests" \
cargo test -p buzz-auth --test '*' -- --nocapture
else
run_test_step "buzz-auth (no integration tests found)" true
fi
run_test_step "workspace integration tests" \
cargo test --test '*' -- --nocapture 2>/dev/null || \
run_test_step "workspace integration tests (none found)" true
}
# ---- Main -------------------------------------------------------------------
START_TIME=$(date +%s)
case "${MODE}" in
unit)
run_unit_tests
;;
integration)
run_integration_tests
;;
all|*)
run_unit_tests
run_integration_tests
;;
esac
END_TIME=$(date +%s)
ELAPSED=$((END_TIME - START_TIME))
# ---- Summary ----------------------------------------------------------------
section "Test Summary"
echo ""
echo -e " Duration: ${ELAPSED}s"
echo ""
if [[ ${#PASSED[@]} -gt 0 ]]; then
echo -e " ${GREEN}Passed (${#PASSED[@]}):${NC}"
for t in "${PASSED[@]}"; do
echo -e " ${GREEN}pass${NC} ${t}"
done
fi
if [[ ${#FAILED[@]} -gt 0 ]]; then
echo ""
echo -e " ${RED}Failed (${#FAILED[@]}):${NC}"
for t in "${FAILED[@]}"; do
echo -e " ${RED}fail${NC} ${t}"
done
echo ""
exit 1
fi
echo ""
success "All tests passed!"
exit 0