mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
Resolves the 100% claim-failure rate introduced by the gateway rewrite
(commit 62bda0c plus 78 follow-ups). Live smoke runs hit
`404 /api/v2/flow/developer/...` on every dev verb plus a manifest
fallback that silently exposed off-role verbs to PMs — confirmed
firing simultaneously in NAS agent logs (be-dev-1, be-pm, main-pm).
Audit reports under docs/internal/audit_2026_05_04/ catalogue 49
defects across gateway, services, prompts, MCP transport, substrate,
and tests (8 detail reports + master synthesis). Six smoking guns;
three proven in production logs.
Phase 0 — unblock claim:
- URL prefix /api/v2/flow/dev → /developer; slug-map board roles
(product_owner, head_marketing) → /board (D-01)
- _i_will_work_on AttributeError on None across pending /
needs_revision / claimed re-entry branches (D-02)
- Seed last_heartbeat_at in _qa_or_doc_claim (D-03)
- Drop misleading i_have_committed verb; dev flow uses commit() (D-04)
- Manifest mount via compose; flow_server + do_server fail loud
instead of exposing all-verbs fallback (D-12)
- MCP _post() surfaces envelope body on 4xx so agents see remediate
hints (D-13)
on git failure so retries aren't blocked by half-state (S-01)
Phase 1 — lifecycle stability:
- _resolve_skill falls back to AgentTable.capabilities (D-06)
- main_pm_complete uses kwargs for escalate_to_ceo (D-07)
- i_am_done auto-runs submit_verification when in_progress (D-08)
- active_claimant_id wired in claim/unclaim paths — single-claimant
invariant now functional (D-05)
- qa_pass/qa_fail assert claimed_by parity with qa_agent_id (D-18)
- Prompt-drift sweep: fail() shape, i_am_done(task_id, notes),
subtask cap (12 hard / 8 soft), error-code symbology rewritten in
base.md + per-role anti-patterns (D-10/11/29/30/31, D-37)
Phase 2 — invariants + architecture:
- Real-DB integration test exercising claim → in_progress → commit
→ submit_for_qa → i_am_done → awaiting_qa (P2-1)
- choreographer.py → package; 3 of 6 role mixins extracted
(board, doc, qa). _impl.py 2,526 → 2,080 lines (-18%). Continuation
plan in docs/internal/audit_2026_05_04/p2_2_decompose_plan.md (P2-2)
- Closure guards consolidated via _subtasks_not_terminal_envelope (P2-3)
- TaskService.unclaim_for_reaper routed through canonical
_validate_and_set_status; in_progress → pending added to
VALID_TRANSITIONS (P2-4)
- Dead code removed: i_am_done_with_catchup verb, _run_catch_up helper
(P2-5)
- 6 state-machine invariants asserted via property test (P2-6)
- attempt_id (uuid4) stamped on every gateway.rejected audit row (P2-7)
- _reconcile_orphan_claims_on_startup rolls back tasks left CLAIMED
with branch_name=NULL from prior crashes (P2-8)
- scripts/regenerate_verb_tables.py introspects Pydantic schemas +
role_config; compose_prompt injects per-role tables as a layer.
Eliminates the prompt-drift class structurally (P2-9)
Other:
- D-48: orchestrator mounts host's ~/.claude.json when present so
agents don't boot from backup recovery on every spawn
- D-49: dev dispatcher rejects role-mismatched spawns (e.g. doc task
assigned to dev agent)
Tests: 553 pass · ruff + mypy clean. Live NAS smoke verification
pending — needs the stack brought back up.
314 lines
12 KiB
YAML
314 lines
12 KiB
YAML
services:
|
|
# ==========================================================================
|
|
# PostgreSQL - Primary Database with pgvector for RAG
|
|
# ==========================================================================
|
|
postgres:
|
|
image: pgvector/pgvector:pg16
|
|
container_name: roboco-postgres
|
|
restart: unless-stopped
|
|
environment:
|
|
POSTGRES_USER: roboco
|
|
POSTGRES_PASSWORD: roboco
|
|
POSTGRES_DB: roboco
|
|
ports:
|
|
- "15432:5432"
|
|
volumes:
|
|
- ${ROBOCO_DATA_DIR:-./data}/postgres:/var/lib/postgresql/data
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U roboco -d roboco"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 5
|
|
|
|
# ==========================================================================
|
|
# Redis - Cache, Sessions, Event Bus
|
|
# ==========================================================================
|
|
redis:
|
|
image: redis:8-alpine
|
|
container_name: roboco-redis
|
|
restart: unless-stopped
|
|
command: redis-server --appendonly yes
|
|
ports:
|
|
- "16379:6379"
|
|
volumes:
|
|
- ${ROBOCO_DATA_DIR:-./data}/redis:/data
|
|
healthcheck:
|
|
test: ["CMD", "redis-cli", "ping"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 5
|
|
|
|
# ==========================================================================
|
|
# Ollama - Local LLM and Embedding Server
|
|
# ==========================================================================
|
|
ollama:
|
|
image: ollama/ollama:latest
|
|
container_name: roboco-ollama
|
|
restart: unless-stopped
|
|
environment:
|
|
OLLAMA_API_KEY: ${OLLAMA_API_KEY}
|
|
ports:
|
|
- "11435:11434"
|
|
volumes:
|
|
- ${ROBOCO_DATA_DIR:-./data}/ollama:/root/.ollama
|
|
healthcheck:
|
|
# Use ollama CLI (guaranteed available) to check if server is responding
|
|
test: ["CMD", "ollama", "list"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 5
|
|
start_period: 10s
|
|
|
|
# Ollama model puller - pulls required models on startup
|
|
# Uses streaming curl to wait for full model download
|
|
ollama-init:
|
|
image: curlimages/curl:latest
|
|
container_name: roboco-ollama-init
|
|
depends_on:
|
|
ollama:
|
|
condition: service_healthy
|
|
restart: "no"
|
|
entrypoint: ["/bin/sh", "-c"]
|
|
command:
|
|
- |
|
|
set -e
|
|
echo "=== Pulling embedding model (qwen3-embedding:0.6b) ==="
|
|
# Ollama /api/pull streams JSON lines until complete - consume full stream
|
|
# Note: $$ escapes $ for docker-compose variable substitution
|
|
curl -sN http://ollama:11434/api/pull -d '{"name":"qwen3-embedding:0.6b"}' | while read -r line; do
|
|
status=$$(echo "$$line" | grep -o '"status":"[^"]*"' | cut -d'"' -f4)
|
|
[ -n "$$status" ] && echo " $$status"
|
|
done
|
|
echo "=== Pulling LLM model (glm-5:cloud) ==="
|
|
curl -sN http://ollama:11434/api/pull -d '{"name":"glm-5:cloud"}' | while read -r line; do
|
|
status=$$(echo "$$line" | grep -o '"status":"[^"]*"' | cut -d'"' -f4)
|
|
[ -n "$$status" ] && echo " $$status"
|
|
done
|
|
echo "=== Verifying models are available ==="
|
|
curl -sf http://ollama:11434/api/tags | grep -q "qwen3-embedding" && echo " qwen3-embedding: OK"
|
|
curl -sf http://ollama:11434/api/tags | grep -q "glm-5" && echo " glm-5: OK"
|
|
echo "=== All models ready! ==="
|
|
|
|
# ==========================================================================
|
|
# Agent Base Image Builder (specialized images built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-base-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-base.Dockerfile
|
|
image: roboco-agent-base
|
|
container_name: roboco-agent-base-builder
|
|
entrypoint: ["/bin/sh", "-c", "echo 'Agent base image built successfully'"]
|
|
restart: "no"
|
|
|
|
# ==========================================================================
|
|
# Agent PM Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-pm-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-pm.Dockerfile
|
|
image: roboco-agent-pm
|
|
entrypoint: ["/bin/sh", "-c", "echo 'Agent PM image built'"]
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent Backend Dev Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-dev-be-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-dev-be.Dockerfile
|
|
image: roboco-agent-dev-be
|
|
entrypoint: ["/bin/sh", "-c", "echo 'Agent Backend Dev image built'"]
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent Frontend Dev Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-dev-fe-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-dev-fe.Dockerfile
|
|
image: roboco-agent-dev-fe
|
|
entrypoint: ["/bin/sh", "-c", "echo 'Agent Frontend Dev image built'"]
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent Backend QA Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-qa-be-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-qa-be.Dockerfile
|
|
image: roboco-agent-qa-be
|
|
entrypoint: ["/bin/sh", "-c", 'echo "Agent Backend QA image built"']
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent Frontend QA Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-qa-fe-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-qa-fe.Dockerfile
|
|
image: roboco-agent-qa-fe
|
|
entrypoint: ["/bin/sh", "-c", 'echo "Agent Frontend QA image built"']
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent UX/UI Dev Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-ux-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-ux.Dockerfile
|
|
image: roboco-agent-ux
|
|
entrypoint: ["/bin/sh", "-c", 'echo "Agent UX/UI image built"']
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Agent Documenter Image Builder (specialized image built on-demand by orchestrator)
|
|
# ==========================================================================
|
|
agent-doc-image:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/agent-doc.Dockerfile
|
|
image: roboco-agent-doc
|
|
entrypoint: ["/bin/sh", "-c", 'echo "Agent Documenter image built"']
|
|
restart: "no"
|
|
depends_on:
|
|
- agent-base-image
|
|
|
|
# ==========================================================================
|
|
# Orchestrator - API Server + Agent Spawner
|
|
# ==========================================================================
|
|
orchestrator:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/orchestrator.Dockerfile
|
|
image: roboco-orchestrator
|
|
container_name: roboco-orchestrator
|
|
restart: unless-stopped
|
|
ports:
|
|
- "8000:8000"
|
|
environment:
|
|
# Database (use container name, not localhost)
|
|
ROBOCO_DATABASE_HOST: roboco-postgres
|
|
ROBOCO_DATABASE_PORT: 5432
|
|
ROBOCO_DATABASE_USER: roboco
|
|
ROBOCO_DATABASE_PASSWORD: roboco
|
|
ROBOCO_DATABASE_NAME: roboco
|
|
# Redis (use container name)
|
|
ROBOCO_REDIS_HOST: roboco-redis
|
|
ROBOCO_REDIS_PORT: 6379
|
|
# API
|
|
ROBOCO_HOST: 0.0.0.0
|
|
ROBOCO_PORT: 8000
|
|
ROBOCO_ENCRYPTION_KEY: ${ROBOCO_ENCRYPTION_KEY:?ROBOCO_ENCRYPTION_KEY is required}
|
|
# HMAC secret for agent auth tokens. Orchestrator signs tokens
|
|
# per-agent at spawn; API middleware verifies them. Generate with:
|
|
# python -c 'import secrets; print(secrets.token_hex(32))'
|
|
ROBOCO_AGENT_AUTH_SECRET: ${ROBOCO_AGENT_AUTH_SECRET:?ROBOCO_AGENT_AUTH_SECRET is required}
|
|
# Set to "true" to require tokens on every API call (fail-closed).
|
|
# Leave unset/false during rollout so the panel + curl still work.
|
|
ROBOCO_AGENT_AUTH_REQUIRED: ${ROBOCO_AGENT_AUTH_REQUIRED:-false}
|
|
# Ollama (use container name)
|
|
ROBOCO_LOCAL_LLM_BASE_URL: http://roboco-ollama:11434/v1
|
|
ROBOCO_LOCAL_LLM_MODEL: glm-5:cloud
|
|
ROBOCO_DEFAULT_EMBEDDING_MODEL: qwen3-embedding:0.6b
|
|
ROBOCO_OLLAMA_BASE_URL: http://roboco-ollama:11434
|
|
# Host paths for spawning agent containers (required for Docker-in-Docker)
|
|
# IMPORTANT: These must be ABSOLUTE paths on the host filesystem
|
|
ROBOCO_HOST_PROJECT_DIR: ${ROBOCO_HOST_PROJECT_DIR:-/volume1/roboco}
|
|
ROBOCO_HOST_CLAUDE_DIR: ${ROBOCO_HOST_CLAUDE_DIR:-/home/renzof/.claude}
|
|
ROBOCO_HOST_DATA_DIR: ${ROBOCO_HOST_DATA_DIR:-/volume1/roboco/data}
|
|
volumes:
|
|
# Docker socket - allows spawning agent containers
|
|
- /var/run/docker.sock:/var/run/docker.sock
|
|
# Claude Code auth - mount your ~/.claude directory
|
|
- ${CLAUDE_AUTH_DIR:-/home/renzof/.claude}:/root/.claude
|
|
# Shared config directory for MCP configs (writable)
|
|
- ${ROBOCO_DATA_DIR:-./data}/mcp-configs:/app/mcp-configs
|
|
# Generated prompts directory - composed at runtime from layers
|
|
- ${ROBOCO_DATA_DIR:-./data}/prompts-generated:/app/prompts-generated
|
|
# Per-agent Claude settings (generated at spawn time)
|
|
- ${ROBOCO_DATA_DIR:-./data}/agent-settings:/app/agent-settings
|
|
# Per-agent SessionStart briefings (pre-rendered task context)
|
|
- ${ROBOCO_DATA_DIR:-./data}/briefings:/app/briefings
|
|
# Per-agent spawn manifests (role-scoped tool list) — written by the
|
|
# orchestrator, bind-mounted into each agent container as
|
|
# /app/tool-manifest.json. Without this mount the file is invisible
|
|
# to the Docker daemon and agents fall back to all-verbs registration.
|
|
- ${ROBOCO_DATA_DIR:-./data}/manifests:/app/manifests
|
|
# Agent workspaces (git clones) - persisted across restarts
|
|
- ${ROBOCO_DATA_DIR:-./data}/workspaces:/data/workspaces
|
|
# Persistent logs — survive `docker compose down/up`. Orchestrator and
|
|
# each spawned agent write structured logs here so we can audit past
|
|
# runs instead of relying on ephemeral `docker logs`.
|
|
- ${ROBOCO_DATA_DIR:-./data}/logs:/data/logs
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
redis:
|
|
condition: service_healthy
|
|
ollama:
|
|
condition: service_healthy
|
|
ollama-init:
|
|
condition: service_completed_successfully
|
|
agent-base-image:
|
|
condition: service_completed_successfully
|
|
# Default agents to spawn (override in .env or command line)
|
|
# command: ["--spawn", "main-pm", "be-dev-1", "be-qa"]
|
|
|
|
# ==========================================================================
|
|
# Next.js Control Panel (Frontend)
|
|
# ==========================================================================
|
|
# Not exposed directly - nginx is the single entry point (port 3000).
|
|
# API/WS traffic is proxied to the orchestrator, everything else to panel.
|
|
panel:
|
|
build:
|
|
context: .
|
|
dockerfile: docker/panel.Dockerfile
|
|
image: roboco-panel
|
|
container_name: roboco-panel
|
|
restart: unless-stopped
|
|
expose:
|
|
- "3000"
|
|
depends_on:
|
|
- orchestrator
|
|
|
|
# ==========================================================================
|
|
# Nginx - Reverse proxy fronting panel + orchestrator
|
|
# ==========================================================================
|
|
# Single entry point on port 3000 so the browser hits one origin and we
|
|
# don't need CORS. /api/* and /ws/* go to the orchestrator, everything
|
|
# else goes to the Next.js panel.
|
|
nginx:
|
|
image: nginx:alpine
|
|
container_name: roboco-nginx
|
|
restart: unless-stopped
|
|
ports:
|
|
- "3000:80"
|
|
volumes:
|
|
- ./docker/nginx.conf:/etc/nginx/conf.d/default.conf:ro
|
|
depends_on:
|
|
- panel
|
|
- orchestrator
|
|
|
|
networks:
|
|
default:
|
|
name: roboco_default
|