Files
roboco/docker-compose.registry.yml
T

423 lines
20 KiB
YAML
Raw Normal View History

# ============================================================================
# RoboCo — pre-built (registry) deployment
# ============================================================================
# This is the "pull and run" compose for USERS: it runs the images the release
# workflow publishes (GHCR + Docker Hub) instead of building from source, so a
# host needs neither the repo's build context nor a build toolchain.
#
# 1. Copy `.env.example` to `.env` and fill in the required secrets.
# 2. docker compose -f docker-compose.registry.yml pull
# 3. docker compose -f docker-compose.registry.yml up -d
#
# Pick the registry + version with two env vars (defaults shown):
# ROBOCO_REGISTRY=ghcr.io/rennf93 # or docker.io/renzof93
# ROBOCO_VERSION=latest # or a pinned release, e.g. 0.5.0
#
# The orchestrator spawns agent containers itself; ROBOCO_AGENT_IMAGE_REGISTRY
# + ROBOCO_AGENT_IMAGE_TAG below tell it to spawn the SAME pre-built agent
# images (it pulls any it doesn't already have). The one-shot agent-*-image
# services exist only so `docker compose pull` fetches every agent image up
# front; they pull then exit.
#
# NOTE: this file is the registry counterpart of docker-compose.yml — when you
# add or change a service there, mirror it here. The infra services
# (postgres/redis/ollama/nginx) are byte-identical to the build compose.
# ============================================================================
services:
postgres:
image: pgvector/pgvector:pg16
container_name: roboco-postgres
restart: unless-stopped
# data-only network: agent containers (on roboco_default) cannot reach
# the production DB; only the multi-homed orchestrator can. Host port
# publishing (15432) is unaffected — roboco_data is a normal bridge.
networks:
- data
environment:
POSTGRES_USER: roboco
POSTGRES_PASSWORD: roboco
POSTGRES_DB: roboco
ports:
- "15432:5432"
volumes:
- ${ROBOCO_DATA_DIR:-./data}/postgres:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U roboco -d roboco"]
interval: 10s
timeout: 5s
retries: 5
redis:
image: redis:8-alpine
container_name: roboco-redis
restart: unless-stopped
# data-only network — see postgres. Redis has no auth, so network
# membership is its ONLY containment against agent containers.
networks:
- data
command: redis-server --appendonly yes
ports:
- "16379:6379"
volumes:
- ${ROBOCO_DATA_DIR:-./data}/redis:/data
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 10s
timeout: 5s
retries: 5
# --------------------------------------------------------------------------
# Backup - periodic pg_dump of the roboco DB (interim; no PITR/WAL yet)
# --------------------------------------------------------------------------
backup:
image: pgvector/pgvector:pg16
container_name: roboco-backup
restart: unless-stopped
# data-only network — pg_dump reaches postgres by container name, same
# as every other data-network consumer; never exposed to the agent mesh.
networks:
- data
environment:
POSTGRES_HOST: roboco-postgres
POSTGRES_PORT: 5432
POSTGRES_USER: roboco
POSTGRES_PASSWORD: roboco
POSTGRES_DB: roboco
entrypoint: ["/bin/bash", "/scripts/backup-entrypoint.sh"]
volumes:
- ./docker/scripts/backup-entrypoint.sh:/scripts/backup-entrypoint.sh:ro
- ${ROBOCO_DATA_DIR:-./data}/backups:/backups
depends_on:
postgres:
condition: service_healthy
ollama:
image: ollama/ollama:latest
container_name: roboco-ollama
restart: unless-stopped
environment:
OLLAMA_API_KEY: ${OLLAMA_API_KEY:-}
ports:
- "11435:11434"
volumes:
- ${ROBOCO_DATA_DIR:-./data}/ollama:/root/.ollama
healthcheck:
test: ["CMD", "ollama", "list"]
interval: 10s
timeout: 5s
retries: 5
start_period: 10s
ollama-init:
image: curlimages/curl:latest
container_name: roboco-ollama-init
depends_on:
ollama:
condition: service_healthy
restart: "no"
entrypoint: ["/bin/sh", "-c"]
command:
- |
# NO `set -e`: pulls are best-effort. Cached models persist in the ollama
# volume and MUST survive a slow/unreachable ollama registry — a manifest
# re-check failure there must never down a fully-cached deployment (it
# used to: a degraded registry made `ollama pull` fail under set -e, which
# blocked the orchestrator's service_completed_successfully gate). Success
# is gated on the models being PRESENT (verify step), not on the pull.
# Note: $$ escapes $ for docker-compose variable substitution.
echo "=== Pulling embedding model (qwen3-embedding:0.6b) — best-effort ==="
curl -sN http://ollama:11434/api/pull -d '{"name":"qwen3-embedding:0.6b"}' | while read -r line; do
status=$$(echo "$$line" | grep -o '"status":"[^"]*"' | cut -d'"' -f4)
[ -n "$$status" ] && echo " $$status"
done || echo " (pull failed — relying on the cached model)"
2026-06-29 05:38:21 +02:00
echo "=== Pulling LLM model (glm-5.2:cloud) — best-effort ==="
curl -sN http://ollama:11434/api/pull -d '{"name":"glm-5.2:cloud"}' | while read -r line; do
status=$$(echo "$$line" | grep -o '"status":"[^"]*"' | cut -d'"' -f4)
[ -n "$$status" ] && echo " $$status"
done || echo " (pull failed — relying on the cached model)"
echo "=== Verifying models are present (the real success gate) ==="
curl -sf http://ollama:11434/api/tags | grep -q "qwen3-embedding" || { echo "FATAL: qwen3-embedding missing and could not be pulled"; exit 1; }
2026-06-29 05:38:21 +02:00
curl -sf http://ollama:11434/api/tags | grep -q "glm-5.2" || { echo "FATAL: glm-5.2 missing and could not be pulled"; exit 1; }
echo "=== All models ready! ==="
# --------------------------------------------------------------------------
# Video Renderer - sidecar for the video-generation engine (renders
# motion/ compositions to MP4 with HyperFrames). Harmless when idle — it
# only renders when POSTed; the video_engine_* feature flags (off by
# default in this compose) gate whether the orchestrator ever sends it
# work. No DB access, no git, no credentials.
# --------------------------------------------------------------------------
video-renderer:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-video-renderer:${ROBOCO_VERSION:-latest}
container_name: roboco-video-renderer
restart: unless-stopped
# Isolated sidecar network: headless Chrome here executes
# agent-authored composition HTML/JS. Only the orchestrator (also
# homed on `render`) can reach it; it can reach nothing else.
networks:
- render
# Chrome headless rendering can crash under Docker's default 64MB
# /dev/shm ("Chrome crashed"); give it real shared memory.
shm_size: "1gb"
mem_limit: "2g"
cpus: 2
healthcheck:
test: ["CMD", "node", "-e", "fetch('http://127.0.0.1:3001/health').then((r) => process.exit(r.ok ? 0 : 1)).catch(() => process.exit(1))"]
interval: 10s
timeout: 5s
retries: 5
start_period: 10s
# --------------------------------------------------------------------------
# Agent image pre-pull (one-shot). Each pulls its published image then exits,
# so `docker compose pull` fetches every agent image up front. The
# orchestrator spawns these same images at runtime.
# --------------------------------------------------------------------------
agent-base-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-base:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-base image present'"]
restart: "no"
agent-pm-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-pm:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-pm image present'"]
restart: "no"
agent-dev-be-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-dev-be:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-dev-be image present'"]
restart: "no"
agent-dev-fe-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-dev-fe:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-dev-fe image present'"]
restart: "no"
agent-qa-be-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-qa-be:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-qa-be image present'"]
restart: "no"
agent-qa-fe-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-qa-fe:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-qa-fe image present'"]
restart: "no"
agent-ux-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-ux:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-ux image present'"]
restart: "no"
agent-doc-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-doc:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-doc image present'"]
restart: "no"
agent-prompter-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-prompter:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-prompter image present'"]
restart: "no"
agent-secretary-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-secretary:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-secretary image present'"]
restart: "no"
agent-pr-reviewer-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-pr-reviewer:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-pr-reviewer image present'"]
restart: "no"
agent-grok-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-grok:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-grok image present'"]
restart: "no"
agent-grok-prompter-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-grok-prompter:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-grok-prompter image present'"]
restart: "no"
agent-grok-secretary-image:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-agent-grok-secretary:${ROBOCO_VERSION:-latest}
entrypoint: ["/bin/sh", "-c", "echo 'agent-grok-secretary image present'"]
restart: "no"
# --------------------------------------------------------------------------
# Orchestrator — API server + agent spawner
# --------------------------------------------------------------------------
orchestrator:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-orchestrator:${ROBOCO_VERSION:-latest}
container_name: roboco-orchestrator
restart: unless-stopped
# Multi-homed: the agent mesh (default) for spawned agents / panel /
# ollama, the data network for postgres/redis, plus render to reach the
# isolated video-renderer sidecar.
networks:
- default
- data
- render
ports:
- "8000:8000"
environment:
# DB network isolation is LIVE in this file (postgres/redis on the
# data-only network): suppress the legacy prod-creds gate-env
# injection — agents can't reach roboco-postgres anyway. This flag
# must always travel with the networks: topology.
ROBOCO_DB_NETWORK_ISOLATED: ${ROBOCO_DB_NETWORK_ISOLATED:-true}
ROBOCO_DATABASE_HOST: roboco-postgres
ROBOCO_DATABASE_PORT: 5432
ROBOCO_DATABASE_USER: roboco
ROBOCO_DATABASE_PASSWORD: roboco
ROBOCO_DATABASE_NAME: roboco
ROBOCO_REDIS_HOST: roboco-redis
ROBOCO_REDIS_PORT: 6379
ROBOCO_HOST: 0.0.0.0
ROBOCO_PORT: 8000
ROBOCO_ENCRYPTION_KEY: ${ROBOCO_ENCRYPTION_KEY:?ROBOCO_ENCRYPTION_KEY is required}
ROBOCO_AGENT_AUTH_SECRET: ${ROBOCO_AGENT_AUTH_SECRET:?ROBOCO_AGENT_AUTH_SECRET is required}
ROBOCO_AGENT_AUTH_REQUIRED: ${ROBOCO_AGENT_AUTH_REQUIRED:-false}
ROBOCO_LOCAL_LLM_BASE_URL: http://roboco-ollama:11434/v1
2026-06-29 05:38:21 +02:00
ROBOCO_LOCAL_LLM_MODEL: glm-5.2:cloud
ROBOCO_DEFAULT_EMBEDDING_MODEL: qwen3-embedding:0.6b
ROBOCO_OLLAMA_BASE_URL: http://roboco-ollama:11434
# Video renderer sidecar (use container name). The video engine
# itself is default-off (video_engine_enabled); this just points the
# client at the sidecar for when it's armed.
ROBOCO_VIDEO_RENDERER_BASE_URL: http://roboco-video-renderer:3001
# Spawn the PRE-BUILT agent images from the same registry instead of
# building them from source (the orchestrator pulls any it lacks).
ROBOCO_AGENT_IMAGE_REGISTRY: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}
ROBOCO_AGENT_IMAGE_TAG: ${ROBOCO_VERSION:-latest}
# Host paths for spawning agent containers (Docker-in-Docker). MUST be
# absolute paths on the host. Default to this compose project's ./data.
ROBOCO_HOST_PROJECT_DIR: ${ROBOCO_HOST_PROJECT_DIR:-/opt/roboco}
ROBOCO_HOST_CLAUDE_DIR: ${ROBOCO_HOST_CLAUDE_DIR:-${HOME}/.claude}
# SuperGrok auth (host ~/.grok) for Grok-CLI agents — the orchestrator
# mounts <dir>/auth.json into each Grok agent. Run `grok login` on the host.
ROBOCO_HOST_GROK_DIR: ${ROBOCO_HOST_GROK_DIR:-${HOME}/.grok}
ROBOCO_HOST_DATA_DIR: ${ROBOCO_HOST_DATA_DIR:-/opt/roboco/data}
# Reachable base URL for commit-trailer links — set to your host's LAN
# address or domain so the links in commit bodies resolve.
ROBOCO_PUBLIC_BASE_URL: ${ROBOCO_PUBLIC_BASE_URL:-http://localhost:8000}
ROBOCO_ENVIRONMENT: production
ROBOCO_CLAIM_STALE_SECONDS: "1800"
ROBOCO_STALE_CLAIM_REAP_SECONDS: "1800"
# Every optional feature ships OFF in this user-facing compose (the build
# compose arms them for the personal deploy). External-PR review included —
# arm it (and any other subsystem) via .env or Settings → Feature Flags.
ROBOCO_EXTERNAL_PR_ENABLED: ${ROBOCO_EXTERNAL_PR_ENABLED:-false}
ROBOCO_EXTERNAL_PR_REQUIRE_HUMAN_CONFIRM: ${ROBOCO_EXTERNAL_PR_REQUIRE_HUMAN_CONFIRM:-true}
# ROBOCO_EXTERNAL_PR_POLL_INTERVAL_SECONDS: "300"
# ROBOCO_EXTERNAL_PR_AUTHOR_ALLOWLIST: '["corey"]' # empty = every external PR
# Production self-healing ("engine 4"). RoboCo watches its OWN repo CI and,
# when red, notifies the CEO and (with originate on) opens a PENDING fix
# task that STOPS for the CEO's Approve-&-Start — it never self-deploys.
# Both toggles default OFF (arm from Settings -> Feature Flags). Set
# PROJECT_SLUG to the registered project that IS RoboCo; CI_WORKFLOW scopes
# the signal to the real CI workflow (RoboCo has several workflows, so the
# unscoped "latest run" would be unreliable).
ROBOCO_SELF_HEAL_ENABLED: ${ROBOCO_SELF_HEAL_ENABLED:-false}
ROBOCO_OBSIDIAN_VAULT_ENABLED: ${ROBOCO_OBSIDIAN_VAULT_ENABLED:-true}
ROBOCO_VAULT_PATH: ${ROBOCO_VAULT_PATH:-/app/vault}
ROBOCO_VAULT_INTAKE_ENABLED: ${ROBOCO_VAULT_INTAKE_ENABLED:-true}
ROBOCO_VAULT_ARCHIVE_DAYS: ${ROBOCO_VAULT_ARCHIVE_DAYS:-30}
ROBOCO_VAULT_REPORT_ENABLED: ${ROBOCO_VAULT_REPORT_ENABLED:-true}
ROBOCO_VAULT_KB_ENABLED: ${ROBOCO_VAULT_KB_ENABLED:-false}
ROBOCO_VAULT_KB_DIRS: ${ROBOCO_VAULT_KB_DIRS:-RoboCo/Notes}
ROBOCO_VAULT_KB_INTERVAL_SECONDS: ${ROBOCO_VAULT_KB_INTERVAL_SECONDS:-900}
ROBOCO_SELF_HEAL_ORIGINATE_ENABLED: ${ROBOCO_SELF_HEAL_ORIGINATE_ENABLED:-false}
ROBOCO_SELF_HEAL_PROJECT_SLUG: ${ROBOCO_SELF_HEAL_PROJECT_SLUG:-roboco-api}
ROBOCO_SELF_HEAL_CI_WORKFLOW: ${ROBOCO_SELF_HEAL_CI_WORKFLOW:-ci.yml}
# Video engine (HyperFrames): NAS-default-on, OFF in public registry —
# heavier optional feature. Arm via .env + uncomment the video-renders
# bind mount above so renders persist.
ROBOCO_VIDEO_ENGINE_ENABLED: ${ROBOCO_VIDEO_ENGINE_ENABLED:-false}
ROBOCO_VIDEO_ON_RELEASE: ${ROBOCO_VIDEO_ON_RELEASE:-false}
ROBOCO_VIDEO_ON_SPOTLIGHT: ${ROBOCO_VIDEO_ON_SPOTLIGHT:-false}
# MinIO object storage for rendered videos is intentionally omitted from
# this registry compose (NAS default-on, registry default-off — the
# established pattern). The minio/minio-init services are absent here and
# ROBOCO_MINIO_* is left unset, so minio_endpoint defaults to empty and the
# media route falls back to FileResponse. Arm via a custom override file.
volumes:
- /var/run/docker.sock:/var/run/docker.sock
- ${CLAUDE_AUTH_DIR:-${HOME}/.claude}:/root/.claude
# SuperGrok auth — mount host ~/.grok at the SAME host path the orchestrator
# hands each Grok agent's `-v`, so its auth.json exists() check passes here
# AND the agent bind resolves on the host. Read-WRITE: the orchestrator
# auto-refreshes the ~6h token in place (grok_auth.refresh_if_stale) so
# agents never mount a dead credential; the agent's own mount stays RO.
- ${ROBOCO_HOST_GROK_DIR:-${HOME}/.grok}:${ROBOCO_HOST_GROK_DIR:-${HOME}/.grok}
- ${ROBOCO_DATA_DIR:-./data}/mcp-configs:/app/mcp-configs
- ${ROBOCO_DATA_DIR:-./data}/vault:/app/vault
- ${ROBOCO_DATA_DIR:-./data}/prompts-generated:/app/prompts-generated
- ${ROBOCO_DATA_DIR:-./data}/agent-settings:/app/agent-settings
- ${ROBOCO_DATA_DIR:-./data}/workspaces:/data/workspaces
# Per-agent GROK usage capture (usage.json -> finalizer).
- ${ROBOCO_DATA_DIR:-./data}/grok-usage:/data/grok-usage
- ${ROBOCO_DATA_DIR:-./data}/logs:/data/logs
# video engine: NAS-only bind mount, off by default in public registry.
# Uncomment + set ROBOCO_VIDEO_ENGINE_ENABLED=true in .env to arm.
# - ${ROBOCO_DATA_DIR:-./data}/video-renders:/data/video-renders
- ${ROBOCO_DATA_DIR:-./data}/briefings:/app/briefings
- ${ROBOCO_DATA_DIR:-./data}/manifests:/app/manifests
depends_on:
postgres:
condition: service_healthy
redis:
condition: service_healthy
ollama:
condition: service_healthy
ollama-init:
condition: service_completed_successfully
agent-base-image:
condition: service_completed_successfully
# --------------------------------------------------------------------------
# Next.js control panel (fronted by nginx; not exposed directly)
# --------------------------------------------------------------------------
panel:
image: ${ROBOCO_REGISTRY:-ghcr.io/rennf93}/roboco-panel:${ROBOCO_VERSION:-latest}
container_name: roboco-panel
restart: unless-stopped
expose:
- "3000"
depends_on:
- orchestrator
# --------------------------------------------------------------------------
# Nginx — single entry point on port 3000
# --------------------------------------------------------------------------
nginx:
image: nginx:alpine
container_name: roboco-nginx
restart: unless-stopped
ports:
- "3000:80"
environment:
ROBOCO_PANEL_AGENT_TOKEN: ${ROBOCO_PANEL_AGENT_TOKEN:-}
NGINX_ENVSUBST_FILTER: "^ROBOCO_"
volumes:
- ./docker/nginx.conf:/etc/nginx/templates/default.conf.template:ro
depends_on:
- panel
- orchestrator
networks:
default:
name: roboco_default
# DB isolation: postgres/redis live ONLY here; the orchestrator is the
# only service homed on both. Spawned agent containers + sandbox sidecars
# join roboco_default (AGENT_NETWORK) and cannot resolve or reach the
# production DB/Redis. Normal bridge (not internal) so host-published
# ports keep working.
data:
name: roboco_data
# Sidecar isolation: video-renderer executes agent-authored composition
# HTML/JS in headless Chrome. It lives ONLY here, reachable only by the
# orchestrator (multi-homed onto this network too); it can reach nothing
# else on roboco_default or roboco_data.
render:
name: roboco_render