mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
fix: post-finale completeness sweep — routing surface, provider config, budgets, compose env, interactive exemption (#661)
This commit is contained in:
@@ -151,6 +151,55 @@ ROBOCO_DEFAULT_EMBEDDING_MODEL=qwen3-embedding:0.6b
|
||||
# 0 disables. Backstops runaway-loop token burn.
|
||||
# ROBOCO_GROK_MAX_COST_USD=0.0
|
||||
|
||||
# =============================================================================
|
||||
# Codex (OpenAI) Provider — optional
|
||||
# =============================================================================
|
||||
# RoboCo can run agents on OpenAI's Codex CLI (ChatGPT subscription auth) via a
|
||||
# mounted ~/.codex/auth.json — run `codex login` once on the host. Enable it by
|
||||
# picking the "Codex" routing mode or a gpt-* model per agent in the panel's AI
|
||||
# routing card. Host dir mounted RO into each Codex agent; the orchestrator
|
||||
# refreshes the token before expiry (codex_auth.py). All vars optional.
|
||||
# ROBOCO_HOST_CODEX_DIR=/home/youruser/.codex
|
||||
# ROBOCO_CODEX_CLI_MODEL=gpt-5.3-codex
|
||||
# If the default OIDC client id is wrong for your account, override it (a bad
|
||||
# refresh never mutates auth.json — worst case is a parked provider):
|
||||
# ROBOCO_CODEX_OAUTH_CLIENT_ID=
|
||||
|
||||
# =============================================================================
|
||||
# Gemini (Google) Provider — optional
|
||||
# =============================================================================
|
||||
# RoboCo can run agents on Google's Gemini CLI (Google-account OAuth) via a
|
||||
# mounted ~/.gemini — run the interactive `gemini` login once on the host.
|
||||
# Enable via the "Gemini" routing mode or a gemini-* model per agent. Each
|
||||
# container copies the RO-mounted creds to a writable local dir and refreshes
|
||||
# in-process (reusable refresh tokens, no orchestrator daemon). All optional.
|
||||
# ROBOCO_HOST_GEMINI_DIR=/home/youruser/.gemini
|
||||
# ROBOCO_GEMINI_CLI_MODEL=gemini-2.5-pro
|
||||
# Hard ceiling on agentic turns per run (loop guard, grok parity):
|
||||
# ROBOCO_GEMINI_MAX_TURNS=200
|
||||
# Park-and-retry delays after a rate-limit / auth failure (seconds):
|
||||
# ROBOCO_GEMINI_RATE_LIMIT_RETRY_AFTER_SECONDS=300
|
||||
# ROBOCO_GEMINI_AUTH_RETRY_AFTER_SECONDS=300
|
||||
|
||||
# =============================================================================
|
||||
# Cost budgets — optional
|
||||
# =============================================================================
|
||||
# Per-task (tasks.budget_usd) and per-project (projects.monthly_budget_usd)
|
||||
# cost caps. Default-off subsystem; also toggleable on the panel's Feature
|
||||
# Flags card. A breached task is BLOCKED (not silently killed) and the CEO is
|
||||
# notified. Build compose arms it true; registry compose leaves it false.
|
||||
# ROBOCO_TASK_BUDGETS_ENABLED=false
|
||||
|
||||
# =============================================================================
|
||||
# Notification re-escalation backoff — optional tuning
|
||||
# =============================================================================
|
||||
# Expired unacked ack-required notifications re-escalate on exponential backoff
|
||||
# (first at expiry, then doubling from the base, capped at 24h) up to a max
|
||||
# count, instead of re-firing every sweep tick. No panel UI — env/compose is
|
||||
# the only tuning path.
|
||||
# ROBOCO_NOTIFICATION_REESCALATION_BASE_SECONDS=3600
|
||||
# ROBOCO_NOTIFICATION_MAX_REESCALATIONS=5
|
||||
|
||||
# =============================================================================
|
||||
# Security
|
||||
# =============================================================================
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Flip the Gemini (Google) provider row to enabled=true.
|
||||
|
||||
Migration 085 seeded the row `enabled=false`, pending an operator OAuth
|
||||
setup — but nothing ever flipped it. Unlike Grok (enabled=true only via the
|
||||
`apply_mode="grok"` write path, which force-enables the row at apply time),
|
||||
Gemini had no equivalent enable step at all: `apply_mode` grew no "gemini"
|
||||
case until this same change, so any Mix-mode assignment to a Gemini model
|
||||
resolved through `resolve_for_agent` against a permanently-disabled row and
|
||||
silently fell back to the legacy Anthropic path — the provider was wired
|
||||
end-to-end everywhere except reachable.
|
||||
|
||||
Codex (migration 083, `083_seed_openai_provider`) is the closer parity
|
||||
target: both are subscription-CLI providers with no API key to withhold
|
||||
behind a disabled row (`~/.codex` / `~/.gemini`, mounted OAuth/subscription
|
||||
credentials, not a stored token), and Codex seeds `enabled=true` directly for
|
||||
exactly that reason. This migration brings Gemini to the same state via an
|
||||
in-place `UPDATE` (the row already exists — no enum touched, no INSERT).
|
||||
|
||||
Revision ID: 086_enable_gemini_provider
|
||||
Revises: 085_seed_gemini_provider
|
||||
Create Date: 2026-07-23
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision = "086_enable_gemini_provider"
|
||||
down_revision = "085_seed_gemini_provider"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
sa.text(
|
||||
"UPDATE provider_configs SET enabled = true WHERE name = 'Gemini (Google)'"
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Honest revert — back to the state migration 085 left it in, not a no-op.
|
||||
op.execute(
|
||||
sa.text(
|
||||
"UPDATE provider_configs SET enabled = false WHERE name = 'Gemini (Google)'"
|
||||
)
|
||||
)
|
||||
@@ -447,6 +447,14 @@ services:
|
||||
# omitted — inert while guard stays off by default, but reaches the
|
||||
# container the moment an operator arms guard by hand-editing this file.
|
||||
ROBOCO_GUARD_EMERGENCY_WHITELIST: ${ROBOCO_GUARD_EMERGENCY_WHITELIST:-}
|
||||
# Trusted local-proxy hop IPs (docker bridge gateway) for the XFF
|
||||
# real-client resolver — carried here (inert until guard is armed) so a
|
||||
# gateway-fronted Tailscale Serve deploy can set it via .env rather than
|
||||
# hand-editing this file. Same reach-the-container rule as above.
|
||||
ROBOCO_GUARD_TRUSTED_CHAIN_PEERS: ${ROBOCO_GUARD_TRUSTED_CHAIN_PEERS:-}
|
||||
# Per-task/project cost budgets (default-off; conservative registry
|
||||
# posture, unlike the build compose which arms it).
|
||||
ROBOCO_TASK_BUDGETS_ENABLED: ${ROBOCO_TASK_BUDGETS_ENABLED:-false}
|
||||
# Cloud auth (FastAPI Users): login-gates the panel/API when exposed
|
||||
# beyond localhost. OFF by default (matches config default, unlike the
|
||||
# build compose which arms it for the personal deploy). Set
|
||||
|
||||
@@ -692,6 +692,15 @@ services:
|
||||
# setting it in .env silently does nothing — only vars listed in this
|
||||
# stanza reach the container.
|
||||
ROBOCO_GUARD_EMERGENCY_WHITELIST: ${ROBOCO_GUARD_EMERGENCY_WHITELIST:-}
|
||||
# Trusted local-proxy hop IPs (docker bridge gateway) for the XFF
|
||||
# real-client resolver — set to the bridge gateway when Tailscale Serve
|
||||
# is gateway-fronted, else the guard sees a whitelisted hop and the WAF
|
||||
# goes inert for /tg. Same "must be listed here to reach the container"
|
||||
# rule as the emergency whitelist above.
|
||||
ROBOCO_GUARD_TRUSTED_CHAIN_PEERS: ${ROBOCO_GUARD_TRUSTED_CHAIN_PEERS:-}
|
||||
# Per-task/project cost budgets (default-off subsystem; also on the
|
||||
# panel feature-flags card).
|
||||
ROBOCO_TASK_BUDGETS_ENABLED: ${ROBOCO_TASK_BUDGETS_ENABLED:-true}
|
||||
volumes:
|
||||
# Docker socket - allows spawning agent containers
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
|
||||
@@ -692,6 +692,15 @@ services:
|
||||
# setting it in .env silently does nothing — only vars listed in this
|
||||
# stanza reach the container.
|
||||
ROBOCO_GUARD_EMERGENCY_WHITELIST: ${ROBOCO_GUARD_EMERGENCY_WHITELIST:-}
|
||||
# Trusted local-proxy hop IPs (docker bridge gateway) for the XFF
|
||||
# real-client resolver — set to the bridge gateway when Tailscale Serve
|
||||
# is gateway-fronted, else the guard sees a whitelisted hop and the WAF
|
||||
# goes inert for /tg. Same "must be listed here to reach the container"
|
||||
# rule as the emergency whitelist above.
|
||||
ROBOCO_GUARD_TRUSTED_CHAIN_PEERS: ${ROBOCO_GUARD_TRUSTED_CHAIN_PEERS:-}
|
||||
# Per-task/project cost budgets (default-off subsystem; also on the
|
||||
# panel feature-flags card).
|
||||
ROBOCO_TASK_BUDGETS_ENABLED: ${ROBOCO_TASK_BUDGETS_ENABLED:-true}
|
||||
volumes:
|
||||
# Docker socket - allows spawning agent containers
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
## Purpose
|
||||
This slice is the agent-runtime + LLM-provider seam plus the in-container agent SDK. The provider layer (roboco/llm/providers/) abstracts how agents are spawned/stopped/health-checked/removed across LLM backends (Claude Code default, Grok CLI) behind an AgentProvider ABC + ProviderRegistry, with a Grok auth-token refresh loop keeping the SuperGrok credential live. The agent SDK (roboco/agent_sdk/) is the FastAPI sidecar running inside every agent container handling A2A messaging, tool-budget/loop/verb-circuit breakers, token-usage capture, and the interactive intake/secretary chat drivers (Claude SDK + Grok CLI). The runtime helpers (spawn_manifest, streaming, transcript_retention) build the per-role tool manifest, wire reasoning-stream callbacks, and select old agent transcripts to prune.
|
||||
This slice is the agent-runtime + LLM-provider seam plus the in-container agent SDK. The provider layer (roboco/llm/providers/) abstracts how agents are spawned/stopped/health-checked/removed across LLM backends (Claude Code default, Grok CLI, Codex CLI, Gemini CLI) behind an AgentProvider ABC + ProviderRegistry, with a Grok auth-token refresh loop keeping the SuperGrok credential live and an orchestrator-side Codex refresh loop keeping the ChatGPT-subscription credential live (Gemini needs neither — its OAuth refresh token is reusable, so each container refreshes its own local copy in-process). The agent SDK (roboco/agent_sdk/) is the FastAPI sidecar running inside every agent container handling A2A messaging, tool-budget/loop/verb-circuit breakers, token-usage capture, and the interactive intake/secretary chat drivers (Claude SDK + Grok CLI — Codex and Gemini are one-shot delivery roles only, no interactive intake/secretary support). The runtime helpers (spawn_manifest, streaming, transcript_retention) build the per-role tool manifest, wire reasoning-stream callbacks, and select old agent transcripts to prune.
|
||||
|
||||
## Files
|
||||
|
||||
@@ -23,6 +23,14 @@ This slice is the agent-runtime + LLM-provider seam plus the in-container agent
|
||||
| roboco/llm/providers/grok_auth.py | SuperGrok token refresh-token grant loop + --check backstop CLI; atomic auth.json rewrite | 317 |
|
||||
| roboco/llm/providers/grok_cli_config.py | Entrypoint renderer: mcp-config -> ~/.grok/config.toml, per-role grok flags, AGENTS.md, bash-guard hook, + default-off fable-mode honesty-nudge hook | 317 |
|
||||
| roboco/llm/providers/grok_cli_usage.py | Capture token usage from grok sessions/updates.jsonl -> usage.json (notional cost) | 201 |
|
||||
| roboco/llm/providers/codex.py | CodexCliProvider: spawns roboco-agent-codex container, mounts ~/.codex dir (RO) + usage dir + codex env | 239 |
|
||||
| roboco/llm/providers/codex_auth.py | ChatGPT-subscription refresh-token grant loop + --check backstop CLI; JWT-exp decode (only expiry signal), atomic auth.json rewrite | 298 |
|
||||
| roboco/llm/providers/codex_cli_config.py | Entrypoint renderer: mcp-config -> ~/.codex/config.toml, Starlark execpolicy deny rules, per-role --sandbox level, combined system+task prompt (no verified system-prompt-file mechanism) | 277 |
|
||||
| roboco/llm/providers/codex_cli_usage.py | Capture token usage from codex exec --json turn.completed events -> usage.json (real input/output/cache-read/cache-write split, priced per-bucket) | 184 |
|
||||
| roboco/llm/providers/codex_cli_sniff.py | Classify a codex run's terminal state (rate_limit/auth/none) from ONLY structured error.message JSONL fields + stderr, never the model's own transcript | 124 |
|
||||
| roboco/llm/providers/gemini.py | GeminiCliProvider: spawns roboco-agent-gemini container, copies host ~/.gemini OAuth creds into a container-local writable copy + usage dir + gemini env | 264 |
|
||||
| roboco/llm/providers/gemini_cli_config.py | Entrypoint renderer: mcp-config -> ~/.gemini/settings.json + per-role TOML Policy Engine deny rules (no native tool-removal flag), GEMINI.md blueprint | 293 |
|
||||
| roboco/llm/providers/gemini_cli_usage.py | Capture token usage from gemini --output-format stream-json terminal result event -> usage.json (per-GA-model pricing); remaps quota/rate-limit errors to exit 75 | 290 |
|
||||
| roboco/agent_sdk/__init__.py | Package docstring only | 10 |
|
||||
| roboco/agent_sdk/models.py | Pydantic models: A2A messages, budget/terminal/verb-circuit/token-usage request+status | 258 |
|
||||
| roboco/agent_sdk/prompt_guard.py | Prompt-injection detector (5 patterns) + CLI for grok entrypoint turn scan | 93 |
|
||||
|
||||
@@ -10,6 +10,8 @@ RoboCo's HTTP request layer is protected by `fastapi-guard` (v7.2.1), implemente
|
||||
|----------|---------|--------|
|
||||
| `ROBOCO_GUARD_ENABLED` | `false` | Master switch. Off = completely inert — no middleware is mounted, the request path is entirely unchanged, and nothing is logged or blocked. |
|
||||
| `ROBOCO_GUARD_PASSIVE_MODE` | see below | When the guard is enabled, controls whether it blocks matching requests or only logs them. |
|
||||
| `ROBOCO_GUARD_EMERGENCY_WHITELIST` | `` (empty) | Comma-separated IPs/CIDRs always allowed through in an active `ROBOCO_GUARD_EMERGENCY` lockdown, in addition to loopback. Empty = loopback only. |
|
||||
| `ROBOCO_GUARD_TRUSTED_CHAIN_PEERS` | `` (empty) | Comma-separated exact IP address(es) — never a CIDR range — trusted to appear as a recorded proxy hop inside `X-Forwarded-For` beyond loopback, e.g. the docker bridge gateway a host-proxied Tailscale Serve chain terminates behind, so the resolved client is the real tailnet/LAN peer instead of that hop's own address. Empty = only a loopback rightmost hop ever peels. |
|
||||
|
||||
As of 2026-07-19 the guard is gated off by default in config, but the NAS build compose arms it ON in ACTIVE enforcement (`ROBOCO_GUARD_PASSIVE_MODE=false`) — passive/log-only calibration came back clean, and the CEO approved the flip now that cloud auth + Tailscale are armed. A matching request on that deploy is actually blocked, not just logged. The registry compose still ships it fully off (see Enforcement Posture below).
|
||||
|
||||
|
||||
@@ -25,3 +25,7 @@ notify_ack(notification_id) # acknowledge after handling
|
||||
```
|
||||
|
||||
When `i_am_idle()` reports unread A2A or @mentions, clear A2A with `read_a2a()` (see `a2a-tools.md`) and clear notifications with list -> get -> ack, then idle again. (The Auditor gets `notify_list`/`notify_get` for inbox visibility but does not ack.)
|
||||
|
||||
## Unacked notifications re-escalate
|
||||
|
||||
An ack-required `notify` left unacked past its `expires_at` is re-escalated to the recipient's up-role (your PM's PM, or the CEO) — but not on every sweep tick. The first re-escalation fires at expiry, each one after that doubles the wait (1h, 2h, 4h, ... capped at 24h), and after a fixed number of attempts it stops and is logged as permanently unacked. Acking promptly is the only way to stop the clock — there is no way to snooze or dismiss a notification other than `notify_ack`.
|
||||
|
||||
@@ -50,6 +50,7 @@ If your task has `dependency_ids` in the same repo, the fresh branch cut also ba
|
||||
- **Self-documentation prevention**: Documenter cannot claim tasks they developed
|
||||
- **Branch requirement**: Branch auto-created on `i_will_work_on`
|
||||
- **Sequence order (strict, assignee-blind)**: if a task has a parent and a `sequence` number, it cannot be claimed while any sibling with a strictly lower sequence is still non-terminal — regardless of who owns which task. Siblings on the SAME sequence run in parallel (independent work ties at 0, or at the wave a delegating PM stamped from the collision graph). This is independent of, and stricter than, `dependency_ids`: a claim attempt on a sequence-held task fails even with no unmet dependency. The error names the blocking sibling by title — `unclaim`/wait is the only remedy, there is no override verb. The dispatcher pre-filters sequence-held (and dependency-held) tasks before attempting a claim, so you should rarely see this in practice — but a claim you make directly (rather than via `give_me_work`) can still hit it.
|
||||
- **Project budget cap** (when task budgets are armed): `i_will_work_on` / `i_will_plan` are refused once the project's `monthly_budget_usd` has been reached this calendar month — a WORK-STARTING claim only, so a QA/doc/PR-review/PM-merge claim on already-in-flight work is never blocked by this. There is no override; wait for the next month or ask the CEO to raise the cap.
|
||||
|
||||
## Releasing a Claimed Task
|
||||
|
||||
|
||||
@@ -489,4 +489,34 @@ describe("EditProjectDialog — Monthly Budget (USD)", () => {
|
||||
};
|
||||
expect(call.updates.monthly_budget_usd).toBe(100);
|
||||
});
|
||||
|
||||
it("shows this month's spend against the cap when monthly_spend_usd is present", async () => {
|
||||
renderDialog(
|
||||
makeProject({ monthly_budget_usd: 100, monthly_spend_usd: 42.5 }),
|
||||
);
|
||||
await screen.findByRole("button", { name: /Save Changes/i });
|
||||
openAutonomySection();
|
||||
|
||||
expect(screen.getByTestId("project-spend").textContent).toBe(
|
||||
"Spent: $42.50 this month / $100.00",
|
||||
);
|
||||
});
|
||||
|
||||
it("hides the ratio (but still shows spend) when there is no monthly cap", async () => {
|
||||
renderDialog(makeProject({ monthly_budget_usd: null, monthly_spend_usd: 10 }));
|
||||
await screen.findByRole("button", { name: /Save Changes/i });
|
||||
openAutonomySection();
|
||||
|
||||
expect(screen.getByTestId("project-spend").textContent).toBe(
|
||||
"Spent: $10.00 this month",
|
||||
);
|
||||
});
|
||||
|
||||
it("hides the spend line entirely when monthly_spend_usd is absent (flag off)", async () => {
|
||||
renderDialog(makeProject({ monthly_budget_usd: 100, monthly_spend_usd: null }));
|
||||
await screen.findByRole("button", { name: /Save Changes/i });
|
||||
openAutonomySection();
|
||||
|
||||
expect(screen.queryByTestId("project-spend")).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -833,6 +833,15 @@ function EditProjectForm({
|
||||
Must be greater than 0 — a 0 budget would block every claim
|
||||
immediately. Leave blank for no cap.
|
||||
</p>
|
||||
{project.monthly_spend_usd != null && (
|
||||
<p className="text-xs text-muted-foreground" data-testid="project-spend">
|
||||
Spent: ${project.monthly_spend_usd.toFixed(2)} this month
|
||||
{monthlyBudgetUsd.trim() &&
|
||||
!Number.isNaN(Number(monthlyBudgetUsd))
|
||||
? ` / $${Number(monthlyBudgetUsd).toFixed(2)}`
|
||||
: ""}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="flex items-center justify-between">
|
||||
|
||||
@@ -50,6 +50,11 @@ const {
|
||||
provider_type: "openai",
|
||||
display_name: "GPT-5.3 Codex",
|
||||
},
|
||||
{
|
||||
model_name: "gemini-2.5-pro",
|
||||
provider_type: "gemini",
|
||||
display_name: "Gemini 2.5 Pro",
|
||||
},
|
||||
]),
|
||||
getOllamaKey: vi.fn(async () => ({ has_key: false, enabled: true })),
|
||||
setOllamaKey: vi.fn(async () => ({ has_key: true, enabled: true })),
|
||||
@@ -387,6 +392,13 @@ function withQueryClient(ui: ReactNode) {
|
||||
return <QueryClientProvider client={client}>{ui}</QueryClientProvider>;
|
||||
}
|
||||
|
||||
// Finds a per-agent Mix row's container div by its agent-id text. `.closest`
|
||||
// on a non-tag-name CSS selector types as `Element | null`, not `HTMLElement`
|
||||
// — cast once here rather than at every call site.
|
||||
function mixRowFor(agentId: string): HTMLElement {
|
||||
return screen.getByText(agentId).closest("div.grid") as HTMLElement;
|
||||
}
|
||||
|
||||
describe("AIRoutingCard", () => {
|
||||
beforeEach(() => {
|
||||
catalog.mockClear();
|
||||
@@ -761,6 +773,102 @@ describe("AIRoutingCard", () => {
|
||||
});
|
||||
});
|
||||
|
||||
// -------------------------------------------------------------------------
|
||||
// Codex and Gemini mode buttons + Mix picker visibility (the headline gap:
|
||||
// both were built but unreachable from the panel — no apply-mode card, no
|
||||
// Mix group).
|
||||
// -------------------------------------------------------------------------
|
||||
|
||||
describe("Codex and Gemini mode buttons", () => {
|
||||
it("renders the Codex button and applies mode='codex' on confirm", async () => {
|
||||
const confirmSpy = vi.spyOn(window, "confirm").mockReturnValue(true);
|
||||
render(withQueryClient(<AIRoutingCard />));
|
||||
await screen.findByText("Grok (xAI) API key");
|
||||
|
||||
fireEvent.click(screen.getByText("Codex"));
|
||||
|
||||
await waitFor(() =>
|
||||
expect(applyMode).toHaveBeenCalledWith({ mode: "codex" }),
|
||||
);
|
||||
confirmSpy.mockRestore();
|
||||
});
|
||||
|
||||
it("renders the Gemini button and applies mode='gemini' on confirm", async () => {
|
||||
const confirmSpy = vi.spyOn(window, "confirm").mockReturnValue(true);
|
||||
render(withQueryClient(<AIRoutingCard />));
|
||||
await screen.findByText("Grok (xAI) API key");
|
||||
|
||||
fireEvent.click(screen.getByText("Gemini"));
|
||||
|
||||
await waitFor(() =>
|
||||
expect(applyMode).toHaveBeenCalledWith({ mode: "gemini" }),
|
||||
);
|
||||
confirmSpy.mockRestore();
|
||||
});
|
||||
|
||||
it("neither button is gated on a key (no key card exists for either provider)", async () => {
|
||||
render(withQueryClient(<AIRoutingCard />));
|
||||
await screen.findByText("Grok (xAI) API key");
|
||||
|
||||
expect(screen.getByText("Codex").closest("button")).not.toBeDisabled();
|
||||
expect(screen.getByText("Gemini").closest("button")).not.toBeDisabled();
|
||||
});
|
||||
});
|
||||
|
||||
describe("Mix picker Codex/Gemini group visibility", () => {
|
||||
it("shows Codex and Gemini provider groups for a delivery role's per-agent select", async () => {
|
||||
render(withQueryClient(<AIRoutingCard />));
|
||||
await screen.findByText("Per-agent override (mix mode)");
|
||||
|
||||
const beDevRow = mixRowFor("be-dev-1");
|
||||
// The catalog query resolves asynchronously — the per-agent groups are
|
||||
// absent on the first render pass, so wait for them (findByText) rather
|
||||
// than asserting synchronously.
|
||||
expect(
|
||||
await within(beDevRow).findByText("Codex (OpenAI)"),
|
||||
).toBeInTheDocument();
|
||||
expect(
|
||||
within(beDevRow).getByText("Gemini (Google)"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("excludes Codex and Gemini from the Intake/Secretary/PR Review group, with an inline note", async () => {
|
||||
render(withQueryClient(<AIRoutingCard />));
|
||||
await screen.findByText("Per-agent override (mix mode)");
|
||||
|
||||
expect(
|
||||
screen.getByText(/Codex and Gemini are delivery-roles-only/i),
|
||||
).toBeInTheDocument();
|
||||
|
||||
// Wait for the catalog query to resolve (an unrelated row's groups)
|
||||
// before asserting absence on this group's rows below.
|
||||
await within(mixRowFor("be-dev-1")).findByText("Codex (OpenAI)");
|
||||
|
||||
const secretaryRow = mixRowFor("secretary-1");
|
||||
expect(
|
||||
within(secretaryRow).queryByText("Codex (OpenAI)"),
|
||||
).not.toBeInTheDocument();
|
||||
expect(
|
||||
within(secretaryRow).queryByText("Gemini (Google)"),
|
||||
).not.toBeInTheDocument();
|
||||
|
||||
const intakeRow = mixRowFor("intake-1");
|
||||
expect(
|
||||
within(intakeRow).queryByText("Codex (OpenAI)"),
|
||||
).not.toBeInTheDocument();
|
||||
expect(
|
||||
within(intakeRow).queryByText("Gemini (Google)"),
|
||||
).not.toBeInTheDocument();
|
||||
|
||||
// The root PR reviewer shares the same group/note, even though it is
|
||||
// technically one-shot-capable — the panel restricts the whole group.
|
||||
const prReviewerRow = mixRowFor("pr-reviewer-1");
|
||||
expect(
|
||||
within(prReviewerRow).queryByText("Codex (OpenAI)"),
|
||||
).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
// -------------------------------------------------------------------------
|
||||
// Mode switches preserve complexity overrides (2026-07-17-style incident:
|
||||
// these same buttons once wiped AGENT_SLUG pins) — the confirm text says so
|
||||
|
||||
@@ -40,8 +40,10 @@ import {
|
||||
import { Separator } from "@/components/ui/separator";
|
||||
import {
|
||||
AlertTriangle,
|
||||
Bot,
|
||||
Cpu,
|
||||
Gauge,
|
||||
Gem,
|
||||
Key,
|
||||
KeyRound,
|
||||
Server,
|
||||
@@ -115,6 +117,12 @@ const AGENT_GROUP_DEFS: {
|
||||
},
|
||||
];
|
||||
|
||||
// Codex/Gemini are V1 delivery-roles-only — no interactive Intake/Secretary
|
||||
// support (see roboco.llm.providers.codex / .gemini). This group's per-agent
|
||||
// picker excludes both providers below instead of offering a route that
|
||||
// would silently misroute the persistent Intake/Secretary session at spawn.
|
||||
const INTERACTIVE_ONLY_GROUP_TITLE = "Intake / Secretary / PR Review";
|
||||
|
||||
// Stable within-group ordering (PM/lead first, devs, QA, doc, reviewer last)
|
||||
// so the picker doesn't churn alphabetically as the live roster loads —
|
||||
// ties (e.g. dev-1/dev-2) break on slug, which already sorts correctly.
|
||||
@@ -280,6 +288,10 @@ export function AIRoutingCard() {
|
||||
(c: { provider_type: ModelProvider }) =>
|
||||
c.provider_type === ModelProvider.OPENAI,
|
||||
);
|
||||
const catalogGeminiOnly = catalog.filter(
|
||||
(c: { provider_type: ModelProvider }) =>
|
||||
c.provider_type === ModelProvider.GEMINI,
|
||||
);
|
||||
const catalogAnthropicOnly = catalog.filter(
|
||||
(c: { provider_type: ModelProvider }) =>
|
||||
c.provider_type === ModelProvider.ANTHROPIC,
|
||||
@@ -326,6 +338,46 @@ export function AIRoutingCard() {
|
||||
}
|
||||
};
|
||||
|
||||
const flipToCodex = async () => {
|
||||
if (
|
||||
!confirm(
|
||||
"Switch every agent to Codex? Per-agent pins and complexity " +
|
||||
"overrides are kept; other role/global assignments are replaced. " +
|
||||
"Intake and Secretary stay on Anthropic (Codex has no interactive " +
|
||||
"chat support).",
|
||||
)
|
||||
)
|
||||
return;
|
||||
try {
|
||||
await applyMode.mutateAsync({ mode: "codex" });
|
||||
toast.success(
|
||||
"Role/global routing now on Codex — pins/overrides kept, Intake & Secretary stay on Anthropic",
|
||||
);
|
||||
} catch (e) {
|
||||
toast.error("Switch failed: " + errMsg(e));
|
||||
}
|
||||
};
|
||||
|
||||
const flipToGemini = async () => {
|
||||
if (
|
||||
!confirm(
|
||||
"Switch every agent to Gemini? Per-agent pins and complexity " +
|
||||
"overrides are kept; other role/global assignments are replaced. " +
|
||||
"Intake and Secretary stay on Anthropic (Gemini has no interactive " +
|
||||
"chat support).",
|
||||
)
|
||||
)
|
||||
return;
|
||||
try {
|
||||
await applyMode.mutateAsync({ mode: "gemini" });
|
||||
toast.success(
|
||||
"Role/global routing now on Gemini — pins/overrides kept, Intake & Secretary stay on Anthropic",
|
||||
);
|
||||
} catch (e) {
|
||||
toast.error("Switch failed: " + errMsg(e));
|
||||
}
|
||||
};
|
||||
|
||||
const flipToOllama = async () => {
|
||||
if (!hasOllamaKey) {
|
||||
toast.error("Save an Ollama API key first");
|
||||
@@ -551,6 +603,126 @@ export function AIRoutingCard() {
|
||||
}
|
||||
};
|
||||
|
||||
// The full per-agent model-picker option list, shared by every group's
|
||||
// Select — factored out so the Codex/Gemini exclusion for the interactive
|
||||
// group (`restrictInteractiveOnly`) doesn't require duplicating the whole
|
||||
// catalog-grouped SelectContent tree.
|
||||
const renderMixSelectOptions = (restrictInteractiveOnly: boolean) => (
|
||||
<>
|
||||
<SelectItem value="__clear__">(inherit global)</SelectItem>
|
||||
|
||||
{/* Anthropic models */}
|
||||
{catalogAnthropicOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="anthropic" />
|
||||
Anthropic
|
||||
</SelectLabel>
|
||||
{catalogAnthropicOnly.map(
|
||||
(c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Grok (xAI) models */}
|
||||
{catalogGrokOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="grok" />
|
||||
Grok (xAI)
|
||||
</SelectLabel>
|
||||
{catalogGrokOnly.map(
|
||||
(c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Codex (OpenAI) models — excluded for the interactive-only group */}
|
||||
{!restrictInteractiveOnly && catalogOpenaiOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="openai" />
|
||||
Codex (OpenAI)
|
||||
</SelectLabel>
|
||||
{catalogOpenaiOnly.map(
|
||||
(c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Gemini (Google) models — excluded for the interactive-only group */}
|
||||
{!restrictInteractiveOnly && catalogGeminiOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="gemini" />
|
||||
Gemini (Google)
|
||||
</SelectLabel>
|
||||
{catalogGeminiOnly.map(
|
||||
(c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Ollama Cloud models */}
|
||||
{catalogOllamaOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="ollama" />
|
||||
Ollama Cloud
|
||||
</SelectLabel>
|
||||
{catalogOllamaOnly.map(
|
||||
(c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Self-Hosted models */}
|
||||
{selfHostedModels.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="self-hosted" />
|
||||
Self-Hosted
|
||||
</SelectLabel>
|
||||
{selfHostedModels.map((m: SelfHostedModel) => (
|
||||
<SelectItem key={m.model_name} value={m.model_name}>
|
||||
{m.display_name}
|
||||
</SelectItem>
|
||||
))}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Fallback: un-grouped catalog when no grouping is possible */}
|
||||
{catalogAnthropicOnly.length === 0 &&
|
||||
catalogOllamaOnly.length === 0 &&
|
||||
selfHostedModels.length === 0 &&
|
||||
catalogForMix.map((c: { model_name: string; display_name: string }) => (
|
||||
<SelectItem key={c.model_name} value={c.model_name}>
|
||||
{c.display_name} — {c.model_name}
|
||||
</SelectItem>
|
||||
))}
|
||||
</>
|
||||
);
|
||||
|
||||
return (
|
||||
<Card>
|
||||
<CardHeader>
|
||||
@@ -560,8 +732,10 @@ export function AIRoutingCard() {
|
||||
<CardDescription>
|
||||
Decide which model backs each agent. Anthropic uses the mounted
|
||||
<code className="px-1"> ~/.claude </code> auth; Grok (xAI) and Ollama
|
||||
Cloud use the API keys you save below; Self-Hosted connects to any
|
||||
OpenAI-compatible endpoint you run locally.
|
||||
Cloud use the API keys you save below; Codex and Gemini authenticate
|
||||
via their own mounted CLI subscriptions (no key needed) — V1:
|
||||
delivery roles only, not Intake/Secretary; Self-Hosted connects to
|
||||
any OpenAI-compatible endpoint you run locally.
|
||||
</CardDescription>
|
||||
</CardHeader>
|
||||
<CardContent className="space-y-6">
|
||||
@@ -700,10 +874,10 @@ export function AIRoutingCard() {
|
||||
|
||||
{/* -------- Mode toggle -------- */}
|
||||
<section className="space-y-3">
|
||||
<HelpTip label="Anthropic / Grok / Ollama / Self-Hosted replace role/global routing with that provider; per-agent pins in the table below survive the switch. Mix keeps whatever's picked in the table.">
|
||||
<HelpTip label="Anthropic / Grok / Codex / Gemini / Ollama / Self-Hosted replace role/global routing with that provider; per-agent pins in the table below survive the switch. Mix keeps whatever's picked in the table.">
|
||||
<Label className="text-sm font-medium">Routing mode</Label>
|
||||
</HelpTip>
|
||||
<div className="grid grid-cols-2 md:grid-cols-3 lg:grid-cols-6 gap-2">
|
||||
<div className="grid grid-cols-2 md:grid-cols-4 lg:grid-cols-8 gap-2">
|
||||
<ModeButton
|
||||
icon={<ShieldCheck className="h-4 w-4" />}
|
||||
label="Anthropic"
|
||||
@@ -724,6 +898,24 @@ export function AIRoutingCard() {
|
||||
onClick={flipToGrok}
|
||||
disabled={applyMode.isPending || !hasGrokKey}
|
||||
/>
|
||||
<ModeButton
|
||||
icon={<Bot className="h-4 w-4" />}
|
||||
label="Codex"
|
||||
description="Every agent uses Codex (gpt-5.3-codex)."
|
||||
active={currentMode === "codex"}
|
||||
onClick={flipToCodex}
|
||||
disabled={applyMode.isPending}
|
||||
labelHint="Codex authenticates via a mounted ~/.codex subscription (ChatGPT, no API key) — always available once the CLI is logged in on the host. V1: delivery roles only, not offered for Intake/Secretary."
|
||||
/>
|
||||
<ModeButton
|
||||
icon={<Gem className="h-4 w-4" />}
|
||||
label="Gemini"
|
||||
description="Every agent uses Gemini (gemini-2.5-pro)."
|
||||
active={currentMode === "gemini"}
|
||||
onClick={flipToGemini}
|
||||
disabled={applyMode.isPending}
|
||||
labelHint="Gemini authenticates via a mounted ~/.gemini OAuth login (no API key) — always available once the CLI is logged in on the host. V1: delivery roles only, not offered for Intake/Secretary."
|
||||
/>
|
||||
<ModeButton
|
||||
icon={<Sparkles className="h-4 w-4" />}
|
||||
label="Ollama"
|
||||
@@ -784,6 +976,22 @@ export function AIRoutingCard() {
|
||||
per-agent cost cap all apply.
|
||||
</p>
|
||||
) : null}
|
||||
{currentMode === "codex" || currentMode === "mix" ? (
|
||||
<p className="text-xs text-muted-foreground">
|
||||
Codex agents run on OpenAI's official Codex CLI (ChatGPT
|
||||
subscription, mounted ~/.codex); the same command /
|
||||
secret-exfiltration guard, prompt-injection guard, and per-agent
|
||||
cost cap apply. V1: delivery roles only — not available for
|
||||
Intake/Secretary.
|
||||
</p>
|
||||
) : null}
|
||||
{currentMode === "gemini" || currentMode === "mix" ? (
|
||||
<p className="text-xs text-muted-foreground">
|
||||
Gemini agents run on Google's official gemini CLI (OAuth
|
||||
login, mounted ~/.gemini); the same guards apply. V1: delivery
|
||||
roles only — not available for Intake/Secretary.
|
||||
</p>
|
||||
) : null}
|
||||
</section>
|
||||
|
||||
{/* -------- Self-Hosted model picker (when self_hosted mode active) -------- */}
|
||||
@@ -953,13 +1161,22 @@ export function AIRoutingCard() {
|
||||
</div>
|
||||
) : (
|
||||
<div className="divide-y rounded-md border">
|
||||
{agentGroups.map((group) => (
|
||||
{agentGroups.map((group) => {
|
||||
const restrictInteractiveOnly =
|
||||
group.title === INTERACTIVE_ONLY_GROUP_TITLE;
|
||||
return (
|
||||
<div key={group.title} className="p-4">
|
||||
<HelpTip label={group.titleHint}>
|
||||
<h4 className="mb-2 w-fit text-xs font-semibold text-muted-foreground uppercase tracking-wider">
|
||||
{group.title}
|
||||
</h4>
|
||||
</HelpTip>
|
||||
{restrictInteractiveOnly ? (
|
||||
<p className="mb-2 text-[11px] text-muted-foreground">
|
||||
Codex and Gemini are delivery-roles-only (V1) — not
|
||||
offered here (no interactive Intake/Secretary support).
|
||||
</p>
|
||||
) : null}
|
||||
<div className="grid grid-cols-1 gap-x-8 gap-y-3 sm:grid-cols-2">
|
||||
{group.agents.map((a) => (
|
||||
<div
|
||||
@@ -987,144 +1204,15 @@ export function AIRoutingCard() {
|
||||
<SelectValue placeholder="(inherit)" />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
<SelectItem value="__clear__">
|
||||
(inherit global)
|
||||
</SelectItem>
|
||||
|
||||
{/* Anthropic models */}
|
||||
{catalogAnthropicOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="anthropic" />
|
||||
Anthropic
|
||||
</SelectLabel>
|
||||
{catalogAnthropicOnly.map(
|
||||
(c: {
|
||||
model_name: string;
|
||||
display_name: string;
|
||||
}) => (
|
||||
<SelectItem
|
||||
key={c.model_name}
|
||||
value={c.model_name}
|
||||
>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Grok (xAI) models */}
|
||||
{catalogGrokOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="grok" />
|
||||
Grok (xAI)
|
||||
</SelectLabel>
|
||||
{catalogGrokOnly.map(
|
||||
(c: {
|
||||
model_name: string;
|
||||
display_name: string;
|
||||
}) => (
|
||||
<SelectItem
|
||||
key={c.model_name}
|
||||
value={c.model_name}
|
||||
>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Codex (OpenAI) models */}
|
||||
{catalogOpenaiOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="openai" />
|
||||
Codex (OpenAI)
|
||||
</SelectLabel>
|
||||
{catalogOpenaiOnly.map(
|
||||
(c: {
|
||||
model_name: string;
|
||||
display_name: string;
|
||||
}) => (
|
||||
<SelectItem
|
||||
key={c.model_name}
|
||||
value={c.model_name}
|
||||
>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Ollama Cloud models */}
|
||||
{catalogOllamaOnly.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="ollama" />
|
||||
Ollama Cloud
|
||||
</SelectLabel>
|
||||
{catalogOllamaOnly.map(
|
||||
(c: {
|
||||
model_name: string;
|
||||
display_name: string;
|
||||
}) => (
|
||||
<SelectItem
|
||||
key={c.model_name}
|
||||
value={c.model_name}
|
||||
>
|
||||
{c.display_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Self-Hosted models */}
|
||||
{selfHostedModels.length > 0 && (
|
||||
<SelectGroup>
|
||||
<SelectLabel>
|
||||
<ProviderBadge variant="self-hosted" />
|
||||
Self-Hosted
|
||||
</SelectLabel>
|
||||
{selfHostedModels.map((m: SelfHostedModel) => (
|
||||
<SelectItem
|
||||
key={m.model_name}
|
||||
value={m.model_name}
|
||||
>
|
||||
{m.display_name}
|
||||
</SelectItem>
|
||||
))}
|
||||
</SelectGroup>
|
||||
)}
|
||||
|
||||
{/* Fallback: un-grouped catalog when no grouping is possible */}
|
||||
{catalogAnthropicOnly.length === 0 &&
|
||||
catalogOllamaOnly.length === 0 &&
|
||||
selfHostedModels.length === 0 &&
|
||||
catalogForMix.map(
|
||||
(c: {
|
||||
model_name: string;
|
||||
display_name: string;
|
||||
}) => (
|
||||
<SelectItem
|
||||
key={c.model_name}
|
||||
value={c.model_name}
|
||||
>
|
||||
{c.display_name} — {c.model_name}
|
||||
</SelectItem>
|
||||
),
|
||||
)}
|
||||
{renderMixSelectOptions(restrictInteractiveOnly)}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
)}
|
||||
{catalogOllamaOnly.length === 0 ? (
|
||||
@@ -1274,7 +1362,13 @@ function errMsg(e: unknown): string {
|
||||
function ProviderBadge({
|
||||
variant,
|
||||
}: {
|
||||
variant: "anthropic" | "grok" | "openai" | "ollama" | "self-hosted";
|
||||
variant:
|
||||
| "anthropic"
|
||||
| "grok"
|
||||
| "openai"
|
||||
| "gemini"
|
||||
| "ollama"
|
||||
| "self-hosted";
|
||||
}) {
|
||||
const styles: Record<string, string> = {
|
||||
anthropic: "bg-blue-500/20 text-blue-700 dark:text-blue-400",
|
||||
@@ -1282,6 +1376,7 @@ function ProviderBadge({
|
||||
"self-hosted": "bg-purple-500/20 text-purple-700 dark:text-purple-400",
|
||||
grok: "bg-teal-500/20 text-teal-700 dark:text-teal-400",
|
||||
openai: "bg-emerald-500/20 text-emerald-700 dark:text-emerald-400",
|
||||
gemini: "bg-sky-500/20 text-sky-700 dark:text-sky-400",
|
||||
};
|
||||
const labels: Record<string, string> = {
|
||||
anthropic: "A",
|
||||
@@ -1289,6 +1384,7 @@ function ProviderBadge({
|
||||
"self-hosted": "S",
|
||||
grok: "G",
|
||||
openai: "C",
|
||||
gemini: "Ge",
|
||||
};
|
||||
return (
|
||||
<span
|
||||
|
||||
@@ -1,12 +1,17 @@
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
|
||||
import { fireEvent, render, screen, waitFor } from "@testing-library/react";
|
||||
|
||||
const { mutateAsync } = vi.hoisted(() => ({
|
||||
const { mutateAsync, spendState } = vi.hoisted(() => ({
|
||||
mutateAsync: vi.fn().mockResolvedValue(undefined),
|
||||
// Mutable per-test stand-in for useTask's query result — mirrors the
|
||||
// real hook's shape ({ data }) so the dialog's spend read-out can be
|
||||
// exercised without a real fetch.
|
||||
spendState: { data: undefined as { spend_usd?: number | null } | undefined },
|
||||
}));
|
||||
|
||||
vi.mock("@/hooks/use-tasks", () => ({
|
||||
useUpdateTask: () => ({ mutateAsync, isPending: false }),
|
||||
useTask: () => spendState,
|
||||
}));
|
||||
|
||||
vi.mock("sonner", () => ({ toast: { success: vi.fn(), error: vi.fn() } }));
|
||||
@@ -82,6 +87,7 @@ function budgetInput(): HTMLInputElement {
|
||||
describe("EditTaskDialog — Budget (USD) input", () => {
|
||||
beforeEach(() => {
|
||||
mutateAsync.mockClear();
|
||||
spendState.data = undefined;
|
||||
});
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
@@ -183,3 +189,63 @@ describe("EditTaskDialog — Budget (USD) input", () => {
|
||||
expect(updates.budget_usd).toBe(2.5);
|
||||
});
|
||||
});
|
||||
|
||||
describe("EditTaskDialog — spend read-out", () => {
|
||||
beforeEach(() => {
|
||||
mutateAsync.mockClear();
|
||||
spendState.data = undefined;
|
||||
});
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("renders spend against the cap once useTask resolves", () => {
|
||||
spendState.data = { spend_usd: 12.34 };
|
||||
render(
|
||||
<EditTaskDialog
|
||||
task={{ ...task, budget_usd: 20 }}
|
||||
open={true}
|
||||
onOpenChange={vi.fn()}
|
||||
/>,
|
||||
);
|
||||
expect(screen.getByTestId("task-spend").textContent).toBe(
|
||||
"Spent: $12.34 / $20.00",
|
||||
);
|
||||
});
|
||||
|
||||
it("hides the ratio (but still shows spend) when there is no budget cap", () => {
|
||||
spendState.data = { spend_usd: 5 };
|
||||
render(
|
||||
<EditTaskDialog
|
||||
task={{ ...task, budget_usd: null }}
|
||||
open={true}
|
||||
onOpenChange={vi.fn()}
|
||||
/>,
|
||||
);
|
||||
expect(screen.getByTestId("task-spend").textContent).toBe("Spent: $5.00");
|
||||
});
|
||||
|
||||
it("renders nothing while the spend fetch hasn't resolved yet", () => {
|
||||
spendState.data = undefined;
|
||||
render(
|
||||
<EditTaskDialog
|
||||
task={{ ...task, budget_usd: 20 }}
|
||||
open={true}
|
||||
onOpenChange={vi.fn()}
|
||||
/>,
|
||||
);
|
||||
expect(screen.queryByTestId("task-spend")).toBeNull();
|
||||
});
|
||||
|
||||
it("renders nothing when the task-budgets flag is off (spend_usd null)", () => {
|
||||
spendState.data = { spend_usd: null };
|
||||
render(
|
||||
<EditTaskDialog
|
||||
task={{ ...task, budget_usd: 20 }}
|
||||
open={true}
|
||||
onOpenChange={vi.fn()}
|
||||
/>,
|
||||
);
|
||||
expect(screen.queryByTestId("task-spend")).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import { useUpdateTask } from "@/hooks/use-tasks";
|
||||
import { useTask, useUpdateTask } from "@/hooks/use-tasks";
|
||||
import { Task, Team, Complexity, TaskNature, TaskType } from "@/types";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Input } from "@/components/ui/input";
|
||||
@@ -126,6 +126,12 @@ function EditTaskDialogInner({
|
||||
const [advancedOpen, setAdvancedOpen] = useState(false);
|
||||
|
||||
const updateTask = useUpdateTask();
|
||||
// Read-only spend, refetched fresh whenever this dialog is mounted (it only
|
||||
// mounts while open — see EditTaskDialog below). null while loading, when
|
||||
// the task-budgets flag is off, or on fetch error — all rendered the same
|
||||
// way: the spend line is simply omitted (never a broken "$undefined").
|
||||
const { data: freshTask } = useTask(task.id);
|
||||
const spendUsd = freshTask?.spend_usd;
|
||||
|
||||
const handleSubmit = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
@@ -369,6 +375,14 @@ function EditTaskDialogInner({
|
||||
before it spends a cent. Leave blank for the task-type
|
||||
default.
|
||||
</p>
|
||||
{spendUsd != null && (
|
||||
<p className="text-xs text-muted-foreground" data-testid="task-spend">
|
||||
Spent: ${spendUsd.toFixed(2)}
|
||||
{budgetUsd.trim() && !Number.isNaN(Number(budgetUsd))
|
||||
? ` / $${Number(budgetUsd).toFixed(2)}`
|
||||
: ""}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Git Configuration Section */}
|
||||
|
||||
@@ -26,15 +26,15 @@ export interface ModelAssignment {
|
||||
model_name: string;
|
||||
}
|
||||
|
||||
// "codex" is READ-only (derive_mode can report it for a pure-OPENAI global
|
||||
// assignment) — there is no apply_mode="codex" write path, so no UI ever
|
||||
// constructs an ApplyModePayload with this value. One shared type (not a
|
||||
// split read/write pair) keeps this file small; nothing calls applyMode with
|
||||
// mode: "codex" since no button exists for it.
|
||||
// One shared read/write type keeps this file small — every value here has
|
||||
// both an apply_mode write path (a ModeButton) and a derive_mode read path
|
||||
// (GET /providers), except "mix"/"cost_tiered" which are additive/table-driven
|
||||
// rather than single mode-button flips.
|
||||
export type RoutingMode =
|
||||
| "anthropic"
|
||||
| "grok"
|
||||
| "codex"
|
||||
| "gemini"
|
||||
| "ollama"
|
||||
| "self_hosted"
|
||||
| "mix"
|
||||
|
||||
@@ -106,6 +106,7 @@ export enum ModelProvider {
|
||||
OPENAI = "openai",
|
||||
LOCAL = "local",
|
||||
GROK = "grok",
|
||||
GEMINI = "gemini",
|
||||
}
|
||||
|
||||
export enum AssignmentScope {
|
||||
@@ -226,6 +227,10 @@ export interface Task {
|
||||
priority: number; // 0=P0(highest), 1=P1, 2=P2, 3=P3(lowest)
|
||||
// Cost cap (ROBOCO_TASK_BUDGETS_ENABLED). null = use the task-type default.
|
||||
budget_usd?: number | null;
|
||||
// This task's own accumulated agent-spawn spend. Only populated by the
|
||||
// single-task detail fetch (GET /tasks/{id}) when the budgets flag is on;
|
||||
// null on list rows and when the flag is off.
|
||||
spend_usd?: number | null;
|
||||
sequence: number; // Order number within siblings
|
||||
team: Team;
|
||||
created_by: string;
|
||||
@@ -1064,6 +1069,10 @@ export interface Project {
|
||||
// Calendar-month cap on summed agent-spawn spend across this project's
|
||||
// tasks; null = no cap. Only enforced when ROBOCO_TASK_BUDGETS_ENABLED is on.
|
||||
monthly_budget_usd: number | null;
|
||||
// This calendar month's summed agent-spawn spend across this project's
|
||||
// tasks (ProjectService.project_month_spend_usd). Only populated when
|
||||
// ROBOCO_TASK_BUDGETS_ENABLED is on; null otherwise.
|
||||
monthly_spend_usd?: number | null;
|
||||
sandbox_services: string[] | null;
|
||||
sandbox_extensions: Record<string, string[]> | null;
|
||||
// Runtime state
|
||||
|
||||
@@ -122,7 +122,21 @@ async def get_project(
|
||||
detail=f"Project not found: {project_id}",
|
||||
)
|
||||
|
||||
return project_to_response(project)
|
||||
response = project_to_response(project)
|
||||
|
||||
# Gated the same as the budgets feature: an extra DB read, so only pay
|
||||
# for it when the panel can actually make use of it (ROBOCO_TASK_BUDGETS_ENABLED).
|
||||
from roboco.config import settings as _settings
|
||||
|
||||
if _settings.task_budgets_enabled:
|
||||
from roboco.services.task import get_task_service
|
||||
|
||||
task_service = get_task_service(db)
|
||||
response.monthly_spend_usd = await task_service.project_month_spend_usd(
|
||||
cast("UUID", project.id)
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
|
||||
# =============================================================================
|
||||
|
||||
@@ -70,6 +70,17 @@ _PROVIDER_REMEDIATION: dict[ModelProvider, str] = {
|
||||
"Configure + test the self-hosted server first (PUT /providers/self-hosted)."
|
||||
),
|
||||
ModelProvider.ANTHROPIC: "The Anthropic provider is disabled — re-enable it first.",
|
||||
ModelProvider.OPENAI: (
|
||||
"Codex authenticates via a mounted ChatGPT-subscription ~/.codex "
|
||||
"directory, not a key — enable it via the Codex mode button, or "
|
||||
"assign a Codex model to an agent in Mix mode (both force-enable "
|
||||
"the row)."
|
||||
),
|
||||
ModelProvider.GEMINI: (
|
||||
"Gemini authenticates via a mounted OAuth ~/.gemini credential, not "
|
||||
"a key — enable it via the Gemini mode button, or assign a Gemini "
|
||||
"model to an agent in Mix mode (both force-enable the row)."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -1119,6 +1119,13 @@ async def get_task(
|
||||
# Enrich with work session and project context
|
||||
response = await enrich_task_with_context(response, db)
|
||||
|
||||
# Gated the same as the budgets feature: an extra DB read, so only pay
|
||||
# for it when the panel can actually make use of it (ROBOCO_TASK_BUDGETS_ENABLED).
|
||||
from roboco.config import settings as _settings
|
||||
|
||||
if _settings.task_budgets_enabled:
|
||||
response.spend_usd = await service.task_spend_usd(task_id)
|
||||
|
||||
return response
|
||||
|
||||
|
||||
|
||||
@@ -58,6 +58,11 @@ class ProjectResponse(BaseModel):
|
||||
dep_update_command: str | None = None
|
||||
dep_update_paths: list[str] | None = None
|
||||
monthly_budget_usd: float | None = None
|
||||
# This calendar month's summed agent-spawn spend across this project's
|
||||
# tasks (TaskService.project_month_spend_usd). Only populated by
|
||||
# GET /projects/{id} (an extra DB read) when ROBOCO_TASK_BUDGETS_ENABLED
|
||||
# is on; null everywhere else (list views, flag-off).
|
||||
monthly_spend_usd: float | None = None
|
||||
sandbox_services: list[str] | None = None
|
||||
sandbox_extensions: dict[str, list[str]] | None = None
|
||||
|
||||
@@ -212,7 +217,8 @@ class ProjectUpdateRequest(BaseModel):
|
||||
video_engine_enabled: bool | None = None
|
||||
dep_update_command: str | None = None
|
||||
dep_update_paths: list[str] | None = None
|
||||
monthly_budget_usd: float | None = None
|
||||
# gt=0 — a 0/negative cap would block every claim immediately (#654).
|
||||
monthly_budget_usd: float | None = Field(default=None, gt=0)
|
||||
sandbox_services: list[str] | None = None
|
||||
sandbox_extensions: dict[str, list[str]] | None = None
|
||||
|
||||
|
||||
@@ -183,6 +183,14 @@ class ApplyModeRequest(BaseModel):
|
||||
ROLE_MODEL_MAP + mounted ~/.claude.
|
||||
- mode="ollama": clear every assignment; set GLOBAL default to
|
||||
`default_model` (if omitted, the service picks a sensible default).
|
||||
- mode="grok": clear every assignment; force-enable the GROK provider;
|
||||
set GLOBAL default to `default_model` (default grok-build-0.1).
|
||||
- mode="codex": clear every assignment; force-enable the OPENAI provider;
|
||||
set GLOBAL default to `default_model` (default gpt-5.3-codex). No key
|
||||
check — subscription-CLI auth (~/.codex), same shape as grok.
|
||||
- mode="gemini": clear every assignment; force-enable the GEMINI provider;
|
||||
set GLOBAL default to `default_model` (default gemini-2.5-pro). No key
|
||||
check — subscription-CLI auth (~/.gemini), same shape as grok.
|
||||
- mode="mix": clear existing per-agent pins; upsert the `per_agent`
|
||||
map verbatim. Role + GLOBAL rows are left untouched so the user can
|
||||
layer with an existing partial setup. Self-hosted model names in
|
||||
@@ -195,22 +203,32 @@ class ApplyModeRequest(BaseModel):
|
||||
routing already exists.
|
||||
"""
|
||||
|
||||
mode: Literal["anthropic", "grok", "ollama", "mix", "self_hosted", "cost_tiered"]
|
||||
mode: Literal[
|
||||
"anthropic",
|
||||
"grok",
|
||||
"codex",
|
||||
"gemini",
|
||||
"ollama",
|
||||
"mix",
|
||||
"self_hosted",
|
||||
"cost_tiered",
|
||||
]
|
||||
default_model: str | None = None
|
||||
per_agent: dict[str, str] | None = None
|
||||
|
||||
|
||||
class ModeResponse(BaseModel):
|
||||
"""Server-side view of the current mode + a snapshot of active rules.
|
||||
|
||||
Read-only ``mode`` values are a superset of what ``ApplyModeRequest``
|
||||
accepts: "codex" (OPENAI) can come back from `derive_mode()` (a pure-Codex
|
||||
global assignment), but there is no `apply_mode="codex"` write path — mix
|
||||
mode's per-agent picker is the only way to route to it.
|
||||
"""
|
||||
"""Server-side view of the current mode + a snapshot of active rules."""
|
||||
|
||||
mode: Literal[
|
||||
"anthropic", "grok", "codex", "ollama", "mix", "self_hosted", "cost_tiered"
|
||||
"anthropic",
|
||||
"grok",
|
||||
"codex",
|
||||
"gemini",
|
||||
"ollama",
|
||||
"mix",
|
||||
"self_hosted",
|
||||
"cost_tiered",
|
||||
]
|
||||
assignments: list[AssignmentResponse]
|
||||
|
||||
@@ -276,7 +294,14 @@ class RoutingPresetApplyResponse(BaseModel):
|
||||
catalog model) — never a partial/silent apply."""
|
||||
|
||||
mode: Literal[
|
||||
"anthropic", "grok", "codex", "ollama", "mix", "self_hosted", "cost_tiered"
|
||||
"anthropic",
|
||||
"grok",
|
||||
"codex",
|
||||
"gemini",
|
||||
"ollama",
|
||||
"mix",
|
||||
"self_hosted",
|
||||
"cost_tiered",
|
||||
]
|
||||
assignments: list[AssignmentResponse]
|
||||
skipped: list[str]
|
||||
|
||||
@@ -213,7 +213,8 @@ class TaskUpdate(BaseModel):
|
||||
# Cost cap (ROBOCO_TASK_BUDGETS_ENABLED). An explicit null clears it back
|
||||
# to "use the TaskType default" — handled at the route layer like the
|
||||
# other _NULLABLE_TASK_FIELDS (TaskService.update() itself skips None).
|
||||
budget_usd: float | None = Field(default=None, ge=0)
|
||||
# gt=0 — a 0/negative cap would block every claim immediately (#654).
|
||||
budget_usd: float | None = Field(default=None, gt=0)
|
||||
target_date: datetime | None = None
|
||||
estimated_complexity: Complexity | None = None
|
||||
|
||||
@@ -312,6 +313,11 @@ class TaskResponse(BaseModel):
|
||||
sequence: int # Order number within siblings
|
||||
# Cost cap (ROBOCO_TASK_BUDGETS_ENABLED). Null = use the TaskType default.
|
||||
budget_usd: float | None = None
|
||||
# This task's own accumulated agent-spawn spend (TaskService.task_spend_usd).
|
||||
# Only populated by GET /tasks/{id} (an extra DB read) when
|
||||
# ROBOCO_TASK_BUDGETS_ENABLED is on; null everywhere else (list views,
|
||||
# flag-off) rather than a stale/misleading $0.
|
||||
spend_usd: float | None = None
|
||||
nature: TaskNature # Technical or non-technical work
|
||||
|
||||
# Task Type & Git Configuration (all tasks follow git workflow)
|
||||
|
||||
@@ -1861,6 +1861,17 @@ class Settings(BaseSettings):
|
||||
"ROBOCO_CODEX_CLI_MODEL"
|
||||
),
|
||||
)
|
||||
# The gemini CLI model id passed via ROBOCO_AGENT_MODEL at spawn. Unlike
|
||||
# the grok path (a raw os.environ read in gemini.py, now fixed to mirror
|
||||
# codex_cli_model above), this is a real Settings field so it shows up in
|
||||
# the settings schema.
|
||||
gemini_cli_model: str = Field(
|
||||
default="gemini-2.5-pro",
|
||||
description=(
|
||||
"Gemini CLI model id passed to the agent at spawn; override via "
|
||||
"ROBOCO_GEMINI_CLI_MODEL"
|
||||
),
|
||||
)
|
||||
# Base retry_after when parking the GEMINI provider on a quota/rate-limit
|
||||
# exit (see roboco.runtime.orchestrator._park_gemini_rate_limited, which
|
||||
# backs this off exponentially on repeated re-parks within one episode —
|
||||
|
||||
@@ -10,6 +10,12 @@ unit-testable, mirroring :mod:`roboco.llm.providers.grok_cli_config`.
|
||||
|
||||
Parity notes (where Codex's runtime model differs from grok's / Claude's):
|
||||
|
||||
* **subagents** — fleet-wide ban (CEO, 2026-07-09): ``config.toml``'s
|
||||
``[agents]`` table (default ``enabled = true``) is rendered with
|
||||
``enabled = false`` unconditionally, the parity analogue of grok's
|
||||
per-role ``--disallowed-tools Agent`` and gemini's
|
||||
``experimental.enableAgents=false`` — a single global switch here too,
|
||||
not a per-role rule, since Codex has no per-role tool-removal flag either.
|
||||
* **tool removal** — the Codex CLI exposes no per-built-in-tool
|
||||
allow/disallow flags (unlike grok's ``--disallowed-tools``). Tool scoping
|
||||
is coarser: a ``--sandbox`` level per role (see :func:`sandbox_level_for_role`)
|
||||
@@ -75,6 +81,14 @@ CODEX_ARGS_PATH = Path(
|
||||
# optimal, docs, playwright) is best-effort.
|
||||
_REQUIRED_MCP_SERVERS = frozenset({"roboco-flow", "roboco-do"})
|
||||
|
||||
# The CLI's default MCP startup timeout (10s) is too tight for a cold uv wheel
|
||||
# cache (first spawn after an image rebuild — see the identical rationale in
|
||||
# roboco.runtime.orchestrator._generate_mcp_config's UV_PROJECT_ENVIRONMENT
|
||||
# comment): a required server not yet ready at 10s fail-fast-aborts the whole
|
||||
# session. Widened per required server so a slow-but-working cold start
|
||||
# doesn't get treated as a dead gateway.
|
||||
_REQUIRED_MCP_STARTUP_TIMEOUT_SEC = 30
|
||||
|
||||
# Only `developer` gets a writable sandbox in Codex V1 — narrower than grok's
|
||||
# per-role `allows_write` (role_config says documenter also writes). Documenter
|
||||
# writes ride the roboco-docs MCP server (a network call, not a local sandboxed
|
||||
@@ -132,13 +146,18 @@ _RAW_PM_PREFIXES: tuple[tuple[str, ...], ...] = (
|
||||
|
||||
|
||||
def render_config_toml(mcp_config: dict[str, Any]) -> str:
|
||||
"""Translate Claude Code ``mcpServers`` into codex's ``[mcp_servers]`` TOML.
|
||||
"""Translate Claude Code ``mcpServers`` into codex's config.toml.
|
||||
|
||||
``{"command": "uv", "args": [...], "env": {...}}`` becomes a
|
||||
``[mcp_servers.<name>]`` table with the same fields, plus ``required =
|
||||
true`` for the gateway pair (``roboco-flow`` / ``roboco-do``) so a
|
||||
gateway-init failure fails the codex session fast. Returns an empty string
|
||||
when there are no servers.
|
||||
true`` + ``startup_timeout_sec = 30`` for the gateway pair (``roboco-flow``
|
||||
/ ``roboco-do``) so a gateway-init failure fails the codex session fast
|
||||
without tripping on a cold uv wheel cache (see
|
||||
``_REQUIRED_MCP_STARTUP_TIMEOUT_SEC``). Always carries a top-level
|
||||
``[agents]`` table disabling Codex's native subagents (fleet-wide ban,
|
||||
CEO 2026-07-09 — parity with grok's ``--disallowed-tools Agent`` and
|
||||
gemini's ``experimental.enableAgents=false``): a global switch, not
|
||||
per-role, so it renders unconditionally even with no MCP servers at all.
|
||||
"""
|
||||
servers: dict[str, dict[str, Any]] = {}
|
||||
for name, spec in (mcp_config.get("mcpServers") or {}).items():
|
||||
@@ -151,8 +170,12 @@ def render_config_toml(mcp_config: dict[str, Any]) -> str:
|
||||
block["env"] = {str(k): str(v) for k, v in env.items()}
|
||||
if str(name) in _REQUIRED_MCP_SERVERS:
|
||||
block["required"] = True
|
||||
block["startup_timeout_sec"] = _REQUIRED_MCP_STARTUP_TIMEOUT_SEC
|
||||
servers[str(name)] = block
|
||||
return tomli_w.dumps({"mcp_servers": servers}) if servers else ""
|
||||
config: dict[str, Any] = {"agents": {"enabled": False}}
|
||||
if servers:
|
||||
config["mcp_servers"] = servers
|
||||
return tomli_w.dumps(config)
|
||||
|
||||
|
||||
def sandbox_level_for_role(role: str) -> str:
|
||||
|
||||
@@ -55,6 +55,7 @@ import os
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Protocol
|
||||
|
||||
from roboco.config import settings
|
||||
from roboco.llm.providers._docker import container_running, stop_container
|
||||
from roboco.llm.providers.base import AgentProvider, ProviderError, SpawnResult
|
||||
|
||||
@@ -70,8 +71,9 @@ _DEFAULT_GEMINI_IMAGE = os.environ.get(
|
||||
)
|
||||
|
||||
# The gemini CLI model id. GA ids: gemini-2.5-pro / gemini-2.5-flash /
|
||||
# gemini-2.5-flash-lite (spike-verified).
|
||||
_GEMINI_CLI_MODEL = os.environ.get("ROBOCO_GEMINI_CLI_MODEL", "gemini-2.5-pro")
|
||||
# gemini-2.5-flash-lite (spike-verified). A real Settings field (parity with
|
||||
# codex_cli_model), not a raw os.environ read.
|
||||
_GEMINI_CLI_MODEL = settings.gemini_cli_model
|
||||
|
||||
# Host directory holding the OAuth credential (``oauth_creds.json``, from a
|
||||
# one-time interactive ``gemini`` login). Mounted into the agent's staging path
|
||||
|
||||
@@ -75,6 +75,10 @@ SYSTEM_PROMPT_PATH = Path(
|
||||
GEMINI_POLICIES_DIR = Path.home() / ".gemini" / "policies"
|
||||
_POLICY_FILE_NAME = "roboco.toml"
|
||||
|
||||
# Hard ceiling on agentic turns (loop guard) — parity with grok's
|
||||
# _DEFAULT_MAX_TURNS. Operator-tunable via ROBOCO_GEMINI_MAX_TURNS (main()).
|
||||
_DEFAULT_MAX_TURNS = 200
|
||||
|
||||
# The auth mode a headless run must declare in settings.json, else the CLI
|
||||
# refuses with exit 41 instead of silently using the mounted OAuth credential
|
||||
# (verified fact). ``oauth-personal`` is the CLI's "Login with Google"
|
||||
@@ -218,15 +222,17 @@ def render_settings_json(mcp_config: dict[str, Any]) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def gemini_cli_args() -> list[str]:
|
||||
def gemini_cli_args(*, max_turns: int = _DEFAULT_MAX_TURNS) -> list[str]:
|
||||
"""The ``gemini -p`` flag tokens (excludes ``-p``/``-m``/``--cwd``).
|
||||
|
||||
Universal across every role — ``--approval-mode yolo`` (headless
|
||||
auto-approval); tool scoping lives entirely in the rendered Policy Engine
|
||||
/ settings.json (see :func:`policy_rules_for_role`), not in a CLI flag,
|
||||
unlike grok's per-role ``grok_cli_args_for_role``.
|
||||
auto-approval) plus ``--max-turns`` (the CLI's own agentic-turn loop
|
||||
guard, dedicated exit code 53 — parity with grok's ``--max-turns``); tool
|
||||
scoping lives entirely in the rendered Policy Engine / settings.json (see
|
||||
:func:`policy_rules_for_role`), not in a CLI flag, unlike grok's per-role
|
||||
``grok_cli_args_for_role``.
|
||||
"""
|
||||
return ["--approval-mode", "yolo"]
|
||||
return ["--approval-mode", "yolo", "--max-turns", str(max_turns)]
|
||||
|
||||
|
||||
def _load_mcp_config(path: str) -> dict[str, Any]:
|
||||
@@ -272,6 +278,12 @@ def main() -> int:
|
||||
agent_id = os.environ.get("ROBOCO_AGENT_ID", "")
|
||||
mcp_path = os.environ.get("ROBOCO_MCP_CONFIG", "/app/mcp-config.json")
|
||||
role = get_agent_role(agent_id) or ""
|
||||
try:
|
||||
max_turns = int(
|
||||
os.environ.get("ROBOCO_GEMINI_MAX_TURNS", str(_DEFAULT_MAX_TURNS))
|
||||
)
|
||||
except ValueError:
|
||||
max_turns = _DEFAULT_MAX_TURNS
|
||||
|
||||
GEMINI_SETTINGS_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
GEMINI_SETTINGS_PATH.write_text(
|
||||
@@ -285,7 +297,9 @@ def main() -> int:
|
||||
write_gemini_memory(source=SYSTEM_PROMPT_PATH, dest=GEMINI_MEMORY_PATH)
|
||||
write_policy_toml(role, policies_dir=GEMINI_POLICIES_DIR)
|
||||
GEMINI_ARGS_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
GEMINI_ARGS_PATH.write_text("\n".join(gemini_cli_args()) + "\n", encoding="utf-8")
|
||||
GEMINI_ARGS_PATH.write_text(
|
||||
"\n".join(gemini_cli_args(max_turns=max_turns)) + "\n", encoding="utf-8"
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -249,11 +249,12 @@ class Project(TimestampMixin):
|
||||
# cap, regardless of the flag — this is purely additive.
|
||||
monthly_budget_usd: float | None = Field(
|
||||
default=None,
|
||||
ge=0,
|
||||
gt=0,
|
||||
description=(
|
||||
"Calendar-month cap on this project's summed agent-spawn spend "
|
||||
"(estimated_cost_usd). Null = no cap. Only enforced at claim time "
|
||||
"when ROBOCO_TASK_BUDGETS_ENABLED is on."
|
||||
"when ROBOCO_TASK_BUDGETS_ENABLED is on. Must be > 0 — a 0/negative "
|
||||
"cap would block every claim immediately."
|
||||
),
|
||||
)
|
||||
|
||||
@@ -331,7 +332,7 @@ class ProjectCreate(RobocoBase):
|
||||
build_command: str | None = None
|
||||
quality_command: str | None = None
|
||||
codegen_command: str | None = None
|
||||
monthly_budget_usd: float | None = None
|
||||
monthly_budget_usd: float | None = Field(default=None, gt=0)
|
||||
|
||||
|
||||
class ProjectUpdate(RobocoBase):
|
||||
@@ -371,7 +372,7 @@ class ProjectUpdate(RobocoBase):
|
||||
video_engine_enabled: bool | None = None
|
||||
dep_update_command: str | None = None
|
||||
dep_update_paths: list[str] | None = None
|
||||
monthly_budget_usd: float | None = None
|
||||
monthly_budget_usd: float | None = Field(default=None, gt=0)
|
||||
sandbox_services: list[str] | None = None
|
||||
sandbox_extensions: dict[str, list[str]] | None = None
|
||||
github_installation_id: int | None = Field(
|
||||
|
||||
@@ -171,9 +171,11 @@ class Task(TimestampMixin):
|
||||
# TASK_TYPE_DEFAULT_BUDGET_USD) when the flag is on; a pure no-op off.
|
||||
budget_usd: float | None = Field(
|
||||
default=None,
|
||||
gt=0,
|
||||
description=(
|
||||
"Cap on this task's own accumulated agent-spawn spend "
|
||||
"(estimated_cost_usd). Null = use the TaskType default."
|
||||
"(estimated_cost_usd). Null = use the TaskType default. Must be "
|
||||
"> 0 — a 0/negative cap would block every claim immediately."
|
||||
),
|
||||
)
|
||||
|
||||
@@ -443,7 +445,7 @@ class TaskUpdate(RobocoBase):
|
||||
description: str | None = None
|
||||
acceptance_criteria: list[str] | None = None
|
||||
priority: int | None = Field(default=None, ge=0, le=3)
|
||||
budget_usd: float | None = Field(default=None, ge=0)
|
||||
budget_usd: float | None = Field(default=None, gt=0)
|
||||
status: TaskStatus | None = None
|
||||
assigned_to: UUID | None = None
|
||||
target_date: datetime | None = None
|
||||
|
||||
@@ -289,6 +289,42 @@ _INTAKE_WORKSPACE_AMBIENT = (
|
||||
# time. Seeded in identity.AGENTS; see roboco/agent_sdk/secretary_main.py.
|
||||
SECRETARY_AGENT_ID = "secretary-1"
|
||||
|
||||
# Codex (OPENAI) and Gemini (GEMINI) are V1 delivery-roles-only (see
|
||||
# roboco.llm.providers.codex / .gemini module docstrings) — neither supports
|
||||
# the persistent interactive Intake/Secretary session (no CLI-flag equivalent
|
||||
# to grok's --disallowed-tools/deny, no interactive-session driver image).
|
||||
# Unlike GROK (which has its own GROK_PROMPTER_IMAGE / GROK_SECRETARY_IMAGE),
|
||||
# routing either of these to Intake/Secretary would fall through to the plain
|
||||
# Claude SDK-driver image with a mismatched provider env instead of refusing —
|
||||
# so both spawn paths reject it explicitly instead of silently misbehaving.
|
||||
# Mirrors roboco.services.llm.INTERACTIVE_UNSUPPORTED_PROVIDERS (kept as a
|
||||
# literal here to avoid a runtime import cycle; parity is pinned by a test).
|
||||
# The resolver exempts interactive agents from GLOBAL/ROLE rows on these
|
||||
# providers (a fleet-wide mode switch keeps the chats on Anthropic); this
|
||||
# guard is the backstop for an EXPLICIT AGENT_SLUG pin, which is refused
|
||||
# loudly rather than silently overridden.
|
||||
_INTERACTIVE_UNSUPPORTED_PROVIDERS: tuple[ModelProvider, ...] = (
|
||||
ModelProvider.OPENAI,
|
||||
ModelProvider.GEMINI,
|
||||
)
|
||||
|
||||
|
||||
def _reject_interactive_unsupported_provider(
|
||||
agent_id: str, provider_type: ModelProvider
|
||||
) -> None:
|
||||
"""Refuse spawning the interactive Intake/Secretary agent on a delivery-
|
||||
roles-only provider. Raise BEFORE any image resolution/container mutation
|
||||
so the guarded wrapper's generic ``except Exception`` surfaces this
|
||||
cleanly on the live relay instead of the spawn silently misrouting."""
|
||||
if provider_type in _INTERACTIVE_UNSUPPORTED_PROVIDERS:
|
||||
raise RuntimeError(
|
||||
f"{provider_type.value} is a delivery-roles-only provider (V1) — "
|
||||
f"it cannot power the interactive {agent_id} session. Route "
|
||||
f"{agent_id} to Anthropic, Grok, Ollama, or Self-Hosted instead "
|
||||
"(Mix mode's per-agent picker)."
|
||||
)
|
||||
|
||||
|
||||
# Role -> Image mapping
|
||||
# Specialized images extend the base with role-specific tools
|
||||
AGENT_IMAGES: dict[str, str] = {
|
||||
@@ -5058,6 +5094,9 @@ class AgentOrchestrator:
|
||||
INTAKE_AGENT_ID, ambient=ambient
|
||||
)
|
||||
route = await self._resolve_agent_route(INTAKE_AGENT_ID)
|
||||
_reject_interactive_unsupported_provider(
|
||||
INTAKE_AGENT_ID, route.provider_type
|
||||
)
|
||||
cli_model = _resolve_agent_cli_model(
|
||||
route.provider_type.value, route.model_name
|
||||
)
|
||||
@@ -5257,6 +5296,9 @@ class AgentOrchestrator:
|
||||
|
||||
prompt_path = self._generate_composed_prompt(SECRETARY_AGENT_ID)
|
||||
route = await self._resolve_agent_route(SECRETARY_AGENT_ID)
|
||||
_reject_interactive_unsupported_provider(
|
||||
SECRETARY_AGENT_ID, route.provider_type
|
||||
)
|
||||
cli_model = _resolve_agent_cli_model(
|
||||
route.provider_type.value, route.model_name
|
||||
)
|
||||
|
||||
+157
-27
@@ -84,6 +84,20 @@ _log = structlog.get_logger(__name__)
|
||||
# rows; nothing else needs editing.
|
||||
_COST_TIERED_SEED: tuple[tuple[str, str, str], ...] = (("developer", "low", "haiku"),)
|
||||
|
||||
# derive_mode()'s single-GLOBAL-assignment lookup — a provider type maps to
|
||||
# its "mode" label 1:1 for every mode `apply_mode` can set via a sole GLOBAL
|
||||
# row. A dict keeps derive_mode's branch count low (a chain of `if` returns
|
||||
# hits ruff's PLR0911 the moment a new provider is added, as GEMINI did).
|
||||
_SINGLE_GLOBAL_MODE_BY_PROVIDER: dict[
|
||||
ModelProvider, Literal["grok", "codex", "gemini", "ollama", "self_hosted"]
|
||||
] = {
|
||||
ModelProvider.GROK: "grok",
|
||||
ModelProvider.OPENAI: "codex",
|
||||
ModelProvider.GEMINI: "gemini",
|
||||
ModelProvider.OLLAMA_CLOUD: "ollama",
|
||||
ModelProvider.LOCAL: "self_hosted",
|
||||
}
|
||||
|
||||
|
||||
async def probe_ollama_tags(base_url: str) -> tuple[list[str], str | None]:
|
||||
"""Fetch the model list from a running Ollama server.
|
||||
@@ -141,6 +155,34 @@ class _ResolvedAssignment:
|
||||
|
||||
provider: ProviderConfigTable
|
||||
model_name: str
|
||||
scope: AssignmentScope
|
||||
|
||||
|
||||
# Interactive agents (Intake chat, Secretary chat) have no V1 support on the
|
||||
# Codex/Gemini providers. A GLOBAL/ROLE assignment pointing them there (e.g.
|
||||
# the one-click Codex/Gemini mode) is treated as not-applicable at resolution
|
||||
# time — they fall back to the legacy Anthropic path, so a fleet-wide mode
|
||||
# switch always yields working chats. An EXPLICIT AGENT_SLUG pin is honored
|
||||
# here and refused loudly by the orchestrator's spawn guard instead — a
|
||||
# deliberate operator choice deserves an error, not a silent override. The
|
||||
# orchestrator imports these as the single source of truth for that guard.
|
||||
INTERACTIVE_AGENT_SLUGS: tuple[str, ...] = ("intake-1", "secretary-1")
|
||||
INTERACTIVE_UNSUPPORTED_PROVIDERS: tuple[ModelProvider, ...] = (
|
||||
ModelProvider.OPENAI,
|
||||
ModelProvider.GEMINI,
|
||||
)
|
||||
|
||||
|
||||
def _interactive_exempt(agent_slug: str, resolved: _ResolvedAssignment) -> bool:
|
||||
"""True iff a GLOBAL/ROLE row lands an interactive agent on a
|
||||
delivery-only provider — the resolver then keeps it on the legacy path.
|
||||
An explicit AGENT_SLUG pin never exempts (kept out of resolve_for_agent
|
||||
for its complexity budget)."""
|
||||
return (
|
||||
agent_slug in INTERACTIVE_AGENT_SLUGS
|
||||
and resolved.provider.type in INTERACTIVE_UNSUPPORTED_PROVIDERS
|
||||
and resolved.scope is not AssignmentScope.AGENT_SLUG
|
||||
)
|
||||
|
||||
|
||||
class ModelRoutingService(BaseService):
|
||||
@@ -167,6 +209,19 @@ class ModelRoutingService(BaseService):
|
||||
"""
|
||||
role = get_agent_role(agent_slug) or ""
|
||||
resolved = await self._resolve_assignment(agent_slug, role, complexity)
|
||||
if resolved is not None and _interactive_exempt(agent_slug, resolved):
|
||||
# A fleet-wide GLOBAL/ROLE row landed an interactive agent on a
|
||||
# delivery-only provider (e.g. the one-click Codex/Gemini mode).
|
||||
# Not applicable to Intake/Secretary — keep their chats working
|
||||
# on the legacy Anthropic path. An explicit AGENT_SLUG pin is
|
||||
# NOT exempted; the orchestrator's spawn guard refuses it loudly.
|
||||
self.log.info(
|
||||
"Interactive agent exempt from delivery-only provider",
|
||||
agent_slug=agent_slug,
|
||||
provider_type=resolved.provider.type.value,
|
||||
scope=resolved.scope.value,
|
||||
)
|
||||
return self._legacy_route(role)
|
||||
if resolved is not None and resolved.provider.enabled:
|
||||
route = await self._route_from_resolved(resolved, agent_slug)
|
||||
if route is not None:
|
||||
@@ -346,9 +401,17 @@ class ModelRoutingService(BaseService):
|
||||
provider = await self._get_seeded_provider(entry.provider_type)
|
||||
provider_type_for_log = entry.provider_type
|
||||
|
||||
# Whenever an assignment resolves to LOCAL, ensure the LOCAL provider
|
||||
# row is enabled so resolve_for_agent() will actually use it.
|
||||
if provider_type_for_log == ModelProvider.LOCAL:
|
||||
# Whenever an assignment resolves to LOCAL/GEMINI/OPENAI, ensure the
|
||||
# provider row is enabled so resolve_for_agent() will actually use it
|
||||
# instead of silently falling back to Anthropic. GROK is deliberately
|
||||
# excluded — its enable state is gated on the xAI key
|
||||
# (set_grok_api_key), unlike LOCAL/Codex/Gemini which have no key to
|
||||
# gate on (self-hosted's own base_url + mounted-subscription auth).
|
||||
if provider_type_for_log in (
|
||||
ModelProvider.LOCAL,
|
||||
ModelProvider.GEMINI,
|
||||
ModelProvider.OPENAI,
|
||||
):
|
||||
provider_svc = ProviderService(self.session)
|
||||
await provider_svc.update_provider(
|
||||
require_uuid(provider.id), ProviderUpdate(enabled=True)
|
||||
@@ -379,20 +442,19 @@ class ModelRoutingService(BaseService):
|
||||
|
||||
async def derive_mode(
|
||||
self,
|
||||
) -> Literal["anthropic", "grok", "codex", "ollama", "mix", "self_hosted"]:
|
||||
) -> Literal[
|
||||
"anthropic", "grok", "codex", "gemini", "ollama", "mix", "self_hosted"
|
||||
]:
|
||||
"""Return the current "mode" label for the Settings UI.
|
||||
|
||||
Decision tree matches what `apply_mode` writes:
|
||||
- no assignments at all → "anthropic"
|
||||
- only a global row, Ollama Cloud → "ollama"
|
||||
- only a global row, LOCAL → "self_hosted"
|
||||
- only a global row, GROK → "grok"
|
||||
- only a global row, OPENAI → "codex"
|
||||
- only a global row, GEMINI → "gemini"
|
||||
- anything else → "mix"
|
||||
|
||||
"codex" (OPENAI) is READ-only here — there is no `apply_mode="codex"`
|
||||
write path (mix mode's per-agent picker is the only way to route to
|
||||
it), so this branch exists purely so a pure-OPENAI global assignment
|
||||
(however it got there) reports its real provider instead of the
|
||||
catch-all "mix".
|
||||
"""
|
||||
assignments = await self.list_assignments()
|
||||
if not assignments:
|
||||
@@ -401,14 +463,9 @@ class ModelRoutingService(BaseService):
|
||||
len(assignments) == 1 and assignments[0].scope == AssignmentScope.GLOBAL
|
||||
)
|
||||
if only_global:
|
||||
if assignments[0].provider.type == ModelProvider.GROK:
|
||||
return "grok"
|
||||
if assignments[0].provider.type == ModelProvider.OPENAI:
|
||||
return "codex"
|
||||
if assignments[0].provider.type == ModelProvider.OLLAMA_CLOUD:
|
||||
return "ollama"
|
||||
if assignments[0].provider.type == ModelProvider.LOCAL:
|
||||
return "self_hosted"
|
||||
mode = _SINGLE_GLOBAL_MODE_BY_PROVIDER.get(assignments[0].provider.type)
|
||||
if mode is not None:
|
||||
return mode
|
||||
return "mix"
|
||||
|
||||
async def set_ollama_api_key(self, api_key: str) -> ProviderConfigTable:
|
||||
@@ -532,6 +589,14 @@ class ModelRoutingService(BaseService):
|
||||
self-hosted model name — not validated against the static catalog).
|
||||
- "grok": wipe role/global assignments, set the GLOBAL default
|
||||
to a Grok (xAI) model (default grok-build-0.1). Requires the xAI key.
|
||||
- "codex": wipe role/global assignments, force-enable the OPENAI
|
||||
provider, set the GLOBAL default to a Codex model (default
|
||||
gpt-5.3-codex). No key check — subscription-CLI auth (~/.codex),
|
||||
same shape as Grok/self_hosted.
|
||||
- "gemini": wipe role/global assignments, force-enable the GEMINI
|
||||
provider, set the GLOBAL default to a Gemini model (default
|
||||
gemini-2.5-pro). No key check — subscription-CLI auth (~/.gemini),
|
||||
same shape as Grok/self_hosted.
|
||||
- "mix": apply per-agent map verbatim. Any agent not in the
|
||||
map falls through to the GLOBAL default — which is whatever it
|
||||
was (preserves prior state). Self-hosted model names (not in the
|
||||
@@ -549,6 +614,10 @@ class ModelRoutingService(BaseService):
|
||||
await self._apply_anthropic()
|
||||
elif mode == "grok":
|
||||
await self._apply_grok(default_model)
|
||||
elif mode == "codex":
|
||||
await self._apply_codex(default_model)
|
||||
elif mode == "gemini":
|
||||
await self._apply_gemini(default_model)
|
||||
elif mode == "ollama":
|
||||
await self._apply_ollama(default_model)
|
||||
elif mode == "self_hosted":
|
||||
@@ -560,8 +629,8 @@ class ModelRoutingService(BaseService):
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Unknown mode '{mode}'."
|
||||
" Use 'anthropic', 'grok', 'ollama', 'self_hosted', 'mix',"
|
||||
" or 'cost_tiered'."
|
||||
" Use 'anthropic', 'grok', 'codex', 'gemini', 'ollama',"
|
||||
" 'self_hosted', 'mix', or 'cost_tiered'."
|
||||
)
|
||||
|
||||
async def _wipe_mode_switch_assignments(self) -> None:
|
||||
@@ -625,6 +694,58 @@ class ModelRoutingService(BaseService):
|
||||
)
|
||||
self.log.info("Mode applied: grok", default_model=model_name)
|
||||
|
||||
async def _apply_codex(self, default_model: str | None) -> None:
|
||||
"""Wipe assignments, set the GLOBAL default to a Codex (OpenAI) model.
|
||||
|
||||
Migration 083 already seeds the OPENAI provider row `enabled=true`
|
||||
(there's no key to withhold behind a disabled row — subscription
|
||||
auth via a mounted `~/.codex`), but this mode's own force-enable is
|
||||
belt-and-suspenders against a row disabled by some other path,
|
||||
mirroring `_apply_grok`. AGENT_SLUG pins and complexity overrides are
|
||||
preserved (see `_wipe_mode_switch_assignments`).
|
||||
"""
|
||||
await self._wipe_mode_switch_assignments()
|
||||
codex = await self._get_seeded_provider(ModelProvider.OPENAI)
|
||||
provider_svc = ProviderService(self.session)
|
||||
await provider_svc.update_provider(
|
||||
require_uuid(codex.id),
|
||||
ProviderUpdate(enabled=True),
|
||||
)
|
||||
model_name = default_model or "gpt-5.3-codex"
|
||||
await self.upsert_assignment(
|
||||
scope=AssignmentScope.GLOBAL,
|
||||
scope_value=None,
|
||||
model_name=model_name,
|
||||
)
|
||||
self.log.info("Mode applied: codex", default_model=model_name)
|
||||
|
||||
async def _apply_gemini(self, default_model: str | None) -> None:
|
||||
"""Wipe assignments, set the GLOBAL default to a Gemini (Google) model.
|
||||
|
||||
Migration 085 seeded the GEMINI provider row `enabled=false`
|
||||
(migration 086 flips it to `enabled=true` at rest, matching Codex),
|
||||
so this mode's force-enable is the same belt-and-suspenders step
|
||||
`_apply_grok` runs for GROK — the mode switch must not depend on the
|
||||
seed migration alone. GeminiCliProvider authenticates via a mounted
|
||||
OAuth credential (`~/.gemini`), not a stored API key, so there is no
|
||||
key-check precondition. AGENT_SLUG pins and complexity overrides are
|
||||
preserved (see `_wipe_mode_switch_assignments`).
|
||||
"""
|
||||
await self._wipe_mode_switch_assignments()
|
||||
gemini = await self._get_seeded_provider(ModelProvider.GEMINI)
|
||||
provider_svc = ProviderService(self.session)
|
||||
await provider_svc.update_provider(
|
||||
require_uuid(gemini.id),
|
||||
ProviderUpdate(enabled=True),
|
||||
)
|
||||
model_name = default_model or "gemini-2.5-pro"
|
||||
await self.upsert_assignment(
|
||||
scope=AssignmentScope.GLOBAL,
|
||||
scope_value=None,
|
||||
model_name=model_name,
|
||||
)
|
||||
self.log.info("Mode applied: gemini", default_model=model_name)
|
||||
|
||||
async def _apply_ollama(self, default_model: str | None) -> None:
|
||||
"""Wipe role/global assignments, set GLOBAL to an Ollama Cloud model.
|
||||
|
||||
@@ -778,10 +899,16 @@ class ModelRoutingService(BaseService):
|
||||
"""Validate one preset payload entry WITHOUT writing anything.
|
||||
|
||||
Replicates every check `upsert_assignment` would apply — scope shape
|
||||
(`_validate_scope`) and a resolvable provider
|
||||
(`resolve_provider_for_model`) — so `apply_routing_preset` can vet the
|
||||
whole payload before touching the DB. Returns the parsed
|
||||
`(scope, scope_value, model_name)` tuple when valid, else `None`.
|
||||
(`_validate_scope`), a resolvable provider (`resolve_provider_for_model`),
|
||||
AND that provider's current `.enabled` state — so `apply_routing_preset`
|
||||
can vet the whole payload before touching the DB. The `.enabled` check
|
||||
catches a preset saved while a provider was live (a key set, self-hosted
|
||||
connected, Codex/Gemini enabled) that has since gone disabled: applying
|
||||
it would otherwise silently restore a dead assignment that resolves
|
||||
through to the legacy Anthropic fallback at spawn — the same class of
|
||||
bug `resolve_for_agent`'s own disabled-provider branch guards against.
|
||||
Returns the parsed `(scope, scope_value, model_name)` tuple when valid,
|
||||
else `None`.
|
||||
"""
|
||||
model_name = entry.get("model_name")
|
||||
scope_raw = entry.get("scope")
|
||||
@@ -794,10 +921,11 @@ class ModelRoutingService(BaseService):
|
||||
except ValueError:
|
||||
return None
|
||||
try:
|
||||
if await self.resolve_provider_for_model(model_name) is None:
|
||||
return None
|
||||
provider = await self.resolve_provider_for_model(model_name)
|
||||
except NotFoundError:
|
||||
return None
|
||||
if provider is None or not provider.enabled:
|
||||
return None
|
||||
return scope, scope_value, model_name
|
||||
|
||||
async def apply_routing_preset(self, preset_id: UUID) -> list[str]:
|
||||
@@ -860,7 +988,9 @@ class ModelRoutingService(BaseService):
|
||||
if row is None:
|
||||
return None
|
||||
# Relationship is lazy="joined" in the ORM so `.provider` is loaded.
|
||||
return _ResolvedAssignment(provider=row.provider, model_name=row.model_name)
|
||||
return _ResolvedAssignment(
|
||||
provider=row.provider, model_name=row.model_name, scope=row.scope
|
||||
)
|
||||
|
||||
async def _find_local_provider(self) -> ProviderConfigTable | None:
|
||||
"""Return the LOCAL provider row, or None if not seeded."""
|
||||
|
||||
@@ -2679,6 +2679,16 @@ class TaskService(BaseService):
|
||||
elif status == "conflict":
|
||||
self._note_base_inheritance_conflict(task, base_branch, result)
|
||||
await self.session.flush()
|
||||
elif status == "merged_push_failed":
|
||||
# Local merge succeeded, push to origin failed. Self-heals on
|
||||
# the dev's next real push, but tell the dev so a failed
|
||||
# submit isn't a mystery.
|
||||
self._append_base_inheritance_dev_note(
|
||||
task,
|
||||
f"Upstream base {base_branch!r} was merged locally but the "
|
||||
f"push to origin failed; your next push carries it forward.",
|
||||
)
|
||||
await self.session.flush()
|
||||
elif status not in ("already_ancestor", "missing_ref"):
|
||||
# missing_ref is quiet: a merged-and-deleted parent branch
|
||||
# simply has nothing left to inherit.
|
||||
@@ -2711,6 +2721,10 @@ class TaskService(BaseService):
|
||||
"base_inheritance_conflict",
|
||||
f"{existing}\n{note}" if existing else note,
|
||||
)
|
||||
# The transition-note marker isn't surfaced to the agent; dev_notes IS
|
||||
# (it rides evidence()/build_task_handoff), so the dev actually sees the
|
||||
# conflict on its next turn instead of only in orchestrator logs.
|
||||
self._append_base_inheritance_dev_note(task, note)
|
||||
self.log.warning(
|
||||
"upstream base inheritance conflict",
|
||||
task_id=str(task.id),
|
||||
@@ -2719,6 +2733,10 @@ class TaskService(BaseService):
|
||||
files=result.get("files"),
|
||||
)
|
||||
|
||||
def _append_base_inheritance_dev_note(self, task: TaskTable, note: str) -> None:
|
||||
"""Surface a base-inheritance note to the assignee via dev_notes."""
|
||||
task.dev_notes = _append_capped(task.dev_notes, f"[BASE INHERITANCE] {note}")
|
||||
|
||||
async def _distinct_projects_for_task(self, task: TaskTable) -> list[UUID]:
|
||||
"""The distinct projects a coordination root's map spans — one
|
||||
``feature/main_pm/{root}`` integration branch each.
|
||||
|
||||
@@ -53,15 +53,21 @@ async def llm_setup(
|
||||
base_url="https://ollama.example.com",
|
||||
)
|
||||
# Mirrors migration 083_seed_openai_provider's contract: enabled=True at
|
||||
# seed time (no apply_mode="codex" write path exists to flip it later —
|
||||
# see that migration's docstring).
|
||||
# seed time.
|
||||
openai = ProviderConfigTable(
|
||||
name="openai-test",
|
||||
type=ModelProvider.OPENAI,
|
||||
enabled=True,
|
||||
base_url="https://api.openai.com/v1",
|
||||
)
|
||||
db_session.add_all([anthropic, grok, ollama, openai])
|
||||
# Mirrors the post-086 seeded state (085 seeds enabled=false, 086 flips it
|
||||
# true to match Codex) — no base_url, subscription OAuth auth only.
|
||||
gemini = ProviderConfigTable(
|
||||
name="gemini-test",
|
||||
type=ModelProvider.GEMINI,
|
||||
enabled=True,
|
||||
)
|
||||
db_session.add_all([anthropic, grok, ollama, openai, gemini])
|
||||
await db_session.flush()
|
||||
yield {"svc": ModelRoutingService(db_session)}
|
||||
|
||||
@@ -217,6 +223,18 @@ async def test_derive_mode_codex_when_only_openai_global(llm_setup: dict) -> Non
|
||||
assert await svc.derive_mode() == "codex"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_derive_mode_gemini_when_only_gemini_global(llm_setup: dict) -> None:
|
||||
"""A pure-GEMINI global assignment reports "gemini", not the catch-all
|
||||
"mix" — mirrors the codex branch derive_mode already carries."""
|
||||
svc = llm_setup["svc"]
|
||||
gemini_model = _first_model_for_type(ModelProvider.GEMINI)
|
||||
await svc.upsert_assignment(
|
||||
scope=AssignmentScope.GLOBAL, scope_value=None, model_name=gemini_model
|
||||
)
|
||||
assert await svc.derive_mode() == "gemini"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_derive_mode_mix_with_per_agent(llm_setup: dict) -> None:
|
||||
svc = llm_setup["svc"]
|
||||
@@ -332,6 +350,75 @@ async def test_apply_mode_grok_enables_grok_provider(llm_setup: dict) -> None:
|
||||
assert refetched.enabled is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_codex_sets_global(llm_setup: dict) -> None:
|
||||
svc = llm_setup["svc"]
|
||||
await svc.apply_mode(mode="codex")
|
||||
assignments = await svc.list_assignments()
|
||||
assert len(assignments) == 1
|
||||
assert assignments[0].scope == AssignmentScope.GLOBAL
|
||||
assert assignments[0].provider.type == ModelProvider.OPENAI
|
||||
assert assignments[0].model_name == "gpt-5.3-codex"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_codex_enables_openai_provider(llm_setup: dict) -> None:
|
||||
"""apply_mode('codex') force-enables the OPENAI row — belt-and-suspenders
|
||||
alongside migration 083's own enabled=true seed."""
|
||||
svc = llm_setup["svc"]
|
||||
provider_svc = ProviderService(svc.session)
|
||||
openai = next(
|
||||
p
|
||||
for p in await provider_svc.list_providers(include_disabled=True)
|
||||
if p.type == ModelProvider.OPENAI
|
||||
)
|
||||
await provider_svc.update_provider(
|
||||
cast("UUID", openai.id), ProviderUpdate(enabled=False)
|
||||
)
|
||||
await svc.session.flush()
|
||||
|
||||
await svc.apply_mode(mode="codex")
|
||||
|
||||
refetched = await provider_svc.get_provider(cast("UUID", openai.id))
|
||||
assert refetched is not None
|
||||
assert refetched.enabled is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_gemini_sets_global(llm_setup: dict) -> None:
|
||||
svc = llm_setup["svc"]
|
||||
await svc.apply_mode(mode="gemini")
|
||||
assignments = await svc.list_assignments()
|
||||
assert len(assignments) == 1
|
||||
assert assignments[0].scope == AssignmentScope.GLOBAL
|
||||
assert assignments[0].provider.type == ModelProvider.GEMINI
|
||||
assert assignments[0].model_name == "gemini-2.5-pro"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_gemini_enables_gemini_provider(llm_setup: dict) -> None:
|
||||
"""apply_mode('gemini') force-enables the GEMINI row — the exact gap this
|
||||
fix closes (migration 085 seeds it disabled and nothing else ever flipped
|
||||
it before this write path + migration 086 existed)."""
|
||||
svc = llm_setup["svc"]
|
||||
provider_svc = ProviderService(svc.session)
|
||||
gemini = next(
|
||||
p
|
||||
for p in await provider_svc.list_providers(include_disabled=True)
|
||||
if p.type == ModelProvider.GEMINI
|
||||
)
|
||||
await provider_svc.update_provider(
|
||||
cast("UUID", gemini.id), ProviderUpdate(enabled=False)
|
||||
)
|
||||
await svc.session.flush()
|
||||
|
||||
await svc.apply_mode(mode="gemini")
|
||||
|
||||
refetched = await provider_svc.get_provider(cast("UUID", gemini.id))
|
||||
assert refetched is not None
|
||||
assert refetched.enabled is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_mix_requires_per_agent(llm_setup: dict) -> None:
|
||||
svc = llm_setup["svc"]
|
||||
@@ -440,6 +527,129 @@ async def test_upsert_and_resolve_openai_assignment_roundtrip(
|
||||
assert route.auth_token is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_upsert_and_resolve_gemini_assignment_roundtrip(
|
||||
llm_setup: dict,
|
||||
) -> None:
|
||||
"""gemini-2.5-pro through upsert_assignment -> resolve_for_agent, against
|
||||
the seeded GEMINI row. Proves resolve_for_agent actually returns a GEMINI
|
||||
spawn route — not a silent Anthropic fallback — the exact gap left open
|
||||
by the row seeding disabled with no enable path (migration 085 alone)."""
|
||||
svc = llm_setup["svc"]
|
||||
gemini_model = _first_model_for_type(ModelProvider.GEMINI)
|
||||
row = await svc.upsert_assignment(
|
||||
scope=AssignmentScope.AGENT_SLUG,
|
||||
scope_value="ux-dev-1",
|
||||
model_name=gemini_model,
|
||||
)
|
||||
assert row.model_name == gemini_model
|
||||
|
||||
route = await svc.resolve_for_agent("ux-dev-1")
|
||||
assert route.provider_type == ModelProvider.GEMINI
|
||||
assert route.model_name == gemini_model
|
||||
# Subscription OAuth auth (~/.gemini), not a decrypted provider token.
|
||||
assert route.auth_token is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_upsert_assignment_enables_disabled_gemini_provider(
|
||||
llm_setup: dict,
|
||||
) -> None:
|
||||
"""Belt-and-suspenders: assigning a Gemini model via Mix (upsert_assignment)
|
||||
force-enables the row even if it was disabled — not just apply_mode('gemini')."""
|
||||
svc = llm_setup["svc"]
|
||||
provider_svc = ProviderService(svc.session)
|
||||
gemini = next(
|
||||
p
|
||||
for p in await provider_svc.list_providers(include_disabled=True)
|
||||
if p.type == ModelProvider.GEMINI
|
||||
)
|
||||
await provider_svc.update_provider(
|
||||
cast("UUID", gemini.id), ProviderUpdate(enabled=False)
|
||||
)
|
||||
await svc.session.flush()
|
||||
|
||||
gemini_model = _first_model_for_type(ModelProvider.GEMINI)
|
||||
await svc.upsert_assignment(
|
||||
scope=AssignmentScope.AGENT_SLUG,
|
||||
scope_value="ux-dev-1",
|
||||
model_name=gemini_model,
|
||||
)
|
||||
|
||||
refetched = await provider_svc.get_provider(cast("UUID", gemini.id))
|
||||
assert refetched is not None
|
||||
assert refetched.enabled is True
|
||||
# And the route actually resolves to GEMINI now that it's enabled.
|
||||
route = await svc.resolve_for_agent("ux-dev-1")
|
||||
assert route.provider_type == ModelProvider.GEMINI
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_gemini_end_to_end_reachable(llm_setup: dict) -> None:
|
||||
"""The full reachability chain the original drill missed: apply_mode
|
||||
-> derive_mode reflects it -> resolve_for_agent actually spawns Gemini."""
|
||||
svc = llm_setup["svc"]
|
||||
await svc.apply_mode(mode="gemini")
|
||||
|
||||
assert await svc.derive_mode() == "gemini"
|
||||
|
||||
route = await svc.resolve_for_agent("ux-dev-1")
|
||||
assert route.provider_type == ModelProvider.GEMINI
|
||||
assert route.model_name == "gemini-2.5-pro"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_codex_end_to_end_reachable(llm_setup: dict) -> None:
|
||||
"""Same reachability chain for Codex, mirroring the Gemini test above."""
|
||||
svc = llm_setup["svc"]
|
||||
await svc.apply_mode(mode="codex")
|
||||
|
||||
assert await svc.derive_mode() == "codex"
|
||||
|
||||
route = await svc.resolve_for_agent("be-dev-1")
|
||||
assert route.provider_type == ModelProvider.OPENAI
|
||||
assert route.model_name == "gpt-5.3-codex"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("mode", ["codex", "gemini"])
|
||||
@pytest.mark.parametrize("interactive_slug", ["intake-1", "secretary-1"])
|
||||
async def test_interactive_agents_exempt_from_delivery_only_global_mode(
|
||||
llm_setup: dict, mode: str, interactive_slug: str
|
||||
) -> None:
|
||||
"""A fleet-wide Codex/Gemini mode must not capture Intake/Secretary —
|
||||
they have no V1 support on those providers, so the resolver keeps them
|
||||
on the legacy Anthropic path (the completeness-drill gap: previously
|
||||
they resolved to the unsupported provider and the spawn guard left both
|
||||
chats refusing to start after a one-click mode switch)."""
|
||||
svc = llm_setup["svc"]
|
||||
await svc.apply_mode(mode=mode)
|
||||
|
||||
# The mode still derives cleanly (single GLOBAL row — no extra pins).
|
||||
assert await svc.derive_mode() == mode
|
||||
|
||||
route = await svc.resolve_for_agent(interactive_slug)
|
||||
assert route.provider_type == ModelProvider.ANTHROPIC
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_agent_explicit_pin_is_not_exempted(
|
||||
llm_setup: dict,
|
||||
) -> None:
|
||||
"""An EXPLICIT AGENT_SLUG pin to a delivery-only provider is honored by
|
||||
the resolver (the orchestrator's spawn guard refuses it loudly) — a
|
||||
deliberate operator choice must error, never be silently overridden."""
|
||||
svc = llm_setup["svc"]
|
||||
await svc.upsert_assignment(
|
||||
scope=AssignmentScope.AGENT_SLUG,
|
||||
scope_value="intake-1",
|
||||
model_name="gpt-5.3-codex",
|
||||
)
|
||||
|
||||
route = await svc.resolve_for_agent("intake-1")
|
||||
assert route.provider_type == ModelProvider.OPENAI
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_for_agent_uses_provider_token(llm_setup: dict) -> None:
|
||||
"""When provider has auth_token_encrypted, it's decrypted (lines 345-346)."""
|
||||
@@ -1107,3 +1317,42 @@ async def test_apply_routing_preset_validates_before_wiping_anything(
|
||||
# rather than a stale expectation of survival.
|
||||
remaining = await svc.list_assignments()
|
||||
assert remaining == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_routing_preset_skips_entry_whose_provider_went_disabled(
|
||||
llm_setup: dict,
|
||||
) -> None:
|
||||
"""A preset entry that resolved fine at save time but whose provider has
|
||||
SINCE been disabled (key cleared, self-hosted disconnected, Codex/Gemini
|
||||
disabled) must be skipped-with-note, never silently restored — applying
|
||||
a preset can't resurrect a dead route behind a success toast."""
|
||||
svc = llm_setup["svc"]
|
||||
gemini_model = _first_model_for_type(ModelProvider.GEMINI)
|
||||
await svc.upsert_assignment(
|
||||
scope=AssignmentScope.GLOBAL, scope_value=None, model_name=gemini_model
|
||||
)
|
||||
preset = await svc.save_routing_preset("gemini-then-disabled")
|
||||
|
||||
# Disable the GEMINI provider AFTER the preset was saved (mirrors an
|
||||
# operator turning it off, or a fresh env where the row starts disabled).
|
||||
provider_svc = ProviderService(svc.session)
|
||||
gemini = next(
|
||||
p
|
||||
for p in await provider_svc.list_providers(include_disabled=True)
|
||||
if p.type == ModelProvider.GEMINI
|
||||
)
|
||||
await provider_svc.update_provider(
|
||||
cast("UUID", gemini.id), ProviderUpdate(enabled=False)
|
||||
)
|
||||
await svc.session.flush()
|
||||
|
||||
# Clear current routing so the preset apply has something to (not) restore.
|
||||
await svc.apply_mode(mode="anthropic")
|
||||
|
||||
notes = await svc.apply_routing_preset(preset.id)
|
||||
assert len(notes) == 1
|
||||
assert "unavailable" in notes[0]
|
||||
|
||||
remaining = await svc.list_assignments()
|
||||
assert remaining == [] # the disabled-provider entry was never written
|
||||
|
||||
@@ -13,6 +13,7 @@ from fastapi import FastAPI
|
||||
from httpx import ASGITransport, AsyncClient
|
||||
from roboco.api.deps import get_agent_context, get_db
|
||||
from roboco.api.routes.project import router as project_router
|
||||
from roboco.config import settings
|
||||
from roboco.db.tables import AgentTable
|
||||
from roboco.models import AgentRole, AgentStatus, Team
|
||||
from roboco.models.permissions import AgentContext
|
||||
@@ -166,6 +167,79 @@ async def test_update_project_explicit_null_clears_field(
|
||||
assert cleared.json()["test_command"] is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_rejects_zero_monthly_budget_usd(
|
||||
project_client: AsyncClient,
|
||||
) -> None:
|
||||
"""#654: a 0 cap would block every claim immediately — rejected at the
|
||||
request boundary, never stored."""
|
||||
create = await project_client.post("/api/projects", json=_payload(), headers=_HDR)
|
||||
pid = create.json()["id"]
|
||||
response = await project_client.patch(
|
||||
f"/api/projects/{pid}",
|
||||
json={"monthly_budget_usd": 0},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.UNPROCESSABLE_ENTITY
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_rejects_negative_monthly_budget_usd(
|
||||
project_client: AsyncClient,
|
||||
) -> None:
|
||||
create = await project_client.post("/api/projects", json=_payload(), headers=_HDR)
|
||||
pid = create.json()["id"]
|
||||
response = await project_client.patch(
|
||||
f"/api/projects/{pid}",
|
||||
json={"monthly_budget_usd": -5},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.UNPROCESSABLE_ENTITY
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_accepts_positive_monthly_budget_usd(
|
||||
project_client: AsyncClient,
|
||||
) -> None:
|
||||
create = await project_client.post("/api/projects", json=_payload(), headers=_HDR)
|
||||
pid = create.json()["id"]
|
||||
cap = 100
|
||||
response = await project_client.patch(
|
||||
f"/api/projects/{pid}",
|
||||
json={"monthly_budget_usd": cap},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["monthly_budget_usd"] == cap
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_project_by_id_includes_spend_when_budgets_enabled(
|
||||
project_client: AsyncClient, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""monthly_spend_usd is populated (0.0 with no spawn sessions yet) once
|
||||
ROBOCO_TASK_BUDGETS_ENABLED is on — the extra DB read only runs then."""
|
||||
monkeypatch.setattr(settings, "task_budgets_enabled", True)
|
||||
create = await project_client.post("/api/projects", json=_payload(), headers=_HDR)
|
||||
pid = create.json()["id"]
|
||||
response = await project_client.get(f"/api/projects/{pid}", headers=_HDR)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["monthly_spend_usd"] == 0.0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_project_by_id_omits_spend_when_budgets_disabled(
|
||||
project_client: AsyncClient, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""Flag off => monthly_spend_usd stays null, same as before this field existed."""
|
||||
monkeypatch.setattr(settings, "task_budgets_enabled", False)
|
||||
create = await project_client.post("/api/projects", json=_payload(), headers=_HDR)
|
||||
pid = create.json()["id"]
|
||||
response = await project_client.get(f"/api/projects/{pid}", headers=_HDR)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["monthly_spend_usd"] is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_not_found(project_client: AsyncClient) -> None:
|
||||
response = await project_client.patch(
|
||||
|
||||
@@ -330,6 +330,117 @@ async def test_apply_mode_ollama_without_provider_returns_404(
|
||||
assert response.status_code == HTTPStatus.NOT_FOUND
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def app_client_with_codex_and_gemini(
|
||||
db_session: AsyncSession,
|
||||
) -> AsyncIterator[AsyncClient]:
|
||||
"""App client pre-seeded with Anthropic + disabled OPENAI/GEMINI providers
|
||||
(mirrors the real seeded state before an operator ever applies either
|
||||
mode: OPENAI seeds enabled=true per migration 083, GEMINI seeds
|
||||
enabled=false per migration 085 — deliberately seeded disabled here so the
|
||||
apply-mode round trip below proves the force-enable, not a pre-enabled
|
||||
no-op)."""
|
||||
app = _make_app(db_session)
|
||||
suffix = uuid4().hex[:8]
|
||||
await db_session.execute(delete(ModelAssignmentTable))
|
||||
await db_session.execute(delete(ProviderConfigTable))
|
||||
await db_session.flush()
|
||||
db_session.add(
|
||||
ProviderConfigTable(
|
||||
name=f"anthropic-cg-{suffix}", type=ModelProvider.ANTHROPIC, enabled=True
|
||||
)
|
||||
)
|
||||
db_session.add(
|
||||
ProviderConfigTable(
|
||||
name=f"codex-cg-{suffix}",
|
||||
type=ModelProvider.OPENAI,
|
||||
enabled=False,
|
||||
base_url="https://api.openai.com/v1",
|
||||
)
|
||||
)
|
||||
db_session.add(
|
||||
ProviderConfigTable(
|
||||
name=f"gemini-cg-{suffix}", type=ModelProvider.GEMINI, enabled=False
|
||||
)
|
||||
)
|
||||
await db_session.flush()
|
||||
transport = ASGITransport(app=app)
|
||||
async with AsyncClient(transport=transport, base_url="http://test") as client:
|
||||
yield client
|
||||
app.dependency_overrides.clear()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_codex_returns_200_reflects_mode_and_enables_provider(
|
||||
app_client_with_codex_and_gemini: AsyncClient,
|
||||
) -> None:
|
||||
"""The full HTTP round trip: POST mode="codex" -> 200, GET reflects
|
||||
mode="codex", and the assignment resolves through the now-enabled OPENAI
|
||||
provider — proving the pydantic Literal + dispatch + enable chain end to
|
||||
end, not just the service-layer call this mirrors."""
|
||||
response = await app_client_with_codex_and_gemini.post(
|
||||
"/api/providers", json={"mode": "codex"}, headers=_HDR_PM
|
||||
)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
body = response.json()
|
||||
assert body["mode"] == "codex"
|
||||
assert body["assignments"][0]["provider_type"] == "openai"
|
||||
assert body["assignments"][0]["model_name"] == "gpt-5.3-codex"
|
||||
|
||||
followup = await app_client_with_codex_and_gemini.get(
|
||||
"/api/providers", headers=_HDR_PM
|
||||
)
|
||||
assert followup.json()["mode"] == "codex"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_gemini_returns_200_reflects_mode_and_enables_provider(
|
||||
app_client_with_codex_and_gemini: AsyncClient,
|
||||
) -> None:
|
||||
"""Same round trip as Codex's, for Gemini — the exact reachability gap
|
||||
this fix closes (the row seeds disabled and nothing else ever flipped it)."""
|
||||
response = await app_client_with_codex_and_gemini.post(
|
||||
"/api/providers", json={"mode": "gemini"}, headers=_HDR_PM
|
||||
)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
body = response.json()
|
||||
assert body["mode"] == "gemini"
|
||||
assert body["assignments"][0]["provider_type"] == "gemini"
|
||||
assert body["assignments"][0]["model_name"] == "gemini-2.5-pro"
|
||||
|
||||
followup = await app_client_with_codex_and_gemini.get(
|
||||
"/api/providers", headers=_HDR_PM
|
||||
)
|
||||
assert followup.json()["mode"] == "gemini"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_mode_gemini_without_provider_returns_404(
|
||||
db_session: AsyncSession,
|
||||
) -> None:
|
||||
"""Apply 'gemini' mode without the GEMINI provider seeded raises
|
||||
NotFoundError -> 404 (mirrors the ollama/grok equivalents)."""
|
||||
# FK-safe: a prior test may have committed a real GEMINI assignment
|
||||
# (model_assignments.provider_config_id references provider_configs.id),
|
||||
# so assignments must be cleared before the provider row can be deleted.
|
||||
await db_session.execute(delete(ModelAssignmentTable))
|
||||
await db_session.execute(
|
||||
delete(ProviderConfigTable).where(
|
||||
ProviderConfigTable.type == ModelProvider.GEMINI
|
||||
)
|
||||
)
|
||||
await db_session.flush()
|
||||
|
||||
app = _make_app(db_session)
|
||||
transport = ASGITransport(app=app)
|
||||
async with AsyncClient(transport=transport, base_url="http://test") as client:
|
||||
response = await client.post(
|
||||
"/api/providers", json={"mode": "gemini"}, headers=_HDR_PM
|
||||
)
|
||||
app.dependency_overrides.clear()
|
||||
assert response.status_code == HTTPStatus.NOT_FOUND
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Self-hosted endpoints
|
||||
# =============================================================================
|
||||
@@ -876,6 +987,14 @@ async def test_save_list_and_apply_preset_round_trip(
|
||||
app_client_with_ollama: AsyncClient,
|
||||
) -> None:
|
||||
"""Save captures the current state; mutating + re-applying restores it."""
|
||||
# Set the Ollama key first — the fixture seeds OLLAMA_CLOUD `enabled=False`
|
||||
# (no key yet), and `_validate_preset_entry` now rejects (skip-with-note)
|
||||
# any preset entry whose provider is disabled, so a meaningful round trip
|
||||
# needs the provider actually live, same as the real UI's key-gated mode
|
||||
# button.
|
||||
await app_client_with_ollama.put(
|
||||
"/api/providers/ollama-key", json={"api_key": "test-key"}, headers=_HDR_PM
|
||||
)
|
||||
# Arrange a distinctive state: a GLOBAL Ollama default.
|
||||
await app_client_with_ollama.post(
|
||||
"/api/providers", json={"mode": "ollama"}, headers=_HDR_PM
|
||||
|
||||
@@ -22,6 +22,7 @@ from roboco.api.routes.tasks import (
|
||||
from roboco.api.routes.tasks import (
|
||||
router as tasks_router,
|
||||
)
|
||||
from roboco.config import settings
|
||||
from roboco.db.tables import AgentTable, ProjectTable, TaskTable, WorkSessionTable
|
||||
from roboco.exceptions import GitError, TaskLifecycleError
|
||||
from roboco.foundation.policy.lifecycle import STATUS_GRAPH
|
||||
@@ -278,6 +279,35 @@ async def test_get_task_by_id(task_client: dict) -> None:
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_task_by_id_includes_spend_when_budgets_enabled(
|
||||
task_client: dict, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""spend_usd is populated (0.0 with no spawn sessions yet) once
|
||||
ROBOCO_TASK_BUDGETS_ENABLED is on — the extra DB read only runs then."""
|
||||
monkeypatch.setattr(settings, "task_budgets_enabled", True)
|
||||
client = task_client["client"]
|
||||
task = _seed_task(task_client)
|
||||
await task_client["db"].flush()
|
||||
response = await client.get(f"/api/tasks/{task.id}", headers=_HDR)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["spend_usd"] == 0.0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_task_by_id_omits_spend_when_budgets_disabled(
|
||||
task_client: dict, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""Flag off => spend_usd stays null, the same as before this field existed."""
|
||||
monkeypatch.setattr(settings, "task_budgets_enabled", False)
|
||||
client = task_client["client"]
|
||||
task = _seed_task(task_client)
|
||||
await task_client["db"].flush()
|
||||
response = await client.get(f"/api/tasks/{task.id}", headers=_HDR)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["spend_usd"] is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_task(task_client: dict) -> None:
|
||||
client = task_client["client"]
|
||||
@@ -291,6 +321,52 @@ async def test_update_task(task_client: dict) -> None:
|
||||
assert response.status_code in (HTTPStatus.OK, HTTPStatus.UNPROCESSABLE_ENTITY)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_task_rejects_zero_budget_usd(task_client: dict) -> None:
|
||||
"""#654: a 0 cap would block every claim immediately — rejected at the
|
||||
request boundary, never stored."""
|
||||
client = task_client["client"]
|
||||
task = _seed_task(task_client)
|
||||
await task_client["db"].flush()
|
||||
response = await client.patch(
|
||||
f"/api/tasks/{task.id}",
|
||||
json={"budget_usd": 0},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.UNPROCESSABLE_ENTITY
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_task_rejects_negative_budget_usd(task_client: dict) -> None:
|
||||
client = task_client["client"]
|
||||
task = _seed_task(task_client)
|
||||
await task_client["db"].flush()
|
||||
response = await client.patch(
|
||||
f"/api/tasks/{task.id}",
|
||||
json={"budget_usd": -5},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.UNPROCESSABLE_ENTITY
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_task_accepts_positive_budget_usd(task_client: dict) -> None:
|
||||
# budget_usd is a _PRIVILEGED_UPDATE_FIELDS / non-"PM lighter" field —
|
||||
# a plain main_pm PATCH would 403 here, so exercise the CEO's full scope.
|
||||
_as_ceo(task_client)
|
||||
client = task_client["client"]
|
||||
task = _seed_task(task_client)
|
||||
await task_client["db"].flush()
|
||||
budget = 12.5
|
||||
response = await client.patch(
|
||||
f"/api/tasks/{task.id}",
|
||||
json={"budget_usd": budget},
|
||||
headers=_HDR,
|
||||
)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
assert response.json()["budget_usd"] == budget
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_task_status_override_recovers_blocked(task_client: dict) -> None:
|
||||
"""A privileged PATCH with ``status`` + ``force`` is applied as an audited
|
||||
|
||||
@@ -277,6 +277,27 @@ def test_task_update_sequence_rejects_negative() -> None:
|
||||
TaskUpdate(sequence=-1)
|
||||
|
||||
|
||||
def test_task_update_budget_usd_accepts_null() -> None:
|
||||
"""null clears the cap back to the TaskType default — always valid."""
|
||||
assert TaskUpdate(budget_usd=None).budget_usd is None
|
||||
|
||||
|
||||
def test_task_update_budget_usd_accepts_positive() -> None:
|
||||
budget = 5.0
|
||||
assert TaskUpdate(budget_usd=budget).budget_usd == budget
|
||||
|
||||
|
||||
def test_task_update_budget_usd_rejects_zero() -> None:
|
||||
"""gt=0 — a 0 budget would block every claim immediately (#654)."""
|
||||
with pytest.raises(ValueError, match="budget_usd"):
|
||||
TaskUpdate(budget_usd=0)
|
||||
|
||||
|
||||
def test_task_update_budget_usd_rejects_negative() -> None:
|
||||
with pytest.raises(ValueError, match="budget_usd"):
|
||||
TaskUpdate(budget_usd=-5)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# task_to_response / task_list_to_response
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -41,9 +41,30 @@ def test_render_config_toml_marks_gateway_pair_required() -> None:
|
||||
assert "required" not in parsed["mcp_servers"]["roboco-optimal"]
|
||||
|
||||
|
||||
def test_render_config_toml_empty_when_no_servers() -> None:
|
||||
assert cc.render_config_toml({}) == ""
|
||||
assert cc.render_config_toml({"mcpServers": {}}) == ""
|
||||
def test_render_config_toml_widens_startup_timeout_on_required_servers() -> None:
|
||||
# The CLI's default 10s MCP startup timeout fail-fast-aborts the session on
|
||||
# a cold uv wheel cache; the gateway pair gets a wider budget.
|
||||
parsed = tomllib.loads(cc.render_config_toml(_SAMPLE_MCP))
|
||||
timeout = cc._REQUIRED_MCP_STARTUP_TIMEOUT_SEC
|
||||
assert parsed["mcp_servers"]["roboco-flow"]["startup_timeout_sec"] == timeout
|
||||
assert parsed["mcp_servers"]["roboco-do"]["startup_timeout_sec"] == timeout
|
||||
assert "startup_timeout_sec" not in parsed["mcp_servers"]["roboco-optimal"]
|
||||
|
||||
|
||||
def test_render_config_toml_disables_subagents_unconditionally() -> None:
|
||||
# Fleet-wide subagent ban (CEO, 2026-07-09) — a global switch, not
|
||||
# per-role, so it renders even with no MCP servers configured at all.
|
||||
no_servers = tomllib.loads(cc.render_config_toml({}))
|
||||
empty_servers = tomllib.loads(cc.render_config_toml({"mcpServers": {}}))
|
||||
with_servers = tomllib.loads(cc.render_config_toml(_SAMPLE_MCP))
|
||||
assert no_servers["agents"]["enabled"] is False
|
||||
assert empty_servers["agents"]["enabled"] is False
|
||||
assert with_servers["agents"]["enabled"] is False
|
||||
|
||||
|
||||
def test_render_config_toml_no_mcp_servers_key_when_no_servers() -> None:
|
||||
assert "mcp_servers" not in tomllib.loads(cc.render_config_toml({}))
|
||||
assert "mcp_servers" not in tomllib.loads(cc.render_config_toml({"mcpServers": {}}))
|
||||
|
||||
|
||||
def test_sandbox_level_developer_is_workspace_write() -> None:
|
||||
|
||||
@@ -125,8 +125,13 @@ def test_write_policy_toml_writes_file(tmp_path: Path) -> None:
|
||||
assert "run_shell_command" in written
|
||||
|
||||
|
||||
def test_gemini_cli_args_is_yolo_only() -> None:
|
||||
assert gc.gemini_cli_args() == ["--approval-mode", "yolo"]
|
||||
def test_gemini_cli_args_is_yolo_plus_default_max_turns() -> None:
|
||||
assert gc.gemini_cli_args() == ["--approval-mode", "yolo", "--max-turns", "200"]
|
||||
|
||||
|
||||
def test_gemini_cli_args_max_turns_is_overridable() -> None:
|
||||
args = gc.gemini_cli_args(max_turns=7)
|
||||
assert args[args.index("--max-turns") + 1] == "7"
|
||||
|
||||
|
||||
def test_main_writes_settings_and_args(
|
||||
@@ -161,4 +166,50 @@ def test_main_writes_settings_and_args(
|
||||
assert args_path.read_text(encoding="utf-8").splitlines() == [
|
||||
"--approval-mode",
|
||||
"yolo",
|
||||
"--max-turns",
|
||||
"200",
|
||||
]
|
||||
|
||||
|
||||
def test_main_honors_max_turns_env_override(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
mcp_path = tmp_path / "mcp-config.json"
|
||||
mcp_path.write_text(json.dumps(_SAMPLE_MCP), encoding="utf-8")
|
||||
args_path = tmp_path / "gemini-args"
|
||||
|
||||
monkeypatch.setattr(gc, "GEMINI_SETTINGS_PATH", tmp_path / ".gemini" / "s.json")
|
||||
monkeypatch.setattr(gc, "GEMINI_MEMORY_PATH", tmp_path / ".gemini" / "GEMINI.md")
|
||||
monkeypatch.setattr(gc, "GEMINI_POLICIES_DIR", tmp_path / ".gemini" / "policies")
|
||||
monkeypatch.setattr(gc, "GEMINI_ARGS_PATH", args_path)
|
||||
monkeypatch.setenv("ROBOCO_AGENT_ID", "be-dev-1")
|
||||
monkeypatch.setenv("ROBOCO_MCP_CONFIG", str(mcp_path))
|
||||
monkeypatch.setenv("ROBOCO_GEMINI_MAX_TURNS", "42")
|
||||
|
||||
assert gc.main() == 0
|
||||
assert args_path.read_text(encoding="utf-8").splitlines()[-2:] == [
|
||||
"--max-turns",
|
||||
"42",
|
||||
]
|
||||
|
||||
|
||||
def test_main_falls_back_to_default_max_turns_on_bad_env(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
mcp_path = tmp_path / "mcp-config.json"
|
||||
mcp_path.write_text(json.dumps(_SAMPLE_MCP), encoding="utf-8")
|
||||
args_path = tmp_path / "gemini-args"
|
||||
|
||||
monkeypatch.setattr(gc, "GEMINI_SETTINGS_PATH", tmp_path / ".gemini" / "s.json")
|
||||
monkeypatch.setattr(gc, "GEMINI_MEMORY_PATH", tmp_path / ".gemini" / "GEMINI.md")
|
||||
monkeypatch.setattr(gc, "GEMINI_POLICIES_DIR", tmp_path / ".gemini" / "policies")
|
||||
monkeypatch.setattr(gc, "GEMINI_ARGS_PATH", args_path)
|
||||
monkeypatch.setenv("ROBOCO_AGENT_ID", "be-dev-1")
|
||||
monkeypatch.setenv("ROBOCO_MCP_CONFIG", str(mcp_path))
|
||||
monkeypatch.setenv("ROBOCO_GEMINI_MAX_TURNS", "not-a-number")
|
||||
|
||||
assert gc.main() == 0
|
||||
assert args_path.read_text(encoding="utf-8").splitlines()[-2:] == [
|
||||
"--max-turns",
|
||||
"200",
|
||||
]
|
||||
|
||||
@@ -15,10 +15,19 @@ from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from roboco.config import settings
|
||||
from roboco.llm.providers import GeminiCliProvider, ProviderError, SpawnResult
|
||||
from roboco.llm.providers import gemini as gemini_module
|
||||
from roboco.models.runtime import OrchestratorAgentConfig
|
||||
|
||||
|
||||
def test_gemini_cli_model_is_a_real_settings_field() -> None:
|
||||
# Parity with codex_cli_model (roboco.config.Settings.codex_cli_model) —
|
||||
# gemini.py reads settings.gemini_cli_model, not a raw os.environ.get.
|
||||
assert settings.gemini_cli_model == gemini_module._GEMINI_CLI_MODEL
|
||||
assert settings.gemini_cli_model == "gemini-2.5-pro"
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_gemini_auth(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
|
||||
"""Point GEMINI_AUTH_HOST_PATH at a fresh tmp dir so tests never mount the
|
||||
|
||||
@@ -13,7 +13,7 @@ from __future__ import annotations
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from roboco.models.base import ModelProvider
|
||||
from roboco.models.base import AssignmentScope, ModelProvider
|
||||
from roboco.services.llm import ModelRoutingService, _ResolvedAssignment
|
||||
|
||||
_AGENT_SLUG = "be-dev-1"
|
||||
@@ -23,7 +23,9 @@ def _disabled_resolved() -> _ResolvedAssignment:
|
||||
provider = MagicMock(
|
||||
enabled=False, id="prov-disabled", type=ModelProvider.OLLAMA_CLOUD
|
||||
)
|
||||
return _ResolvedAssignment(provider=provider, model_name="grok-build")
|
||||
return _ResolvedAssignment(
|
||||
provider=provider, model_name="grok-build", scope=AssignmentScope.GLOBAL
|
||||
)
|
||||
|
||||
|
||||
def _svc() -> ModelRoutingService:
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
"""Task.budget_usd / Project.monthly_budget_usd validation (#654).
|
||||
|
||||
The task-budgets feature's own design says "0 rejected — a zero budget
|
||||
silently blocks everything" (every claim is refused from the first tick),
|
||||
so every schema that can set these fields must reject 0 and negative values
|
||||
at the pydantic boundary — a 422, never a stored self-DoS. Null ("no cap")
|
||||
stays valid throughout. Mirrors test_project_sandbox_services.py's style
|
||||
(domain-model `pytest.raises(ValidationError)` coverage).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
from roboco.models.base import Team
|
||||
from roboco.models.project import Project, ProjectCreate, ProjectUpdate
|
||||
from roboco.models.task import Task, TaskUpdate
|
||||
|
||||
|
||||
def _task(budget_usd: float | None = None) -> Task:
|
||||
return Task(
|
||||
title="Add user lookup endpoint",
|
||||
description="Add GET /v1/users/{id} returning user JSON.",
|
||||
acceptance_criteria=["returns 404 for unknown user"],
|
||||
created_by=uuid4(),
|
||||
team=Team.BACKEND,
|
||||
budget_usd=budget_usd,
|
||||
)
|
||||
|
||||
|
||||
def _project(monthly_budget_usd: float | None = None) -> Project:
|
||||
return Project(
|
||||
name="P",
|
||||
slug="p",
|
||||
git_url="https://example.com/r.git",
|
||||
assigned_cell=Team.BACKEND,
|
||||
created_by=uuid4(),
|
||||
monthly_budget_usd=monthly_budget_usd,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Task.budget_usd
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_task_defaults_budget_usd_to_none() -> None:
|
||||
assert _task().budget_usd is None
|
||||
|
||||
|
||||
def test_task_accepts_positive_budget_usd() -> None:
|
||||
budget = 12.5
|
||||
assert _task(budget_usd=budget).budget_usd == budget
|
||||
|
||||
|
||||
def test_task_rejects_zero_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="budget_usd"):
|
||||
_task(budget_usd=0)
|
||||
|
||||
|
||||
def test_task_rejects_negative_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="budget_usd"):
|
||||
_task(budget_usd=-5)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# roboco.models.task.TaskUpdate.budget_usd (domain update model)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_task_update_accepts_null_budget_usd() -> None:
|
||||
assert TaskUpdate(budget_usd=None).budget_usd is None
|
||||
|
||||
|
||||
def test_task_update_accepts_positive_budget_usd() -> None:
|
||||
budget = 3.0
|
||||
assert TaskUpdate(budget_usd=budget).budget_usd == budget
|
||||
|
||||
|
||||
def test_task_update_rejects_zero_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="budget_usd"):
|
||||
TaskUpdate(budget_usd=0)
|
||||
|
||||
|
||||
def test_task_update_rejects_negative_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="budget_usd"):
|
||||
TaskUpdate(budget_usd=-1)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Project.monthly_budget_usd
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_project_defaults_monthly_budget_usd_to_none() -> None:
|
||||
assert _project().monthly_budget_usd is None
|
||||
|
||||
|
||||
def test_project_accepts_positive_monthly_budget_usd() -> None:
|
||||
cap = 100.0
|
||||
assert _project(monthly_budget_usd=cap).monthly_budget_usd == cap
|
||||
|
||||
|
||||
def test_project_rejects_zero_monthly_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="monthly_budget_usd"):
|
||||
_project(monthly_budget_usd=0)
|
||||
|
||||
|
||||
def test_project_rejects_negative_monthly_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="monthly_budget_usd"):
|
||||
_project(monthly_budget_usd=-5)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ProjectCreate.monthly_budget_usd
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_project_create_accepts_null_monthly_budget_usd() -> None:
|
||||
assert (
|
||||
ProjectCreate(
|
||||
name="P",
|
||||
slug="p",
|
||||
git_url="https://example.com/r.git",
|
||||
assigned_cell=Team.BACKEND,
|
||||
).monthly_budget_usd
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_project_create_rejects_zero_monthly_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="monthly_budget_usd"):
|
||||
ProjectCreate(
|
||||
name="P",
|
||||
slug="p",
|
||||
git_url="https://example.com/r.git",
|
||||
assigned_cell=Team.BACKEND,
|
||||
monthly_budget_usd=0,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ProjectUpdate.monthly_budget_usd
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_project_update_accepts_null_monthly_budget_usd() -> None:
|
||||
assert ProjectUpdate(monthly_budget_usd=None).monthly_budget_usd is None
|
||||
|
||||
|
||||
def test_project_update_accepts_positive_monthly_budget_usd() -> None:
|
||||
cap = 50.0
|
||||
assert ProjectUpdate(monthly_budget_usd=cap).monthly_budget_usd == cap
|
||||
|
||||
|
||||
def test_project_update_rejects_zero_monthly_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="monthly_budget_usd"):
|
||||
ProjectUpdate(monthly_budget_usd=0)
|
||||
|
||||
|
||||
def test_project_update_rejects_negative_monthly_budget_usd() -> None:
|
||||
with pytest.raises(ValidationError, match="monthly_budget_usd"):
|
||||
ProjectUpdate(monthly_budget_usd=-5)
|
||||
@@ -0,0 +1,195 @@
|
||||
"""Codex (OPENAI) and Gemini (GEMINI) are V1 delivery-roles-only — neither has
|
||||
an interactive-session driver image (unlike GROK's dedicated
|
||||
GROK_PROMPTER_IMAGE / GROK_SECRETARY_IMAGE). Routing either to the persistent
|
||||
Intake/Secretary agent must refuse loudly instead of silently falling through
|
||||
to the plain Claude SDK-driver image with a mismatched provider env.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from roboco.models.base import ModelProvider
|
||||
from roboco.runtime.orchestrator import (
|
||||
_INTERACTIVE_UNSUPPORTED_PROVIDERS,
|
||||
INTAKE_AGENT_ID,
|
||||
SECRETARY_AGENT_ID,
|
||||
AgentOrchestrator,
|
||||
_reject_interactive_unsupported_provider,
|
||||
)
|
||||
from roboco.services import prompter_live
|
||||
from roboco.services.llm import (
|
||||
INTERACTIVE_AGENT_SLUGS,
|
||||
INTERACTIVE_UNSUPPORTED_PROVIDERS,
|
||||
)
|
||||
|
||||
|
||||
def _make_minimal_orchestrator() -> AgentOrchestrator:
|
||||
with patch.object(AgentOrchestrator, "__init__", return_value=None):
|
||||
orch = AgentOrchestrator.__new__(AgentOrchestrator)
|
||||
orch._instances = {}
|
||||
orch._bg_tasks = set()
|
||||
orch._running = True
|
||||
orch._intake_spawn_lock = asyncio.Lock()
|
||||
orch._secretary_spawn_lock = asyncio.Lock()
|
||||
return orch
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _fresh_registry() -> Any:
|
||||
prev = prompter_live._RegistryHolder.instance
|
||||
prompter_live._RegistryHolder.instance = prompter_live.PrompterLiveRegistry()
|
||||
yield
|
||||
prompter_live._RegistryHolder.instance = prev
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unit-level: the pure guard function itself.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestRejectInteractiveUnsupportedProvider:
|
||||
def test_guard_set_matches_the_resolver_exemption_set(self) -> None:
|
||||
"""The orchestrator's literal must track the resolver's canonical
|
||||
tuple (kept separate to avoid a runtime import cycle)."""
|
||||
assert tuple(_INTERACTIVE_UNSUPPORTED_PROVIDERS) == tuple(
|
||||
INTERACTIVE_UNSUPPORTED_PROVIDERS
|
||||
)
|
||||
|
||||
def test_resolver_slugs_match_the_orchestrator_agent_ids(self) -> None:
|
||||
"""The resolver's exemption must cover exactly the two interactive
|
||||
agents the orchestrator spawns — a renamed id would silently
|
||||
un-exempt a chat."""
|
||||
assert set(INTERACTIVE_AGENT_SLUGS) == {INTAKE_AGENT_ID, SECRETARY_AGENT_ID}
|
||||
|
||||
@pytest.mark.parametrize("provider", [ModelProvider.OPENAI, ModelProvider.GEMINI])
|
||||
def test_raises_for_delivery_only_providers(self, provider: ModelProvider) -> None:
|
||||
with pytest.raises(RuntimeError, match="delivery-roles-only"):
|
||||
_reject_interactive_unsupported_provider(INTAKE_AGENT_ID, provider)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"provider",
|
||||
[
|
||||
ModelProvider.ANTHROPIC,
|
||||
ModelProvider.GROK,
|
||||
ModelProvider.OLLAMA_CLOUD,
|
||||
ModelProvider.LOCAL,
|
||||
],
|
||||
)
|
||||
def test_passes_for_interactive_capable_providers(
|
||||
self, provider: ModelProvider
|
||||
) -> None:
|
||||
_reject_interactive_unsupported_provider(INTAKE_AGENT_ID, provider) # no raise
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Intake spawn refusal — surfaces on the relay, container never launched.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestIntakeSpawnRefusesDeliveryOnlyProvider:
|
||||
@pytest.mark.parametrize("provider", [ModelProvider.OPENAI, ModelProvider.GEMINI])
|
||||
@pytest.mark.asyncio
|
||||
async def test_refuses_before_any_container_work(
|
||||
self, monkeypatch: pytest.MonkeyPatch, provider: ModelProvider
|
||||
) -> None:
|
||||
orch = _make_minimal_orchestrator()
|
||||
|
||||
async def _clone(*_a: Any, **_k: Any) -> tuple[str, list[str]]:
|
||||
return "/data/workspaces/roboco/board/intake-1", ["/cwd"]
|
||||
|
||||
async def _route(_aid: str) -> Any:
|
||||
return SimpleNamespace(
|
||||
provider_type=provider,
|
||||
model_name="whatever",
|
||||
base_url=None,
|
||||
auth_token=None,
|
||||
)
|
||||
|
||||
run_calls: list[list[str]] = []
|
||||
|
||||
async def _run(cmd: list[str]) -> str:
|
||||
run_calls.append(cmd)
|
||||
return "containerid0123456789"
|
||||
|
||||
monkeypatch.setattr(orch, "_clone_intake_scope", _clone)
|
||||
monkeypatch.setattr(orch, "_resolve_agent_route", _route)
|
||||
monkeypatch.setattr(
|
||||
orch, "_generate_composed_prompt", lambda *_a, **_k: Path("/tmp/p.md")
|
||||
)
|
||||
monkeypatch.setattr(orch, "_run_container_cmd", _run)
|
||||
|
||||
registry = prompter_live.get_live_registry()
|
||||
pushed: list[tuple[str, dict[str, Any]]] = []
|
||||
closed: list[str] = []
|
||||
monkeypatch.setattr(registry, "push", lambda sid, ev: pushed.append((sid, ev)))
|
||||
monkeypatch.setattr(registry, "close", closed.append)
|
||||
registry.open("sess-refuse", INTAKE_AGENT_ID)
|
||||
|
||||
await orch._spawn_intake_container_guarded(
|
||||
"sess-refuse", project_slug="roboco", product_id=None, initial_message=None
|
||||
)
|
||||
|
||||
assert not run_calls # no container was ever launched
|
||||
assert len(pushed) == 1
|
||||
assert pushed[0][1]["kind"] == "error"
|
||||
assert "delivery-roles-only" in pushed[0][1]["text"]
|
||||
assert closed == ["sess-refuse"]
|
||||
assert INTAKE_AGENT_ID not in orch._instances
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Secretary spawn refusal — same shape, same guard.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestSecretarySpawnRefusesDeliveryOnlyProvider:
|
||||
@pytest.mark.parametrize("provider", [ModelProvider.OPENAI, ModelProvider.GEMINI])
|
||||
@pytest.mark.asyncio
|
||||
async def test_refuses_before_any_container_work(
|
||||
self, monkeypatch: pytest.MonkeyPatch, provider: ModelProvider
|
||||
) -> None:
|
||||
orch = _make_minimal_orchestrator()
|
||||
|
||||
async def _route(_aid: str) -> Any:
|
||||
return SimpleNamespace(
|
||||
provider_type=provider,
|
||||
model_name="whatever",
|
||||
base_url=None,
|
||||
auth_token=None,
|
||||
)
|
||||
|
||||
run_calls: list[list[str]] = []
|
||||
|
||||
async def _run(cmd: list[str]) -> str:
|
||||
run_calls.append(cmd)
|
||||
return "containerid0123456789"
|
||||
|
||||
monkeypatch.setattr(orch, "_resolve_agent_route", _route)
|
||||
monkeypatch.setattr(
|
||||
orch, "_generate_composed_prompt", lambda *_a, **_k: Path("/tmp/p.md")
|
||||
)
|
||||
monkeypatch.setattr(orch, "_run_container_cmd", _run)
|
||||
|
||||
registry = prompter_live.get_live_registry()
|
||||
pushed: list[tuple[str, dict[str, Any]]] = []
|
||||
closed: list[str] = []
|
||||
monkeypatch.setattr(registry, "push", lambda sid, ev: pushed.append((sid, ev)))
|
||||
monkeypatch.setattr(registry, "close", closed.append)
|
||||
registry.open("sess-sec-refuse", SECRETARY_AGENT_ID)
|
||||
|
||||
await orch._spawn_secretary_container_guarded(
|
||||
"sess-sec-refuse", initial_message=None
|
||||
)
|
||||
|
||||
assert not run_calls # no container was ever launched
|
||||
assert len(pushed) == 1
|
||||
assert pushed[0][1]["kind"] == "error"
|
||||
assert "delivery-roles-only" in pushed[0][1]["text"]
|
||||
assert closed == ["sess-sec-refuse"]
|
||||
assert SECRETARY_AGENT_ID not in orch._instances
|
||||
@@ -47,6 +47,7 @@ def _claim_task(
|
||||
last_heartbeat_at=None,
|
||||
active_claimant_id=None,
|
||||
orchestration_markers={},
|
||||
dev_notes=None,
|
||||
)
|
||||
|
||||
|
||||
@@ -229,6 +230,30 @@ async def test_inherit_conflict_notes_the_task() -> None:
|
||||
assert note is not None
|
||||
assert "a.py, b.py" in note
|
||||
assert "sync_branch" in note
|
||||
# The dev must actually SEE it — dev_notes rides evidence(), the marker
|
||||
# does not.
|
||||
assert task.dev_notes is not None
|
||||
assert "a.py, b.py" in task.dev_notes
|
||||
assert "[BASE INHERITANCE]" in task.dev_notes
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_inherit_merged_push_failed_notes_the_dev() -> None:
|
||||
svc = _service()
|
||||
task = _claim_task("feature/backend/AAA--BBB")
|
||||
proj_svc, git_svc, _ = _patched_deps(svc, {"status": "merged_push_failed"})
|
||||
|
||||
with (
|
||||
patch(
|
||||
"roboco.services.project.get_project_service",
|
||||
MagicMock(return_value=proj_svc),
|
||||
),
|
||||
patch("roboco.services.git.get_git_service", MagicMock(return_value=git_svc)),
|
||||
):
|
||||
await svc._inherit_upstream_base(task, uuid4())
|
||||
|
||||
assert task.dev_notes is not None
|
||||
assert "push to origin failed" in task.dev_notes
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
Reference in New Issue
Block a user