mirror of
https://github.com/rennf93/roboco.git
synced 2026-08-03 07:23:24 +02:00
feat(grok): reasoning-effort by role (cut grok-build cost on cheap roles)
grok-build-0.1 reasons heavily by default and reasoning bills at the output rate (a live "say ok" call emitted ~300 reasoning tokens, ~85% of its cost). Confirmed live that opencode's `--variant minimal` cuts reasoning ~54% (298 -> 136 tokens, same prompt). GrokProvider now picks reasoning effort by role: code-quality roles (developer, qa, pr_reviewer) keep full reasoning; coordination / docs / board roles (cell_pm, main_pm, documenter, product_owner, head_marketing, auditor, prompter, secretary) run "minimal". It's passed to opencode via the entrypoint's `--variant`. Operators can force one effort for ALL grok agents with the ROBOCO_GROK_REASONING_EFFORT env (minimal | high | max, or default/full). Tests cover the role map, the env override, and the spawn env wiring.
This commit is contained in:
@@ -20,6 +20,16 @@ python -m roboco.llm.providers.opencode_config
|
|||||||
# `< /dev/null` is REQUIRED: without a closed stdin, `opencode run` hangs after
|
# `< /dev/null` is REQUIRED: without a closed stdin, `opencode run` hangs after
|
||||||
# init in a headless / no-TTY environment (it blocks waiting on stdin). Verified
|
# init in a headless / no-TTY environment (it blocks waiting on stdin). Verified
|
||||||
# live — closing stdin lets the run proceed to the model call and exit cleanly.
|
# live — closing stdin lets the run proceed to the model call and exit cleanly.
|
||||||
|
#
|
||||||
|
# Reasoning effort: GrokProvider sets ROBOCO_GROK_VARIANT per role (e.g.
|
||||||
|
# "minimal" for coordination/docs roles to cut reasoning cost). Absent =
|
||||||
|
# opencode default (full reasoning).
|
||||||
|
variant_arg=()
|
||||||
|
if [ -n "${ROBOCO_GROK_VARIANT:-}" ]; then
|
||||||
|
variant_arg=(--variant "$ROBOCO_GROK_VARIANT")
|
||||||
|
fi
|
||||||
|
|
||||||
exec opencode run \
|
exec opencode run \
|
||||||
--model "xai/${ROBOCO_AGENT_MODEL:-grok-build-0.1}" \
|
--model "xai/${ROBOCO_AGENT_MODEL:-grok-build-0.1}" \
|
||||||
|
"${variant_arg[@]}" \
|
||||||
-- "${ROBOCO_INITIAL_PROMPT:-}" < /dev/null
|
-- "${ROBOCO_INITIAL_PROMPT:-}" < /dev/null
|
||||||
|
|||||||
@@ -41,6 +41,7 @@ import dataclasses
|
|||||||
import os
|
import os
|
||||||
from typing import TYPE_CHECKING, Protocol
|
from typing import TYPE_CHECKING, Protocol
|
||||||
|
|
||||||
|
from roboco.agents_config import get_agent_role
|
||||||
from roboco.llm.providers._docker import container_running, stop_container
|
from roboco.llm.providers._docker import container_running, stop_container
|
||||||
from roboco.llm.providers.base import AgentProvider, ProviderError, SpawnResult
|
from roboco.llm.providers.base import AgentProvider, ProviderError, SpawnResult
|
||||||
|
|
||||||
@@ -67,6 +68,41 @@ _SYSTEM_PROMPT_IN_CONTAINER = "/app/system-prompt.md"
|
|||||||
# the OpenAI-protocol CLI via the recognised `--tools` flag (not --allowed-tools).
|
# the OpenAI-protocol CLI via the recognised `--tools` flag (not --allowed-tools).
|
||||||
_DEFAULT_TOOLS = "Read,Write,Edit,Bash,Grep,Glob,TodoWrite"
|
_DEFAULT_TOOLS = "Read,Write,Edit,Bash,Grep,Glob,TodoWrite"
|
||||||
|
|
||||||
|
# Reasoning effort by role. grok-build-0.1 reasons heavily by default, and
|
||||||
|
# reasoning bills at the output rate — it dominates cost (a live "say ok" call
|
||||||
|
# emitted ~300 reasoning tokens). Code-quality roles (developer, qa, pr_reviewer)
|
||||||
|
# keep full reasoning; coordination / docs / board roles run "minimal" (a live
|
||||||
|
# test cut reasoning ~54% with no quality cost for that work). opencode applies
|
||||||
|
# this via its `--variant` flag. Operators can force one effort for ALL grok
|
||||||
|
# agents with the ROBOCO_GROK_REASONING_EFFORT env on the orchestrator
|
||||||
|
# (value "minimal" | "high" | "max", or "default"/"full" to use full reasoning).
|
||||||
|
_MINIMAL_REASONING_ROLES = frozenset(
|
||||||
|
{
|
||||||
|
"cell_pm",
|
||||||
|
"main_pm",
|
||||||
|
"documenter",
|
||||||
|
"product_owner",
|
||||||
|
"head_marketing",
|
||||||
|
"auditor",
|
||||||
|
"prompter",
|
||||||
|
"secretary",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
_FULL_REASONING_OVERRIDES = frozenset({"default", "full", "none", ""})
|
||||||
|
|
||||||
|
|
||||||
|
def _reasoning_effort_for(agent_id: str) -> str | None:
|
||||||
|
"""Resolve the opencode --variant reasoning effort for an agent.
|
||||||
|
|
||||||
|
Returns ``None`` to use opencode's default (full) reasoning. A global
|
||||||
|
override env wins over the per-role default.
|
||||||
|
"""
|
||||||
|
override = os.environ.get("ROBOCO_GROK_REASONING_EFFORT", "").strip()
|
||||||
|
if override:
|
||||||
|
return None if override.lower() in _FULL_REASONING_OVERRIDES else override
|
||||||
|
role = get_agent_role(agent_id) or ""
|
||||||
|
return "minimal" if role in _MINIMAL_REASONING_ROLES else None
|
||||||
|
|
||||||
|
|
||||||
def _container_name(agent_id: str) -> str:
|
def _container_name(agent_id: str) -> str:
|
||||||
return f"roboco-agent-{agent_id}"
|
return f"roboco-agent-{agent_id}"
|
||||||
@@ -188,6 +224,10 @@ class GrokProvider(AgentProvider):
|
|||||||
# Reused as the generic agent session id so the transcript stays
|
# Reused as the generic agent session id so the transcript stays
|
||||||
# locatable at finalize, exactly as on the Claude Code path.
|
# locatable at finalize, exactly as on the Claude Code path.
|
||||||
cmd.extend(["-e", f"ROBOCO_AGENT_SESSION_ID={config.claude_session_id}"])
|
cmd.extend(["-e", f"ROBOCO_AGENT_SESSION_ID={config.claude_session_id}"])
|
||||||
|
# Reasoning effort (opencode --variant) by role; omitted = full reasoning.
|
||||||
|
variant = _reasoning_effort_for(config.agent_id)
|
||||||
|
if variant:
|
||||||
|
cmd.extend(["-e", f"ROBOCO_GROK_VARIANT={variant}"])
|
||||||
|
|
||||||
async def stop(self, instance_id: str, graceful: bool = True) -> None:
|
async def stop(self, instance_id: str, graceful: bool = True) -> None:
|
||||||
await stop_container(instance_id, graceful)
|
await stop_container(instance_id, graceful)
|
||||||
|
|||||||
@@ -23,19 +23,21 @@ from roboco.llm.providers import (
|
|||||||
ProviderRegistry,
|
ProviderRegistry,
|
||||||
SpawnResult,
|
SpawnResult,
|
||||||
)
|
)
|
||||||
|
from roboco.llm.providers.grok import _reasoning_effort_for
|
||||||
from roboco.models.base import ModelProvider
|
from roboco.models.base import ModelProvider
|
||||||
from roboco.models.runtime import OrchestratorAgentConfig
|
from roboco.models.runtime import OrchestratorAgentConfig
|
||||||
|
|
||||||
|
|
||||||
def _config(
|
def _config(
|
||||||
*,
|
*,
|
||||||
|
agent_id: str = "be-dev-1",
|
||||||
provider_type: str = "grok",
|
provider_type: str = "grok",
|
||||||
provider_base_url: str | None = "https://api.x.ai/v1",
|
provider_base_url: str | None = "https://api.x.ai/v1",
|
||||||
provider_auth_token: str | None = "xai-secret-key",
|
provider_auth_token: str | None = "xai-secret-key",
|
||||||
mcp_config_path: Path | None = Path("/host/mcp-configs/be-dev-1.json"),
|
mcp_config_path: Path | None = Path("/host/mcp-configs/be-dev-1.json"),
|
||||||
) -> OrchestratorAgentConfig:
|
) -> OrchestratorAgentConfig:
|
||||||
return OrchestratorAgentConfig(
|
return OrchestratorAgentConfig(
|
||||||
agent_id="be-dev-1",
|
agent_id=agent_id,
|
||||||
blueprint_path=Path("/app/system-prompt.md"),
|
blueprint_path=Path("/app/system-prompt.md"),
|
||||||
model="grok-build-0.1",
|
model="grok-build-0.1",
|
||||||
mcp_config_path=mcp_config_path,
|
mcp_config_path=mcp_config_path,
|
||||||
@@ -243,6 +245,52 @@ async def test_grok_spawn_raises_on_docker_failure() -> None:
|
|||||||
await provider.spawn(_config())
|
await provider.spawn(_config())
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Reasoning effort by role
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_reasoning_effort_full_for_code_roles() -> None:
|
||||||
|
# developer / qa / pr_reviewer keep full reasoning (no variant).
|
||||||
|
assert _reasoning_effort_for("be-dev-1") is None
|
||||||
|
assert _reasoning_effort_for("be-qa") is None
|
||||||
|
assert _reasoning_effort_for("pr-reviewer-1") is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_reasoning_effort_minimal_for_coordination_roles() -> None:
|
||||||
|
for slug in ("be-pm", "main-pm", "be-doc", "auditor", "product-owner"):
|
||||||
|
assert _reasoning_effort_for(slug) == "minimal", slug
|
||||||
|
|
||||||
|
|
||||||
|
def test_reasoning_effort_override(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
monkeypatch.setenv("ROBOCO_GROK_REASONING_EFFORT", "max")
|
||||||
|
assert _reasoning_effort_for("be-dev-1") == "max" # override wins over role
|
||||||
|
monkeypatch.setenv("ROBOCO_GROK_REASONING_EFFORT", "default")
|
||||||
|
assert _reasoning_effort_for("be-pm") is None # "default" => full reasoning
|
||||||
|
|
||||||
|
|
||||||
|
async def test_grok_spawn_sets_variant_for_minimal_role() -> None:
|
||||||
|
host = _FakeHost()
|
||||||
|
provider = GrokProvider(host)
|
||||||
|
with patch(
|
||||||
|
"asyncio.create_subprocess_exec", AsyncMock(return_value=_proc())
|
||||||
|
) as exec_mock:
|
||||||
|
await provider.spawn(_config(agent_id="be-pm")) # cell_pm -> minimal
|
||||||
|
cmd = list(exec_mock.call_args.args)
|
||||||
|
assert "ROBOCO_GROK_VARIANT=minimal" in cmd
|
||||||
|
|
||||||
|
|
||||||
|
async def test_grok_spawn_no_variant_for_dev_role() -> None:
|
||||||
|
host = _FakeHost()
|
||||||
|
provider = GrokProvider(host)
|
||||||
|
with patch(
|
||||||
|
"asyncio.create_subprocess_exec", AsyncMock(return_value=_proc())
|
||||||
|
) as exec_mock:
|
||||||
|
await provider.spawn(_config(agent_id="be-dev-1")) # developer -> full
|
||||||
|
cmd = list(exec_mock.call_args.args)
|
||||||
|
assert not any(c.startswith("ROBOCO_GROK_VARIANT=") for c in cmd)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# ClaudeCodeProvider
|
# ClaudeCodeProvider
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user