From b4b66f8ecbe3cb6e8d49297e97764505fa0b8ecb Mon Sep 17 00:00:00 2001 From: npub1223z34hd7vtwc6qj4s7flsxkj644nlre2nthu7lrrmkumhu3xddsrx9r6w <52a228d6edf316ec6812ac3c9fc0d696ab59fc7954d77e7be31eedcddf91335b@sprout-oss.stage.blox.sqprod.co> Date: Mon, 6 Jul 2026 22:39:20 -0700 Subject: [PATCH] feat(agent): emit friendly tool titles with async fast-pass summaries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port the Goose two-stage tool-title pattern for individual tool calls. emit_pending now sends a deterministic friendly title (server: tool plus a detail excerpt from path/command/query-style args, 60-char first-line cap, multibyte-safe) alongside an explicit toolName so the desktop never loses tool identity. For runnable tools a fire-and-forget fast pass asks the LLM (64 output tokens, 300-char args excerpt, one retry) for a friendlier title and publishes it as a title-only tool_call_update tagged _meta.buzz.toolSummary — deliberately without a status field so it can never clobber pending/executing/terminal state. Cancel-aware via the session watch channel; silent on failure. Kill switch BUZZ_AGENT_NO_TOOL_SUMMARY=1 disables the fast pass; model override via BUZZ_AGENT_TOOL_SUMMARY_MODEL. The regression harness sets the kill switch since those tests queue exact LLM response sequences and count requests; golden_transcripts.rs carries the explicit summary-path coverage (pending title + toolName, title-only async update, kill switch). No chain summaries in this slice. Co-authored-by: Taylor Ho Signed-off-by: Taylor Ho --- crates/buzz-agent/src/agent.rs | 260 +++++++++++++++++- crates/buzz-agent/src/config.rs | 13 + crates/buzz-agent/src/llm.rs | 2 + crates/buzz-agent/tests/golden_transcripts.rs | 195 +++++++++++++ crates/buzz-agent/tests/regressions.rs | 7 +- 5 files changed, 474 insertions(+), 3 deletions(-) diff --git a/crates/buzz-agent/src/agent.rs b/crates/buzz-agent/src/agent.rs index 730e87b2e..47dd56742 100644 --- a/crates/buzz-agent/src/agent.rs +++ b/crates/buzz-agent/src/agent.rs @@ -28,7 +28,7 @@ pub struct RunCtx<'a> { pub effective_model: &'a str, pub session_id: &'a str, pub system_prompt: &'a str, - pub llm: &'a Llm, + pub llm: &'a Arc, pub mcp: &'a Arc, /// Skills discovered at session creation; used by the built-in `load_skill` tool. pub skills: &'a [SkillEntry], @@ -333,6 +333,20 @@ impl RunCtx<'_> { results[idx] = Some(synthetic_tool_result(call, err)); continue; } + // Second stage of the friendly-title flow: fire-and-forget fast + // summarization for real (runnable) tool calls only — fast-failed + // and built-in calls keep their deterministic title. + if self.cfg.tool_summary_enabled { + spawn_tool_summary( + Arc::clone(self.llm), + self.cfg.clone(), + self.effective_model.to_owned(), + self.wire.clone(), + self.session_id.to_owned(), + call.clone(), + self.cancel.clone(), + ); + } runnable.push(idx); } @@ -557,7 +571,11 @@ async fn emit_pending(wire: &WireSender, sid: &str, call: &ToolCall) { json!({ "sessionUpdate": "tool_call", "toolCallId": call.provider_id, - "title": call.name, + // Deterministic friendly title, available immediately. The + // exact tool identity rides alongside in `toolName` so + // clients never lose it to the friendlier phrasing. + "title": friendly_tool_title(&call.name, &call.arguments), + "toolName": call.name, "kind": "other", "status": "pending", "rawInput": call.arguments, @@ -567,6 +585,175 @@ async fn emit_pending(wire: &WireSender, sid: &str, call: &ToolCall) { .await; } +/// Format a qualified tool name (`server__tool`) as a human-readable base +/// label. Mirrors goose's `format_tool_name`. +fn format_tool_name(tool_name: &str) -> String { + if let Some((server, tool)) = tool_name.split_once("__") { + format!("{}: {}", server.replace('_', " "), tool.replace('_', " ")) + } else { + tool_name.replace('_', " ") + } +} + +/// Build a short deterministic title from the tool name plus the most useful +/// argument value (file path, command, query, url, etc.). Port of goose's +/// `summarize_tool_call` fallback-title builder. +fn friendly_tool_title(tool_name: &str, arguments: &serde_json::Value) -> String { + const DETAIL_KEYS: [&str; 9] = [ + "path", "file", "command", "query", "url", "uri", "name", "pattern", "source", + ]; + const MAX_DETAIL_CHARS: usize = 60; + + let base = format_tool_name(tool_name); + let detail = arguments.as_object().and_then(|obj| { + for key in &DETAIL_KEYS { + if let Some(v) = obj.get(*key) { + let s = match v { + serde_json::Value::String(s) => s.clone(), + other => other.to_string(), + }; + if !s.is_empty() { + let first_line = s.lines().next().unwrap_or(&s); + let mut out: String = first_line.chars().take(MAX_DETAIL_CHARS).collect(); + if first_line.chars().count() > MAX_DETAIL_CHARS { + out.push('…'); + } + return Some(out); + } + } + } + None + }); + match detail { + Some(d) => format!("{base} · {d}"), + None => base, + } +} + +const TOOL_SUMMARY_SYSTEM_PROMPT: &str = + "Summarize this tool call in a short lowercase phrase (3-8 words). \ + No punctuation. No quotes. Examples: reading project configuration, \ + checking network connectivity, listing files in src directory"; + +/// Max output tokens for a tool-title summary — a short phrase, not prose. +const TOOL_SUMMARY_MAX_OUTPUT_TOKENS: u32 = 64; + +/// Cap on the serialized-arguments excerpt sent to the fast model. +const TOOL_SUMMARY_MAX_ARGS_CHARS: usize = 300; + +/// Cap on the accepted summary phrase; longer responses are rejected as +/// non-conforming rather than truncated mid-thought. +const TOOL_SUMMARY_MAX_TITLE_CHARS: usize = 80; + +/// Normalize a fast-model response into a row-label phrase, or `None` when +/// the response is unusable (empty, multi-paragraph rambling, oversized). +fn sanitize_tool_summary(raw: &str) -> Option { + let first_line = raw.trim().lines().next()?.trim(); + let cleaned = first_line.trim_matches(|c| c == '"' || c == '\'' || c == '`'); + if cleaned.is_empty() || cleaned.chars().count() > TOOL_SUMMARY_MAX_TITLE_CHARS { + return None; + } + Some(cleaned.to_string()) +} + +/// Fire-and-forget fast summarization pass for one tool call (goose's +/// two-stage title pattern). Publishes a title-only `tool_call_update` +/// tagged `_meta.buzz.toolSummary` — deliberately no `status` field, so a +/// late summary can never clobber pending/executing/terminal state. Failures +/// are silent: the deterministic title from `emit_pending` stays. +fn spawn_tool_summary( + llm: Arc, + cfg: Config, + model: String, + wire: WireSender, + sid: String, + call: ToolCall, + cancel: watch::Receiver, +) { + tokio::spawn(async move { + let args_json = { + let s = call.arguments.to_string(); + if s.chars().count() > TOOL_SUMMARY_MAX_ARGS_CHARS { + let mut t: String = s.chars().take(TOOL_SUMMARY_MAX_ARGS_CHARS).collect(); + t.push('…'); + t + } else { + s + } + }; + let user_prompt = format!("Tool: {}\nArguments: {args_json}", call.name); + let effective_model = cfg.tool_summary_model.as_deref().unwrap_or(&model); + + // The fast model occasionally returns an empty/errored response under + // load. One retry with a short backoff recovers the common cases. + let mut title: Option = None; + for attempt in 0..2 { + if *cancel.borrow() { + return; + } + match llm + .summarize( + &cfg, + TOOL_SUMMARY_SYSTEM_PROMPT, + &user_prompt, + TOOL_SUMMARY_MAX_OUTPUT_TOKENS, + effective_model, + ) + .await + { + Ok(s) => { + if let Some(clean) = sanitize_tool_summary(&s) { + title = Some(clean); + break; + } + if attempt == 0 { + tracing::debug!( + "tool summary: empty/unusable response for {} ({}), retrying once", + call.provider_id, + call.name + ); + tokio::time::sleep(std::time::Duration::from_millis(150)).await; + } + } + Err(e) => { + if attempt == 0 { + tracing::debug!( + "tool summary: fast pass errored for {} ({}): {e}, retrying once", + call.provider_id, + call.name + ); + tokio::time::sleep(std::time::Duration::from_millis(150)).await; + } else { + tracing::debug!( + "tool summary: fast pass errored for {} ({}) after retry: {e}", + call.provider_id, + call.name + ); + } + } + } + } + let Some(title) = title else { return }; + if *cancel.borrow() { + return; + } + wire::send( + &wire, + wire::session_update( + &sid, + json!({ + "sessionUpdate": "tool_call_update", + "toolCallId": call.provider_id, + "title": title, + "toolName": call.name, + "_meta": { "buzz": { "toolSummary": true } }, + }), + ), + ) + .await; + }); +} + async fn emit_in_progress(wire: &WireSender, sid: &str, call: &ToolCall) { wire::send( wire, @@ -744,3 +931,72 @@ fn map_stop(p: ProviderStop) -> StopReason { ProviderStop::Refusal => StopReason::Refusal, } } + +#[cfg(test)] +mod tests { + use super::{friendly_tool_title, sanitize_tool_summary}; + use serde_json::json; + + #[test] + fn friendly_title_formats_qualified_name_with_detail() { + assert_eq!( + friendly_tool_title("developer__shell", &json!({ "command": "git status" })), + "developer: shell · git status" + ); + } + + #[test] + fn friendly_title_prefers_path_over_later_keys() { + assert_eq!( + friendly_tool_title( + "fs__read_file", + &json!({ "name": "x", "path": "/tmp/a.txt" }) + ), + "fs: read file · /tmp/a.txt" + ); + } + + #[test] + fn friendly_title_without_useful_args_is_just_the_name() { + assert_eq!(friendly_tool_title("do_thing", &json!({})), "do thing"); + assert_eq!(friendly_tool_title("do_thing", &json!(null)), "do thing"); + } + + #[test] + fn friendly_title_truncates_long_first_line_only() { + let long = "x".repeat(200); + let title = friendly_tool_title("t", &json!({ "command": format!("{long}\nsecond") })); + assert!(title.ends_with('…')); + // "t · " + 60 chars + ellipsis. + assert_eq!(title.chars().count(), 4 + 60 + 1); + assert!(!title.contains("second")); + } + + #[test] + fn friendly_title_handles_multibyte_without_panicking() { + let title = friendly_tool_title("t", &json!({ "query": "héllo wörld 🚀".repeat(20) })); + assert!(title.starts_with("t · ")); + } + + #[test] + fn sanitize_accepts_short_phrase_and_strips_quotes() { + assert_eq!( + sanitize_tool_summary("\"checking repository state\"\n"), + Some("checking repository state".to_string()) + ); + } + + #[test] + fn sanitize_rejects_empty_and_oversized() { + assert_eq!(sanitize_tool_summary(" \n "), None); + assert_eq!(sanitize_tool_summary(&"x".repeat(200)), None); + } + + #[test] + fn sanitize_keeps_only_first_line() { + assert_eq!( + sanitize_tool_summary("reading config\nextra rambling"), + Some("reading config".to_string()) + ); + } +} diff --git a/crates/buzz-agent/src/config.rs b/crates/buzz-agent/src/config.rs index 606edf2d2..0eefe266d 100644 --- a/crates/buzz-agent/src/config.rs +++ b/crates/buzz-agent/src/config.rs @@ -637,6 +637,15 @@ pub struct Config { /// Thinking/reasoning effort level. `None` = use provider default (no /// thinking config sent). Set via `BUZZ_AGENT_THINKING_EFFORT`. pub thinking_effort: Option, + /// Per-tool friendly title summarization: after emitting a tool call, + /// spawn a fast LLM pass that publishes a short human phrase as a + /// title-only `tool_call_update`. Default on; disable via + /// `BUZZ_AGENT_NO_TOOL_SUMMARY=1`. + pub tool_summary_enabled: bool, + /// Optional model override for tool title summaries (a cheaper/faster + /// model than the session model). Set via `BUZZ_AGENT_TOOL_SUMMARY_MODEL`; + /// `None` falls back to the session's effective model. + pub tool_summary_model: Option, } impl Config { @@ -731,6 +740,8 @@ impl Config { hook_servers: parse_hook_servers_env("MCP_HOOK_SERVERS"), hints_enabled: parse_env("BUZZ_AGENT_NO_HINTS", 0u8)? == 0, thinking_effort: parse_thinking_effort(env("BUZZ_AGENT_THINKING_EFFORT").as_deref())?, + tool_summary_enabled: parse_env("BUZZ_AGENT_NO_TOOL_SUMMARY", 0u8)? == 0, + tool_summary_model: env("BUZZ_AGENT_TOOL_SUMMARY_MODEL"), }; cfg.validate()?; Ok(cfg) @@ -771,6 +782,8 @@ impl Config { hook_servers: HookServers::None, hints_enabled: false, thinking_effort: None, + tool_summary_enabled: false, + tool_summary_model: None, } } diff --git a/crates/buzz-agent/src/llm.rs b/crates/buzz-agent/src/llm.rs index 07c2071aa..672ef657a 100644 --- a/crates/buzz-agent/src/llm.rs +++ b/crates/buzz-agent/src/llm.rs @@ -1193,6 +1193,8 @@ mod tests { openai_api: OpenAiApi::Chat, hints_enabled: true, thinking_effort: None, + tool_summary_enabled: false, + tool_summary_model: None, } } diff --git a/crates/buzz-agent/tests/golden_transcripts.rs b/crates/buzz-agent/tests/golden_transcripts.rs index 4ac350346..ef9d363ea 100644 --- a/crates/buzz-agent/tests/golden_transcripts.rs +++ b/crates/buzz-agent/tests/golden_transcripts.rs @@ -808,3 +808,198 @@ async fn test_cancel_notification_no_reply() { h.shutdown().await; } + +/// Per-tool friendly titles, stage 1: the immediate `tool_call` (pending) +/// update must carry the deterministic friendly title plus the exact tool +/// identity in `toolName`, and receipts in `rawInput`. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn test_pending_tool_call_carries_friendly_title_and_tool_name() { + let url = spawn_fake_llm(vec![ + openai_tool_call( + "call_t1", + "fake__do_thing", + json!({ "command": "git status" }), + ), + openai_text("done"), + ]) + .await; + let mut h = Harness::spawn(&[("OPENAI_COMPAT_BASE_URL", &url)]).await; + + let sid = handshake(&mut h).await; + let p = h + .send( + "session/prompt", + json!({ + "sessionId": sid, + "prompt": [{ "type": "text", "text": "use the tool" }], + }), + ) + .await; + + let pending = h + .recv_until(|v| { + v.get("method") == Some(&json!("session/update")) + && v["params"]["update"]["sessionUpdate"] == "tool_call" + }) + .await; + let update = &pending["params"]["update"]; + assert_eq!(update["title"], "fake: do thing · git status"); + assert_eq!(update["toolName"], "fake__do_thing"); + assert_eq!(update["status"], "pending"); + assert_eq!(update["rawInput"]["command"], "git status"); + + let final_resp = h.recv_for_id(p).await; + assert_eq!(final_resp["result"]["stopReason"], "end_turn"); + h.shutdown().await; +} + +/// Per-tool friendly titles, stage 2: the async fast pass publishes a +/// title-only `tool_call_update` tagged `_meta.buzz.toolSummary` with NO +/// status field, so it can never regress pending/executing/terminal state. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn test_async_tool_summary_publishes_title_only_update() { + // Queue: round-1 tool call, then the fast summary response (the tool is + // delayed 1s so the summary request reaches the fake LLM first), then the + // round-2 final text. + let url = spawn_fake_llm(vec![ + openai_tool_call( + "call_s1", + "fake__tool_0", + json!({ "command": "git status" }), + ), + openai_text("checking repository state"), + openai_text("done"), + ]) + .await; + let mut h = Harness::spawn(&[("OPENAI_COMPAT_BASE_URL", &url)]).await; + + let init_id = h + .send( + "initialize", + json!({ "protocolVersion": 2, "clientCapabilities": {} }), + ) + .await; + let _ = h.recv_for_id(init_id).await; + let fake_mcp = env!("CARGO_BIN_EXE_fake-mcp"); + let new_id = h + .send( + "session/new", + json!({ + "cwd": "/tmp", + "mcpServers": [{ + "name": "fake", + "command": fake_mcp, + "args": [], + "env": [ + { "name": "FAKE_MCP_TOOL_COUNT", "value": "1" }, + { "name": "FAKE_MCP_TOOL_DELAY", "value": "1" }, + ], + }], + }), + ) + .await; + let new = h.recv_for_id(new_id).await; + let sid = new["result"]["sessionId"].as_str().unwrap().to_owned(); + + let p = h + .send( + "session/prompt", + json!({ + "sessionId": sid, + "prompt": [{ "type": "text", "text": "use the tool" }], + }), + ) + .await; + + let summary = h + .recv_until(|v| { + v.get("method") == Some(&json!("session/update")) + && v["params"]["update"]["sessionUpdate"] == "tool_call_update" + && v["params"]["update"]["_meta"]["buzz"]["toolSummary"] == json!(true) + }) + .await; + let update = &summary["params"]["update"]; + assert_eq!(update["toolCallId"], "call_s1"); + assert_eq!(update["title"], "checking repository state"); + assert_eq!(update["toolName"], "fake__tool_0"); + assert!( + update.get("status").is_none(), + "summary update must not carry a status: {update}" + ); + + let final_resp = h.recv_for_id(p).await; + assert_eq!(final_resp["result"]["stopReason"], "end_turn"); + h.shutdown().await; +} + +/// Kill switch: BUZZ_AGENT_NO_TOOL_SUMMARY=1 must suppress the fast pass +/// entirely — no summary update on the wire, and no extra LLM request +/// (the canned-queue accounting would break the turn if one fired). +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn test_tool_summary_kill_switch() { + let url = spawn_fake_llm(vec![ + openai_tool_call("call_k1", "fake__tool_0", json!({ "command": "ls" })), + openai_text("done"), + ]) + .await; + let mut h = Harness::spawn(&[ + ("OPENAI_COMPAT_BASE_URL", &url), + ("BUZZ_AGENT_NO_TOOL_SUMMARY", "1"), + ]) + .await; + + let init_id = h + .send( + "initialize", + json!({ "protocolVersion": 2, "clientCapabilities": {} }), + ) + .await; + let _ = h.recv_for_id(init_id).await; + let fake_mcp = env!("CARGO_BIN_EXE_fake-mcp"); + let new_id = h + .send( + "session/new", + json!({ + "cwd": "/tmp", + "mcpServers": [{ + "name": "fake", + "command": fake_mcp, + "args": [], + "env": [{ "name": "FAKE_MCP_TOOL_COUNT", "value": "1" }], + }], + }), + ) + .await; + let new = h.recv_for_id(new_id).await; + let sid = new["result"]["sessionId"].as_str().unwrap().to_owned(); + + let p = h + .send( + "session/prompt", + json!({ + "sessionId": sid, + "prompt": [{ "type": "text", "text": "use the tool" }], + }), + ) + .await; + + // Drain everything up to the final response; no notification along the + // way may carry the toolSummary marker. If the fast pass had fired it + // would also have consumed the "done" response and broken the turn. + let mut saw_summary = false; + let final_resp = loop { + let v = h.recv().await; + if v["params"]["update"]["_meta"]["buzz"]["toolSummary"] == json!(true) { + saw_summary = true; + } + if v["id"] == json!(p) { + break v; + } + }; + assert!( + !saw_summary, + "kill switch did not suppress the summary pass" + ); + assert_eq!(final_resp["result"]["stopReason"], "end_turn"); + h.shutdown().await; +} diff --git a/crates/buzz-agent/tests/regressions.rs b/crates/buzz-agent/tests/regressions.rs index 3c2237ac9..d8841f157 100644 --- a/crates/buzz-agent/tests/regressions.rs +++ b/crates/buzz-agent/tests/regressions.rs @@ -102,7 +102,12 @@ impl Harness { .env("BUZZ_AGENT_LLM_TIMEOUT_SECS", "5") .env("BUZZ_AGENT_TOOL_TIMEOUT_SECS", "5") .env("BUZZ_AGENT_MAX_ROUNDS", "8") - .env("BUZZ_AGENT_MCP_INIT_TIMEOUT_SECS", "2"); + .env("BUZZ_AGENT_MCP_INIT_TIMEOUT_SECS", "2") + // These tests queue exact LLM response sequences and count + // requests; the async tool-summary fast pass would consume + // queued responses nondeterministically. Covered explicitly + // in golden_transcripts.rs instead. + .env("BUZZ_AGENT_NO_TOOL_SUMMARY", "1"); for (k, v) in extra { cmd.env(k, v); }