From a031fc96f5145b9ff85579b8ba72ffea59cfa958 Mon Sep 17 00:00:00 2001 From: Michael Neale Date: Tue, 4 Aug 2026 17:51:44 +1000 Subject: [PATCH] feat(mesh): live gossip topology, MeshLLM headline, inferred inbound work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Drops the relay/roster overlay from the topology in favour of a single source: the runtime's own gossip view. A node is drawn because we are connected to it right now, not because a status note is still inside its 120s freshness window. That window is why a departed device could sit on the graph as a solid, capacity-contributing dot for up to two minutes — a reconciliation problem this removes rather than referees. Community-roster ghosts go too: a member who never starts a node discloses no hardware, so they added a denominator and little else. New mesh_live_view command projects peers[] (label, state, capacity honoring the peer's own cap, models, rtt). Client-mode peers report vram_gb: 0, normalized to absent so the UI never prints "0 GB" for a machine that shares none. The relay snapshot is retained for exactly one job: the headline when no local runtime exists, where it is the only view available. Card headline is now "MeshLLM · 115 GB, 2 peers", with the participation hint below it: "You're sharing", "· serving another member", or the consuming split. Inbound work turns out to be derivable after all. mesh-llm exposes no counter for it (fronted_request_count merely aliases request_count), but sharing, with inflight > 0, while our own dispatch count stays flat between samples, means the work cannot be ours. inferInboundWork encodes that. It is sampled, so it can undercount a request that starts and ends between polls — it never over-claims, which is the direction that matters. Also corrects the inverted doc comments on MeshServingUsage, which still described the backwards model that caused the original card bug. UX only — no routing, admission, or lifecycle change. Signed-off-by: Michael Neale --- .../src-tauri/src/commands/mesh_live_view.rs | 31 +++ desktop/src-tauri/src/commands/mod.rs | 4 + desktop/src-tauri/src/lib.rs | 1 + desktop/src-tauri/src/mesh_llm/mod.rs | 16 ++ desktop/src-tauri/src/mesh_llm/peers.rs | 263 ++++++++++++++++++ desktop/src-tauri/src/mesh_llm/usage.rs | 39 ++- desktop/src-tauri/src/mesh_llm_stubs.rs | 5 + .../mesh-compute/hooks/useMeshLiveView.ts | 57 ++++ .../src/features/mesh-compute/meshActivity.ts | 133 +++++++++ .../mesh-compute/meshCardModel.test.mjs | 243 ++++++++-------- .../features/mesh-compute/meshCardModel.ts | 117 ++++---- .../mesh-compute/meshDetailModel.test.mjs | 244 ++++++++++++---- .../features/mesh-compute/meshDetailModel.ts | 140 +++++----- .../mesh-compute/ui/MeshDetailPopover.tsx | 50 ++-- .../mesh-compute/ui/MeshTopologyRadial.tsx | 186 +++++-------- .../ui/SidebarMeshComputeCard.tsx | 31 ++- desktop/src/shared/api/tauriMesh.ts | 38 +++ 17 files changed, 1172 insertions(+), 426 deletions(-) create mode 100644 desktop/src-tauri/src/commands/mesh_live_view.rs create mode 100644 desktop/src-tauri/src/mesh_llm/peers.rs create mode 100644 desktop/src/features/mesh-compute/hooks/useMeshLiveView.ts create mode 100644 desktop/src/features/mesh-compute/meshActivity.ts diff --git a/desktop/src-tauri/src/commands/mesh_live_view.rs b/desktop/src-tauri/src/commands/mesh_live_view.rs new file mode 100644 index 000000000..c03deefc8 --- /dev/null +++ b/desktop/src-tauri/src/commands/mesh_live_view.rs @@ -0,0 +1,31 @@ +//! Live gossip view of the mesh — the peers this machine's runtime is actually +//! connected to right now. +//! +//! Separate from `mesh_llm.rs` because that file sits at the 1000-line ceiling +//! enforced by `desktop/scripts/check-file-sizes.mjs`, and separate from +//! `mesh_snapshot.rs` because the two answer different questions: the snapshot +//! reports every member's last published note (valid 120s, readable with no +//! local node), while this reports live adjacency. See `mesh_llm/peers.rs` for +//! why both exist. + +use crate::app_state::AppState; +use crate::commands::CmdResult; +use crate::mesh_llm; +use tauri::State; + +/// This machine's live gossip view of the mesh: the peers our runtime is +/// actually connected to right now. +/// +/// `connected: false` with no peers means we are not participating, which is +/// distinct from "participating but alone". Never errors on absence — a stopped +/// node is a normal state, not a failure. +#[tauri::command] +pub async fn mesh_live_view(state: State<'_, AppState>) -> CmdResult { + let runtime = state.mesh_llm_runtime.lock().await; + match runtime.as_ref() { + // A runtime that exists but is mid-start has no gossip view yet; report + // "not connected" rather than surfacing a transient error. + Some(runtime) => Ok(runtime.live_view().await.unwrap_or_default()), + None => Ok(mesh_llm::MeshLiveView::default()), + } +} diff --git a/desktop/src-tauri/src/commands/mod.rs b/desktop/src-tauri/src/commands/mod.rs index e7c8783da..c95dbbcb1 100644 --- a/desktop/src-tauri/src/commands/mod.rs +++ b/desktop/src-tauri/src/commands/mod.rs @@ -31,6 +31,8 @@ mod media_gif; mod media_snapshot_png; mod media_transcode; #[cfg(feature = "mesh-llm")] +mod mesh_live_view; +#[cfg(feature = "mesh-llm")] pub(crate) mod mesh_llm; #[cfg(feature = "mesh-llm")] mod mesh_snapshot; @@ -88,6 +90,8 @@ pub use link_preview::*; pub use media::*; pub use media_download::*; #[cfg(feature = "mesh-llm")] +pub use mesh_live_view::*; +#[cfg(feature = "mesh-llm")] pub use mesh_llm::*; #[cfg(feature = "mesh-llm")] pub use mesh_snapshot::*; diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index 386c63577..7d0fe86da 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -774,6 +774,7 @@ pub fn run() { mesh_installed_models, mesh_model_catalog, mesh_snapshot, + mesh_live_view, update_managed_agent, discover_backend_providers, probe_backend_provider, diff --git a/desktop/src-tauri/src/mesh_llm/mod.rs b/desktop/src-tauri/src/mesh_llm/mod.rs index 743f8766d..87b3ac347 100644 --- a/desktop/src-tauri/src/mesh_llm/mod.rs +++ b/desktop/src-tauri/src/mesh_llm/mod.rs @@ -33,6 +33,9 @@ pub(crate) use recovery::{ MeshRuntimeRecovery, }; +mod peers; +pub use peers::{live_view_from_payload, MeshLiveView}; + mod usage; pub use usage::{serving_usage_from_payload, MeshServingUsage}; @@ -635,6 +638,19 @@ impl DesktopMeshRuntime { Ok(serving_usage_from_payload(&status.payload)) } + /// This machine's live gossip view of the mesh — who we are actually + /// connected to right now. Distinct from the relay snapshot, which reports + /// last-published notes valid for 120s. See [`peers`] for why both exist. + pub async fn live_view(&self) -> anyhow::Result { + let mut handle = self.handle.lock().await; + Self::promote_finished_startup(&mut handle).await; + let DesktopMeshHandle::Ready(ready) = &*handle else { + anyhow::bail!("mesh management status is not ready"); + }; + let status = ready.status().await?; + Ok(live_view_from_payload(&status.payload)) + } + pub async fn dial_endpoint_addr(&self, endpoint_addr: impl Into) -> anyhow::Result<()> { let endpoint_addr = endpoint_addr.into(); let validated = validate_advertised_endpoint(&endpoint_addr)?; diff --git a/desktop/src-tauri/src/mesh_llm/peers.rs b/desktop/src-tauri/src/mesh_llm/peers.rs new file mode 100644 index 000000000..451079a5c --- /dev/null +++ b/desktop/src-tauri/src/mesh_llm/peers.rs @@ -0,0 +1,263 @@ +use serde::{Deserialize, Serialize}; + +/// A node this machine's mesh runtime is **actually talking to right now**. +/// +/// ## Why this exists instead of reusing the relay snapshot +/// +/// Buzz has two views of the mesh and they answer different questions: +/// +/// - **Relay status notes** (`snapshot.rs`) — every member's last published +/// note, valid for 120s. Readable with no local node, so it is the only way +/// to see the pool *before* joining. But a note outlives the node that wrote +/// it: a machine that vanished 30s ago still looks present, and still counts +/// toward capacity. +/// - **Live peers** (this file) — the runtime's own gossip view. A node appears +/// here because our node is connected to it. No freshness window to guess at, +/// no reconciliation rule to invent. +/// +/// A topology view has to be trustworthy, so it draws *this*. Mixing the two +/// sources meant refereeing disagreements between them, which produced a graph +/// that showed devices gossip already knew were gone. +/// +/// The tradeoff, stated plainly: with no local node there are no peers, and the +/// relay snapshot remains the only available view. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] +#[serde(rename_all = "camelCase")] +pub struct MeshPeer { + /// Short node id from the runtime. Stable within a session. + pub id: String, + /// Human label — hostname as the peer reports it. + pub label: String, + /// Coarse role/state, normalized. See [`MeshPeerState`]. + pub state: MeshPeerState, + /// Shared AI memory in GB, honoring the peer's own `--max-vram` cap. + /// `None` when the peer reports nothing (a client-mode node reports 0, which + /// is normalized to `None` rather than displayed as "0 GB"). + pub capacity_gb: Option, + /// Models this peer is serving right now. Empty for a consuming peer. + pub models: Vec, + /// Round-trip latency in ms, when measured. + pub rtt_ms: Option, +} + +/// What a peer is doing, as *it* reports — never inferred from traffic. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MeshPeerState { + /// Advertising at least one model. + Serving, + /// Warming up — a host that has not advertised a model yet. + Loading, + /// Present and able to host, but advertising nothing. + Standby, + /// Client-mode: on the mesh to consume, not to contribute. + Consuming, +} + +/// This machine's live view of the mesh. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Default)] +#[serde(rename_all = "camelCase")] +pub struct MeshLiveView { + /// True when a local runtime is up. Distinguishes "no peers" (alone on the + /// mesh) from "unknown" (not participating, so nothing to report). + pub connected: bool, + /// This machine's own shared capacity, when serving. + pub self_capacity_gb: Option, + /// Peers currently connected, sharing first then by label. + pub peers: Vec, +} + +fn f64_at(value: &serde_json::Value, key: &str) -> Option { + value.get(key).and_then(serde_json::Value::as_f64) +} + +/// Normalize a peer's reported role/state into [`MeshPeerState`]. +/// +/// Trusts the peer's own `state`/`role` first and only falls back to inferring +/// from advertised models. A client-mode node that happens to list a model is +/// still consuming: role wins over model presence. +fn peer_state(value: &serde_json::Value, has_models: bool) -> MeshPeerState { + let raw = value + .get("state") + .or_else(|| value.get("role")) + .and_then(serde_json::Value::as_str) + .unwrap_or_default() + .to_ascii_lowercase(); + match raw.as_str() { + "client" => MeshPeerState::Consuming, + "serving" => MeshPeerState::Serving, + "loading" => MeshPeerState::Loading, + "standby" => MeshPeerState::Standby, + _ if has_models => MeshPeerState::Serving, + _ => MeshPeerState::Standby, + } +} + +fn string_list(value: &serde_json::Value, key: &str) -> Vec { + value + .get(key) + .and_then(serde_json::Value::as_array) + .map(|items| { + items + .iter() + .filter_map(serde_json::Value::as_str) + .map(str::to_string) + .collect() + }) + .unwrap_or_default() +} + +/// Pure extractor: project a raw SDK status payload into [`MeshLiveView`]. +/// +/// Defensive throughout (missing → absent) so an SDK shape change degrades to +/// "no peers shown" rather than an error. +pub fn live_view_from_payload(payload: &serde_json::Value) -> MeshLiveView { + let mut peers = payload + .get("peers") + .and_then(serde_json::Value::as_array) + .map(|items| { + items + .iter() + .filter_map(|peer| { + let id = peer + .get("id") + .and_then(serde_json::Value::as_str) + .unwrap_or_default() + .to_string(); + if id.is_empty() { + return None; + } + let mut models = string_list(peer, "serving_models"); + if models.is_empty() { + models = string_list(peer, "hosted_models"); + } + models.sort(); + models.dedup(); + let label = peer + .get("hostname") + .and_then(serde_json::Value::as_str) + .filter(|hostname| !hostname.is_empty()) + .map(str::to_string) + // Fall back to a short id rather than "Unknown": the id + // is at least actionable in logs. + .unwrap_or_else(|| id.chars().take(10).collect()); + Some(MeshPeer { + state: peer_state(peer, !models.is_empty()), + // 0 GB means "did not report" (client mode), not "has no + // memory" — keep it absent so the UI omits the figure. + capacity_gb: f64_at(peer, "vram_gb").filter(|gb| *gb > 0.0), + models, + rtt_ms: peer + .get("rtt_ms") + .and_then(serde_json::Value::as_u64) + .or_else(|| peer.get("latency_ms").and_then(serde_json::Value::as_u64)), + id, + label, + }) + }) + .collect::>() + }) + .unwrap_or_default(); + + peers.sort_by(|a, b| { + let serving = |peer: &MeshPeer| peer.state == MeshPeerState::Serving; + serving(b) + .cmp(&serving(a)) + .then_with(|| a.label.cmp(&b.label)) + }); + + MeshLiveView { + connected: true, + self_capacity_gb: f64_at(payload, "my_vram_gb").filter(|gb| *gb > 0.0), + peers, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn payload(peers: serde_json::Value) -> serde_json::Value { + serde_json::json!({ "my_vram_gb": 115.0, "peers": peers }) + } + + #[test] + fn projects_a_serving_peer_with_its_reported_figures() { + let view = live_view_from_payload(&payload(serde_json::json!([{ + "id": "abc123def456", + "hostname": "studio54.lan", + "state": "serving", + "vram_gb": 64.0, + "rtt_ms": 30, + "serving_models": ["gemma-4"], + }]))); + assert!(view.connected); + assert_eq!(view.self_capacity_gb, Some(115.0)); + assert_eq!(view.peers.len(), 1); + let peer = &view.peers[0]; + assert_eq!(peer.label, "studio54.lan"); + assert_eq!(peer.state, MeshPeerState::Serving); + assert_eq!(peer.capacity_gb, Some(64.0)); + assert_eq!(peer.rtt_ms, Some(30)); + assert_eq!(peer.models, vec!["gemma-4".to_string()]); + } + + #[test] + fn a_client_peer_reports_no_capacity_rather_than_zero_gb() { + // Client-mode nodes report vram_gb: 0. Showing "0 GB" would imply a + // machine with no memory instead of one that shares none. + let view = live_view_from_payload(&payload(serde_json::json!([{ + "id": "cli1", + "hostname": "mac.lan", + "state": "client", + "vram_gb": 0.0, + }]))); + assert_eq!(view.peers[0].state, MeshPeerState::Consuming); + assert_eq!(view.peers[0].capacity_gb, None); + } + + #[test] + fn a_declared_role_wins_over_advertised_models() { + // A client that lists a model is still consuming. + let view = live_view_from_payload(&payload(serde_json::json!([{ + "id": "cli2", + "hostname": "host.lan", + "state": "client", + "serving_models": ["gemma-4"], + }]))); + assert_eq!(view.peers[0].state, MeshPeerState::Consuming); + } + + #[test] + fn serving_peers_sort_before_consumers() { + let view = live_view_from_payload(&payload(serde_json::json!([ + { "id": "b", "hostname": "zulu.lan", "state": "client" }, + { "id": "a", "hostname": "alpha.lan", "state": "serving", "serving_models": ["m"] }, + ]))); + assert_eq!(view.peers[0].label, "alpha.lan"); + assert_eq!(view.peers[1].label, "zulu.lan"); + } + + #[test] + fn peers_without_an_id_are_dropped() { + let view = live_view_from_payload(&payload(serde_json::json!([ + { "hostname": "nameless.lan", "state": "serving" }, + ]))); + assert!(view.peers.is_empty()); + } + + #[test] + fn a_missing_peers_array_is_alone_not_broken() { + let view = live_view_from_payload(&serde_json::json!({ "my_vram_gb": 8.0 })); + assert!(view.connected); + assert!(view.peers.is_empty()); + } + + #[test] + fn an_unlabelled_peer_falls_back_to_a_short_id() { + let view = live_view_from_payload(&payload(serde_json::json!([ + { "id": "0123456789abcdef", "state": "standby" }, + ]))); + assert_eq!(view.peers[0].label, "0123456789"); + } +} diff --git a/desktop/src-tauri/src/mesh_llm/usage.rs b/desktop/src-tauri/src/mesh_llm/usage.rs index 56046cc71..49b9b7e9c 100644 --- a/desktop/src-tauri/src/mesh_llm/usage.rs +++ b/desktop/src-tauri/src/mesh_llm/usage.rs @@ -1,31 +1,46 @@ use serde::{Deserialize, Serialize}; -/// Host-side "who is using the compute I'm sharing" snapshot. +/// This node's own routing activity, plus the peers it can currently see. /// -/// Read-only projection of the serving node's own runtime metrics (the same -/// `routing_metrics` / `inflight_requests` the SDK already exposes on the local -/// console). No new trust surface: it reads the node's own status payload. +/// Read-only projection of the node's own status payload — no new trust +/// surface. /// -/// The local/remote/endpoint attempt split is what distinguishes *my own* -/// agent (local) from *another member consuming my compute* (remote/endpoint). +/// **Direction matters.** Every counter here comes from `routing_metrics`, +/// which is incremented only by this node's local OpenAI ingress +/// (`network/openai/transport.rs`) when *this* node dispatches a request. The +/// inbound peer-serving path (`mesh/stage_transport.rs`) never touches it — it +/// only observes inflight. So the attempt/served split says *where my requests +/// ran*, never *who asked me*: +/// +/// - local → my own GPU ran it +/// - remote → a peer ran it for me (I am CONSUMING someone else's compute) +/// +/// mesh-llm exposes no inbound counter (`fronted_request_count` merely aliases +/// `request_count`), so nothing here may be labelled "served for others". +/// Inbound work can only be *inferred*: serving, with inflight > 0, while our +/// own dispatch count stays flat, means the work is necessarily not ours. #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Default)] #[serde(rename_all = "camelCase")] pub struct MeshServingUsage { - /// Requests being served right now. + /// Inference in flight on this node right now. Does NOT distinguish a local + /// agent's request from a remote member's — the only live signal available. pub inflight: u64, /// Highest concurrent in-flight seen this session. pub peak_inflight: u64, - /// Total requests routed through this node. + /// Requests this node's ingress has routed (outbound), NOT work served for + /// others. Named `requests_served` for wire compatibility; treat as + /// "requests routed". pub requests_served: u64, - /// Completion tokens produced. + /// Completion tokens this node observed on requests it dispatched. pub tokens_served: u64, /// Recent decode throughput. pub tokens_per_second: f64, - /// Requests served for this machine's own agents. + /// Attempts this node dispatched to its own GPU. pub local_attempts: u64, - /// Requests served for a remote peer (someone else consuming my compute). + /// Attempts this node dispatched to a peer — i.e. this machine consuming + /// someone else's compute. pub remote_attempts: u64, - /// Requests served via an advertised endpoint (also a remote consumer). + /// Attempts this node dispatched to a configured endpoint. pub endpoint_attempts: u64, /// Other nodes currently visible as peers. pub peers: u64, diff --git a/desktop/src-tauri/src/mesh_llm_stubs.rs b/desktop/src-tauri/src/mesh_llm_stubs.rs index 5b93cf8c7..f9fc52460 100644 --- a/desktop/src-tauri/src/mesh_llm_stubs.rs +++ b/desktop/src-tauri/src/mesh_llm_stubs.rs @@ -47,3 +47,8 @@ pub async fn mesh_model_catalog() -> CmdResult { pub async fn mesh_snapshot(_state: State<'_, AppState>) -> CmdResult { Err("mesh-llm feature not enabled".to_string()) } + +#[tauri::command] +pub async fn mesh_live_view(_state: State<'_, AppState>) -> CmdResult<()> { + Err("mesh-llm feature is not enabled".to_string()) +} diff --git a/desktop/src/features/mesh-compute/hooks/useMeshLiveView.ts b/desktop/src/features/mesh-compute/hooks/useMeshLiveView.ts new file mode 100644 index 000000000..a73a73cae --- /dev/null +++ b/desktop/src/features/mesh-compute/hooks/useMeshLiveView.ts @@ -0,0 +1,57 @@ +import * as React from "react"; + +import { type MeshLiveView, meshLiveView } from "@/shared/api/tauriMesh"; + +/** + * Poll this machine's live gossip view of the mesh. + * + * Faster than the relay snapshot's 30s cadence because this is local IPC + * reading an in-process status payload — no relay round trip — and peers join + * and leave on a human timescale that a 30s poll makes look broken. + * + * Only polls while a runtime is expected to exist: with no node there is + * nothing to report, and hammering a command that always returns + * `connected:false` is waste. + */ +const POLL_MS = 5000; + +export function useMeshLiveView(enabled: boolean): { + view: MeshLiveView | null; + refresh: () => void; +} { + const [view, setView] = React.useState(null); + const [nonce, setNonce] = React.useState(0); + + // biome-ignore lint/correctness/useExhaustiveDependencies: nonce is an intentional re-run trigger for refresh(), never a value read in the body — bumping it re-polls the live view immediately after a local node transition + React.useEffect(() => { + if (!enabled) { + // Drop the stale view rather than leaving the last peers on screen after + // the node stops — they are no longer reachable. + setView(null); + return; + } + let cancelled = false; + const tick = async () => { + try { + const next = await meshLiveView(); + if (!cancelled) { + setView(next); + } + } catch { + // A transient IPC failure is not evidence the mesh is empty; keep the + // previous view rather than blanking the topology. + } + }; + void tick(); + const timer = setInterval(() => void tick(), POLL_MS); + return () => { + cancelled = true; + clearInterval(timer); + }; + }, [enabled, nonce]); + + return { + view, + refresh: React.useCallback(() => setNonce((value) => value + 1), []), + }; +} diff --git a/desktop/src/features/mesh-compute/meshActivity.ts b/desktop/src/features/mesh-compute/meshActivity.ts new file mode 100644 index 000000000..914767955 --- /dev/null +++ b/desktop/src/features/mesh-compute/meshActivity.ts @@ -0,0 +1,133 @@ +import type { MeshServingUsage } from "@/shared/api/tauriMesh"; + +/** + * Inferring inbound work — the "someone is using my machine" signal. + * + * mesh-llm exposes no inbound counter. `routing_metrics` is incremented only by + * this node's own OpenAI ingress, and `fronted_request_count` merely aliases + * `request_count`, so there is no field that counts work done for other + * members. + * + * But it is still *derivable*, because two facts we do hold are enough: + * + * 1. `inflight` counts inference in flight on this node, whoever asked. + * 2. `requestsServed` (= `request_count`) counts requests **we** dispatched. + * + * So if we are serving, there is work in flight, and our own dispatch count has + * not moved between samples — that work is not ours. It arrived from a peer. + * This is inference by elimination from data we already poll, not a guess. + * + * Its one weakness, stated so nobody mistakes it for a counter: it is sampled. + * A request that starts and finishes entirely between two polls is invisible, + * so the signal can undercount. It never *over*-claims, which is the direction + * that matters — we would rather miss a flicker than assert traffic that never + * happened. + */ + +export type MeshActivitySample = { + /** `inflight` at the time of the sample. */ + inflight: number; + /** `requestsServed` (outbound dispatch count) at the time of the sample. */ + requestsRouted: number; +}; + +export function activitySample( + usage: MeshServingUsage | null, +): MeshActivitySample | null { + if (!usage) { + return null; + } + return { inflight: usage.inflight, requestsRouted: usage.requestsServed }; +} + +/** + * Is the work currently in flight coming from someone else? + * + * Requires a previous sample: without one we cannot tell whether our own + * dispatch count is moving, so we decline to claim anything. + */ +export function inferInboundWork({ + isSharing, + current, + previous, +}: { + isSharing: boolean; + current: MeshActivitySample | null; + previous: MeshActivitySample | null; +}): boolean { + // Only a sharing node can receive inbound work at all. + if (!isSharing || !current || !previous) { + return false; + } + if (current.inflight === 0) { + return false; + } + // Our own dispatch count moved, so at least some of this is ours. Attributing + // it to a peer would be a claim we cannot support. + return current.requestsRouted === previous.requestsRouted; +} + +/** + * Where this machine's completed requests actually ran. + * + * Uses the `pressure` split, which counts *completed* requests rather than + * attempts, so it is a fact about our own consumption rather than a retry + * artefact. Returns null before anything completes rather than claiming + * "0 remote". + */ +export function describeRequestOrigin( + usage: MeshServingUsage | null, +): string | null { + if (!usage) { + return null; + } + const remote = usage.remotelyServed + usage.endpointServed; + const total = usage.locallyServed + remote; + if (total === 0) { + return null; + } + const label = (n: number, noun: string) => + `${n} ${n === 1 ? noun : `${noun}s`}`; + if (remote === 0) { + return `${label(total, "request")} ran here`; + } + if (usage.locallyServed === 0) { + return `${label(remote, "request")} ran on shared compute`; + } + return `${remote} of ${total} requests ran on shared compute`; +} + +/** + * The one-line participation hint under the card headline. + * + * Ordered by what the operator most wants to know: live inbound work first + * (their machine is earning its keep), then live work generally, then the + * cumulative record. Never asserts inbound work without the elimination check + * above holding. + */ +export function describeParticipationHint({ + isSharing, + isConsuming, + inboundWork, + usage, +}: { + isSharing: boolean; + isConsuming: boolean; + inboundWork: boolean; + usage: MeshServingUsage | null; +}): string { + if (isConsuming) { + return describeRequestOrigin(usage) ?? "Using shared compute"; + } + if (!isSharing) { + return "Share compute to run models"; + } + if (inboundWork) { + return "You're sharing · serving another member"; + } + if ((usage?.inflight ?? 0) > 0) { + return "You're sharing · working now"; + } + const origin = describeRequestOrigin(usage); + return origin ? `You're sharing · ${origin}` : "You're sharing"; +} diff --git a/desktop/src/features/mesh-compute/meshCardModel.test.mjs b/desktop/src/features/mesh-compute/meshCardModel.test.mjs index e7f7e3eac..22ccef7aa 100644 --- a/desktop/src/features/mesh-compute/meshCardModel.test.mjs +++ b/desktop/src/features/mesh-compute/meshCardModel.test.mjs @@ -15,7 +15,7 @@ import assert from "node:assert/strict"; import { describeMeshCapacity, - describeParticipation, + describeMeshHeadline, describeReadyModels, deriveMeshCardModel, formatCapacityGb, @@ -57,11 +57,42 @@ function snapshot(overrides = {}) { models: ["unsloth/gemma-4-26B-A4B-it-GGUF:UD-Q4_K_M"], devices: [device()], includesSelf: false, + memberCount: 1, reason: null, ...overrides, }; } +function usage(overrides = {}) { + return { + inflight: 0, + peakInflight: 0, + requestsServed: 0, + tokensServed: 0, + tokensPerSecond: 0, + localAttempts: 0, + remoteAttempts: 0, + endpointAttempts: 0, + peers: 0, + locallyServed: 0, + remotelyServed: 0, + endpointServed: 0, + ...overrides, + }; +} + +function peer(overrides = {}) { + return { + id: "peer-1", + label: "studio54.lan", + state: "serving", + capacityGb: 64, + models: ["m"], + rttMs: 30, + ...overrides, + }; +} + function derive(overrides = {}) { return deriveMeshCardModel({ snapshot: snapshot(), @@ -69,8 +100,9 @@ function derive(overrides = {}) { toggle: OFF_TOGGLE, pendingAction: null, canShare: true, - busyNow: false, - requestsRouted: 0, + view: null, + usage: usage(), + inboundWork: false, ...overrides, }); } @@ -138,105 +170,129 @@ test("ready models collapse to a count past one", () => { test("consuming never renders as sharing", () => { const model = derive({ toggle: CONSUMING_TOGGLE, - snapshot: snapshot({ devices: [device({ label: "Mac Mini" })] }), + usage: usage({ remotelyServed: 4 }), }); assert.equal(model.tone, "consuming"); assert.equal(model.switchOn, false, "the Share switch must stay off"); - assert.match(model.headline, /Using shared compute/); - assert.match(model.detail, /Mac Mini/); + // The headline names the mesh; the detail line carries this machine's role, + // and must say we are taking rather than giving. + assert.match(model.detail, /ran on shared compute/); + assert.ok( + !/You're sharing/.test(model.detail), + `consuming must not read as sharing: ${model.detail}`, + ); // A consuming client may be replaced by a serve runtime, so the switch stays // usable rather than locked. assert.equal(model.switchDisabled, false); }); -test("sharing leads with mesh capacity and omits the model name", () => { +const SERVE_STATUS = { + state: "running", + mode: "serve", + health: { status: "ok", reason: null }, + modelId: "unsloth/gemma-4-26B-A4B-it-GGUF:UD-Q4_K_M", + modelName: null, + apiBaseUrl: null, + consoleUrl: null, +}; + +test("the headline names MeshLLM, live capacity and peer count", () => { + assert.equal( + describeMeshHeadline({ + view: { + connected: true, + selfCapacityGb: 115, + peers: [peer({ capacityGb: 64 })], + }, + snapshot: snapshot(), + }), + "MeshLLM · 179 GB, 1 peer", + ); +}); + +test("a consuming-only mesh shows peers without inventing capacity", () => { + // Client-mode peers share no memory, so a mesh of consumers has no GB to + // report — but the peers are still really there. + assert.equal( + describeMeshHeadline({ + view: { + connected: true, + selfCapacityGb: null, + peers: [peer({ capacityGb: null, state: "consuming" })], + }, + snapshot: snapshot(), + }), + "MeshLLM · 1 peer", + ); +}); + +test("with no runtime the headline falls back to the relay snapshot", () => { + // Peers are unknowable without a node, so the community view is all we have. + assert.equal( + describeMeshHeadline({ + view: { connected: false, selfCapacityGb: null, peers: [] }, + snapshot: snapshot({ sharingDeviceCount: 3, sharedCapacityGb: 42 }), + }), + "42 GB · 3 devices", + ); +}); + +test("sharing omits the model name and hints participation", () => { const model = derive({ toggle: SHARING_TOGGLE, - status: { - state: "running", - mode: "serve", - health: { status: "ok", reason: null }, - modelId: "unsloth/gemma-4-26B-A4B-it-GGUF:UD-Q4_K_M", - modelName: null, - apiBaseUrl: null, - consoleUrl: null, - }, - snapshot: snapshot({ - sharingDeviceCount: 1, - sharedCapacityGb: 115, - devices: [device({ isSelf: true })], - includesSelf: true, - }), + status: SERVE_STATUS, + view: { connected: true, selfCapacityGb: 115, peers: [peer()] }, }); assert.equal(model.tone, "sharing"); assert.equal(model.switchOn, true); - assert.equal(model.headline, "115 GB · 1 device"); + assert.equal(model.headline, "MeshLLM · 179 GB, 1 peer"); + assert.match(model.detail, /You're sharing/); // The model belongs in the detail view, not on a 256px card. assert.ok(!/Gemma/i.test(model.detail ?? ""), "card must not name the model"); }); -test("inflight work reads as working, never as 'someone is using you'", () => { +test("inferred inbound work is the one place a peer may be named", () => { const model = derive({ toggle: SHARING_TOGGLE, - status: { - state: "running", - mode: "serve", - health: { status: "ok", reason: null }, - modelId: "m", - modelName: null, - apiBaseUrl: null, - consoleUrl: null, - }, - busyNow: true, + status: SERVE_STATUS, + usage: usage({ inflight: 1 }), + inboundWork: true, + }); + assert.match(model.detail, /serving another member/); +}); + +test("inflight work alone reads as working, never as being used", () => { + // Without the elimination check the work may be our own agent's, so the copy + // must not attribute it to anyone. + const model = derive({ + toggle: SHARING_TOGGLE, + status: SERVE_STATUS, + usage: usage({ inflight: 1 }), + inboundWork: false, }); assert.match(model.detail, /working now/); -}); - -test("routed requests are a subtle used-ness signal, not a served claim", () => { - const model = derive({ - toggle: SHARING_TOGGLE, - status: { - state: "running", - mode: "serve", - health: { status: "ok", reason: null }, - modelId: "m", - modelName: null, - apiBaseUrl: null, - consoleUrl: null, - }, - requestsRouted: 7, - }); - assert.match(model.detail, /7 requests this session/); -}); - -test("participation copy never claims work was served for other members", () => { - // mesh-llm exposes no inbound counter, so no wording here may imply that - // another member consumed this machine's compute. - const claims = [ - describeParticipation({ busyNow: false, requestsRouted: 0 }), - describeParticipation({ busyNow: true, requestsRouted: 3 }), - describeParticipation({ busyNow: false, requestsRouted: 3 }), - ]; - assert.deepEqual(claims, [ - "Sharing · ready", - "Sharing · working now", - "Sharing · 3 requests this session", - ]); - for (const claim of claims) { - assert.ok( - !/served|for (others|members|someone)|consumed/i.test(claim), - `must not claim served-for-others: ${claim}`, - ); - } -}); - -test("one routed request is singular", () => { - assert.equal( - describeParticipation({ busyNow: false, requestsRouted: 1 }), - "Sharing · 1 request this session", + assert.ok( + !/another member|someone|served for/i.test(model.detail), + `must not attribute the work: ${model.detail}`, ); }); +test("a solo sharer is a live-view fact, not a lone relay note", () => { + // One status note may simply be a stale one; only gossip can say "alone". + const alone = derive({ + toggle: SHARING_TOGGLE, + status: SERVE_STATUS, + view: { connected: true, selfCapacityGb: 115, peers: [] }, + }); + assert.equal(alone.isSolo, true); + const populated = derive({ + toggle: SHARING_TOGGLE, + status: SERVE_STATUS, + view: { connected: true, selfCapacityGb: 115, peers: [peer()] }, + }); + assert.equal(populated.isSolo, false); +}); + test("a serve node with no advertised model is warming up, not serving", () => { const model = derive({ toggle: SHARING_TOGGLE, @@ -294,34 +350,3 @@ test("an unknown runtime occupant is not silently replaceable", () => { }); assert.equal(model.switchDisabled, true); }); - -test("a solo sharer gets the waiting hint; a populated mesh does not", () => { - const sharingStatus = { - state: "running", - mode: "serve", - health: { status: "ok", reason: null }, - modelId: "m", - modelName: null, - apiBaseUrl: null, - consoleUrl: null, - }; - const solo = derive({ - toggle: SHARING_TOGGLE, - status: sharingStatus, - snapshot: snapshot({ devices: [device({ isSelf: true })] }), - }); - assert.equal(solo.showSoloHint, true); - - const populated = derive({ - toggle: SHARING_TOGGLE, - status: sharingStatus, - snapshot: snapshot({ - sharingDeviceCount: 2, - devices: [ - device({ isSelf: true }), - device({ deviceId: "e2", label: "Studio 2" }), - ], - }), - }); - assert.equal(populated.showSoloHint, false); -}); diff --git a/desktop/src/features/mesh-compute/meshCardModel.ts b/desktop/src/features/mesh-compute/meshCardModel.ts index 91406627e..a2fd4312a 100644 --- a/desktop/src/features/mesh-compute/meshCardModel.ts +++ b/desktop/src/features/mesh-compute/meshCardModel.ts @@ -1,8 +1,11 @@ import type { + MeshLiveView, MeshNodeStatus, + MeshServingUsage, MeshSnapshot, MeshSnapshotDevice, } from "@/shared/api/tauriMesh"; +import { describeParticipationHint } from "./meshActivity"; import type { MeshShareToggleModel } from "./shareToggleState"; /** @@ -93,6 +96,41 @@ export function describeMeshCapacity(snapshot: MeshSnapshot | null): string { : `${formatCapacityGb(gb)} · ${devices}`; } +/** + * The card headline once we have a live view: name the mesh, its capacity, and + * how many nodes we can actually see. + * + * Prefers the live gossip view over the relay snapshot. Peers listed here are + * ones our runtime is talking to *now*, whereas relay status notes stay valid + * for 120s and so outlive the node that wrote them — a graph built on notes + * shows devices gossip already knows are gone. + * + * Capacity sums this machine plus its serving peers. Consuming peers contribute + * a peer count but no GB, because they share none. + */ +export function describeMeshHeadline({ + view, + snapshot, +}: { + view: MeshLiveView | null; + snapshot: MeshSnapshot | null; +}): string { + // Not participating: the relay snapshot is the only view of the pool, and the + // reason to consider joining. + if (!view?.connected) { + return describeMeshCapacity(snapshot); + } + const capacityGb = [ + view.selfCapacityGb ?? 0, + ...view.peers.map((peer) => peer.capacityGb ?? 0), + ].reduce((total, gb) => total + gb, 0); + const peerCount = view.peers.length; + const peers = `${peerCount} ${plural(peerCount, "peer")}`; + return capacityGb > 0 + ? `MeshLLM · ${formatCapacityGb(capacityGb)}, ${peers}` + : `MeshLLM · ${peers}`; +} + /** Short label for what is ready to run, or null when nothing is. */ export function describeReadyModels( snapshot: MeshSnapshot | null, @@ -107,30 +145,6 @@ export function describeReadyModels( return `${models.length} models ready`; } -/** - * The sharing detail line: proof this machine is participating, and that the - * mesh has actually been used. - * - * Scrupulously avoids claiming someone else consumed this machine's compute — - * mesh-llm exposes no inbound counter, so "requests routed" is the strongest - * true statement available. Worded as "requests" without "served for others". - */ -export function describeParticipation({ - busyNow, - requestsRouted, -}: { - busyNow: boolean; - requestsRouted: number; -}): string { - if (busyNow) { - return "Sharing · working now"; - } - if (requestsRouted > 0) { - return `Sharing · ${requestsRouted} ${plural(requestsRouted, "request")} this session`; - } - return "Sharing · ready"; -} - /** * Trim a model reference down to something that fits a 256px sidebar. * `unsloth/gemma-4-26B-A4B-it-GGUF:UD-Q4_K_M` → `Gemma 4 26B A4B`. @@ -161,8 +175,9 @@ export function deriveMeshCardModel({ toggle, pendingAction, canShare, - busyNow, - requestsRouted, + view, + usage, + inboundWork, }: { snapshot: MeshSnapshot | null; status: MeshNodeStatus | null; @@ -170,27 +185,28 @@ export function deriveMeshCardModel({ pendingAction: "start" | "stop" | null; /** False when no model can be resolved yet (catalog still loading). */ canShare: boolean; + /** Live gossip view. Null/disconnected falls back to the relay snapshot. */ + view: MeshLiveView | null; + /** This node's own routing counters. All outbound — see `meshActivity.ts`. */ + usage: MeshServingUsage | null; /** - * This node has inference in flight (`inflight > 0`). - * - * Honest but coarse: mesh-llm's inflight counter does not say whether the - * work is for a local agent or a remote member, so this only ever claims - * "working", never "someone is using your compute". + * Inbound work inferred by elimination (serving + inflight + our own dispatch + * count flat). Sampled, so it can undercount; it never over-claims. */ - busyNow: boolean; - /** - * Requests this node's ingress has routed this session (`request_count`). - * - * Outbound routing, NOT work served for others — mesh-llm exposes no inbound - * counter. Used only as a subtle "this has been used" signal. - */ - requestsRouted: number; + inboundWork: boolean; }): MeshCardModel { const devices = snapshot?.devices ?? []; - const participantCount = devices.length; - const capacity = describeMeshCapacity(snapshot); + const headline = describeMeshHeadline({ view, snapshot }); const ready = describeReadyModels(snapshot); - const isSolo = participantCount === 1; + // Solo means "connected but nobody else is here" — a live-view fact. The + // relay snapshot cannot tell us this: a lone note may just be a stale one. + const isSolo = view?.connected === true && view.peers.length === 0; + const hint = describeParticipationHint({ + isSharing: toggle.isSharing, + isConsuming: toggle.isConsuming, + inboundWork, + usage, + }); const base = { devices, @@ -232,16 +248,11 @@ export function deriveMeshCardModel({ // Consuming: this machine is TAKING compute, not giving it. Say so plainly — // the switch is off here and that must not read as "nothing is happening". if (toggle.isConsuming) { - const peer = devices.find( - (device) => !device.isSelf && device.state === "serving", - ); return { ...base, tone: "consuming", - headline: "Using shared compute", - detail: peer - ? `Running on ${peer.label}. Turn on to share this computer too.` - : "Running on another member's computer.", + headline, + detail: hint, showSoloHint: false, }; } @@ -274,8 +285,8 @@ export function deriveMeshCardModel({ return { ...base, tone: "sharing", - headline: capacity, - detail: describeParticipation({ busyNow, requestsRouted }), + headline, + detail: hint, showSoloHint: isSolo, }; } @@ -285,8 +296,8 @@ export function deriveMeshCardModel({ return { ...base, tone: "idle", - headline: capacity, - detail: ready ?? "Share compute to run models.", + headline, + detail: ready ?? hint, showSoloHint: false, }; } diff --git a/desktop/src/features/mesh-compute/meshDetailModel.test.mjs b/desktop/src/features/mesh-compute/meshDetailModel.test.mjs index 2ea546d72..8adb35740 100644 --- a/desktop/src/features/mesh-compute/meshDetailModel.test.mjs +++ b/desktop/src/features/mesh-compute/meshDetailModel.test.mjs @@ -10,10 +10,11 @@ import test from "node:test"; */ import { - describeActivity, + describeParticipationHint, describeRequestOrigin, - deriveMeshDetailModel, -} from "./meshDetailModel.ts"; + inferInboundWork, +} from "./meshActivity.ts"; +import { describeActivity, deriveMeshDetailModel } from "./meshDetailModel.ts"; function usage(overrides = {}) { return { @@ -45,6 +46,27 @@ function device(overrides = {}) { }; } +function peer(overrides = {}) { + return { + id: "peer-1", + label: "studio54.lan", + state: "serving", + capacityGb: 64, + models: ["m"], + rttMs: 30, + ...overrides, + }; +} + +function view(overrides = {}) { + return { + connected: true, + selfCapacityGb: 115, + peers: [], + ...overrides, + }; +} + function snapshot(overrides = {}) { return { sharingDeviceCount: 1, @@ -61,15 +83,15 @@ function snapshot(overrides = {}) { test("request origin distinguishes borrowed compute from local work", () => { assert.equal( describeRequestOrigin(usage({ locallyServed: 5 })), - "5 requests · all on this computer", + "5 requests ran here", ); assert.equal( describeRequestOrigin(usage({ remotelyServed: 4 })), - "4 requests · all on shared compute", + "4 requests ran on shared compute", ); assert.equal( describeRequestOrigin(usage({ locallyServed: 9, remotelyServed: 3 })), - "3 of 12 requests on shared compute", + "3 of 12 requests ran on shared compute", ); }); @@ -77,7 +99,7 @@ test("endpoint-served requests count as shared, not local", () => { // An endpoint is not this machine's GPU, so it must not read as local work. assert.equal( describeRequestOrigin(usage({ locallyServed: 1, endpointServed: 1 })), - "1 of 2 requests on shared compute", + "1 of 2 requests ran on shared compute", ); }); @@ -89,97 +111,219 @@ test("no completed requests yields no origin claim, never '0 remote'", () => { test("singular grammar for a single request", () => { assert.equal( describeRequestOrigin(usage({ locallyServed: 1 })), - "1 request · all on this computer", + "1 request ran here", ); }); test("activity reports inflight work without attributing it to anyone", () => { - const live = describeActivity(usage({ inflight: 2 }), true); + const live = describeActivity({ + usage: usage({ inflight: 2 }), + isSharing: true, + inboundWork: false, + }); assert.equal(live, "2 requests in flight"); - // The counter cannot distinguish a local agent from a remote member, so the + // Without the elimination check the work may be our own agent's, so the // phrasing must never imply someone else is using this machine. assert.ok( - !/(served|for (others|members|someone)|using your|consumed)/i.test(live), + !/(served|for (others|members|someone)|using your|consumed|another member)/i.test( + live, + ), `must not attribute inbound work: ${live}`, ); }); +test("inferred inbound work is the one case a peer may be credited", () => { + assert.equal( + describeActivity({ + usage: usage({ inflight: 2 }), + isSharing: true, + inboundWork: true, + }), + "2 requests in flight · from another member", + ); +}); + test("a warm sharing runtime with no traffic is its own state", () => { - assert.equal(describeActivity(usage(), true), "Ready · no requests yet"); + assert.equal( + describeActivity({ usage: usage(), isSharing: true, inboundWork: false }), + "Ready · no requests yet", + ); // Not sharing and idle is not noteworthy — stay silent. - assert.equal(describeActivity(usage(), false), null); - assert.equal(describeActivity(null, true), null); + assert.equal( + describeActivity({ usage: usage(), isSharing: false, inboundWork: false }), + null, + ); + assert.equal( + describeActivity({ usage: null, isSharing: true, inboundWork: false }), + null, + ); }); -test("participation names the roster denominator", () => { - const model = deriveMeshDetailModel({ - snapshot: snapshot({ - sharingDeviceCount: 2, - memberCount: 12, - sharedCapacityGb: 115, - devices: [device({ isSelf: true }), device({ deviceId: "d2" })], +test("inbound work requires two samples and a flat dispatch count", () => { + const base = { isSharing: true }; + // Our own dispatch count moved, so at least some of the work is ours. + assert.equal( + inferInboundWork({ + ...base, + current: { inflight: 1, requestsRouted: 6 }, + previous: { inflight: 0, requestsRouted: 5 }, }), - usage: usage(), - isSharing: true, - }); - assert.equal(model.participationLabel, "2 of 12 members sharing"); - assert.equal(model.capacityLabel, "115 GB"); - // 12 members, 2 with published status notes -> 10 ghosts. - assert.equal(model.ghostCount, 10); -}); - -test("ghost count never goes negative when the roster lags devices", () => { - // A device can report while a stale roster page omits it. A negative ghost - // count is nonsense, so it clamps at zero. - const model = deriveMeshDetailModel({ - snapshot: snapshot({ - memberCount: 1, - devices: [device({ isSelf: true }), device({ deviceId: "d2" })], + false, + ); + // Flat dispatch count with work in flight: it cannot be ours. + assert.equal( + inferInboundWork({ + ...base, + current: { inflight: 1, requestsRouted: 5 }, + previous: { inflight: 0, requestsRouted: 5 }, }), - usage: usage(), - isSharing: true, - }); - assert.equal(model.ghostCount, 0); + true, + ); + // No history yet — decline to claim anything. + assert.equal( + inferInboundWork({ + ...base, + current: { inflight: 1, requestsRouted: 5 }, + previous: null, + }), + false, + ); + // Nothing in flight, and a non-sharing node cannot receive inbound work. + assert.equal( + inferInboundWork({ + ...base, + current: { inflight: 0, requestsRouted: 5 }, + previous: { inflight: 0, requestsRouted: 5 }, + }), + false, + ); + assert.equal( + inferInboundWork({ + isSharing: false, + current: { inflight: 1, requestsRouted: 5 }, + previous: { inflight: 0, requestsRouted: 5 }, + }), + false, + ); }); -test("an empty roster falls back to a device count", () => { +test("participation counts live peers, not relay notes", () => { const model = deriveMeshDetailModel({ - snapshot: snapshot({ sharingDeviceCount: 1, memberCount: 0 }), + view: view({ peers: [peer(), peer({ id: "p2", label: "mac.lan" })] }), + snapshot: snapshot(), usage: usage(), isSharing: true, + inboundWork: false, }); - assert.equal(model.participationLabel, "1 device sharing"); + assert.equal(model.participationLabel, "2 peers connected"); + assert.equal(model.connected, true); + // 115 self + 64 + 64 peers. + assert.equal(model.capacityLabel, "243 GB"); }); -test("unknown capacity drops the figure rather than printing 0 GB", () => { - const model = deriveMeshDetailModel({ - snapshot: snapshot({ sharedCapacityGb: null }), +test("connected but alone is distinct from not participating", () => { + const alone = deriveMeshDetailModel({ + view: view({ peers: [] }), + snapshot: snapshot(), usage: usage(), isSharing: true, + inboundWork: false, }); + assert.equal(alone.participationLabel, "No other devices yet"); + assert.equal(alone.connected, true); + + // With no runtime, peers are unknowable and the relay view is all we have. + const off = deriveMeshDetailModel({ + view: { connected: false, selfCapacityGb: null, peers: [] }, + snapshot: snapshot({ sharingDeviceCount: 3 }), + usage: usage(), + isSharing: false, + inboundWork: false, + }); + assert.equal(off.participationLabel, "3 sharing in this community"); + assert.equal(off.connected, false); +}); + +test("consuming peers add presence but never invented capacity", () => { + const model = deriveMeshDetailModel({ + view: view({ + selfCapacityGb: null, + peers: [peer({ capacityGb: null, state: "consuming", models: [] })], + }), + snapshot: snapshot(), + usage: usage(), + isSharing: false, + inboundWork: false, + }); + assert.equal(model.participationLabel, "1 peer connected"); + // Nobody reported a figure, so no GB is claimed rather than "0 GB". assert.equal(model.capacityLabel, null); }); -test("a null snapshot is renderable and claims nothing", () => { +test("a null view and snapshot are renderable and claim nothing", () => { const model = deriveMeshDetailModel({ + view: null, snapshot: null, usage: null, isSharing: false, + inboundWork: false, }); assert.equal(model.capacityLabel, null); - assert.equal(model.ghostCount, 0); + assert.equal(model.connected, false); assert.equal(model.busyNow, false); assert.equal(model.activityLabel, null); assert.equal(model.originLabel, null); + assert.equal(model.modelCount, 0); }); test("capacity formatting keeps small figures meaningful", () => { assert.equal( deriveMeshDetailModel({ - snapshot: snapshot({ sharedCapacityGb: 4.62 }), + view: view({ selfCapacityGb: 4.62, peers: [] }), + snapshot: snapshot(), usage: usage(), isSharing: true, + inboundWork: false, }).capacityLabel, "4.6 GB", ); }); + +test("the card hint distinguishes giving, taking, and neither", () => { + assert.equal( + describeParticipationHint({ + isSharing: false, + isConsuming: false, + inboundWork: false, + usage: usage(), + }), + "Share compute to run models", + ); + assert.equal( + describeParticipationHint({ + isSharing: true, + isConsuming: false, + inboundWork: false, + usage: usage(), + }), + "You're sharing", + ); + assert.equal( + describeParticipationHint({ + isSharing: true, + isConsuming: false, + inboundWork: true, + usage: usage({ inflight: 1 }), + }), + "You're sharing · serving another member", + ); + assert.equal( + describeParticipationHint({ + isSharing: false, + isConsuming: true, + inboundWork: false, + usage: usage({ remotelyServed: 3, locallyServed: 1 }), + }), + "3 of 4 requests ran on shared compute", + ); +}); diff --git a/desktop/src/features/mesh-compute/meshDetailModel.ts b/desktop/src/features/mesh-compute/meshDetailModel.ts index e84ec82f6..3b06d4760 100644 --- a/desktop/src/features/mesh-compute/meshDetailModel.ts +++ b/desktop/src/features/mesh-compute/meshDetailModel.ts @@ -1,13 +1,23 @@ -import type { MeshServingUsage, MeshSnapshot } from "@/shared/api/tauriMesh"; +import type { + MeshLiveView, + MeshServingUsage, + MeshSnapshot, +} from "@/shared/api/tauriMesh"; +import { describeRequestOrigin } from "./meshActivity"; /** * Pure projection for the mesh detail popover. * * The popover answers three questions the 256px card cannot: - * 1. How big is the pool, and how much of the community is in it? - * 2. Is the spice flowing — is work actually happening right now? + * 1. Who is on the mesh right now, and how much do they bring? + * 2. Is the spice flowing — is work actually happening? * 3. Am I running on someone else's machine, or my own? * + * Peers come from the **live gossip view**, not the relay snapshot. Status notes + * stay valid for 120s and so outlive the node that wrote them; a topology built + * on notes shows devices gossip already knows are gone. The relay snapshot is + * used only when no local runtime exists, where it is the sole available view. + * * The hard constraint this file exists to encode: **mesh-llm exposes no inbound * counter.** `routing_metrics` is incremented only by this node's own OpenAI * ingress, so every number available describes work *this machine dispatched*. @@ -15,66 +25,34 @@ import type { MeshServingUsage, MeshSnapshot } from "@/shared/api/tauriMesh"; * * - "I used someone else's machine" → provable (`remotelyServed`) * - "my own GPU did the work" → provable (`locallyServed`) - * - "someone used MY machine" → NOT provable, only ever hinted + * - "someone used MY machine" → no counter; INFERRED by elimination * - * `inflight` is the one live signal, and it does not say who the work is for. - * That is why the busy state is worded "working" and never "someone is using - * your compute". + * That last one is derivable without a counter: sharing, with work in flight, + * while our own dispatch count stays flat, means the work is not ours. See + * `inferInboundWork` in `meshActivity.ts`. It is sampled, so it can undercount + * — but it never over-claims, which is the direction that matters. */ -/** Ghost = a community member with no published status note. */ export type MeshDetailModel = { - /** e.g. "115 GB" — pool capacity, or null when nobody reported a figure. */ + /** e.g. "115 GB" — live pool capacity, or null when nothing is shared. */ capacityLabel: string | null; - /** e.g. "2 of 12 members sharing". */ + /** e.g. "2 peers connected", or the relay view when not participating. */ participationLabel: string; - /** Members with no status note at all. Count only — never GB. */ - ghostCount: number; /** True when inference is in flight on this node right now. */ busyNow: boolean; /** - * Live activity phrase, or null when idle. Deliberately vague about *who*: - * the counters cannot attribute inbound work. + * Live activity phrase, or null when idle. Only claims inbound work when the + * elimination check in `meshActivity.ts` holds. */ activityLabel: string | null; /** Where this machine's completed requests actually ran. */ originLabel: string | null; - /** Distinct models ready across the pool. */ + /** Distinct models serving across the live mesh. */ modelCount: number; + /** True when a local runtime is up, so peers are knowable at all. */ + connected: boolean; }; -function plural(n: number, one: string, many = `${one}s`): string { - return n === 1 ? one : many; -} - -/** - * Where this machine's work ran — the honest "am I borrowing someone's GPU" - * line. - * - * `remotelyServed` counts *completed* requests a peer answered for us, so it is - * a fact about our own consumption, not a guess. Returns null before any - * request completes rather than claiming "0 remote". - */ -export function describeRequestOrigin( - usage: MeshServingUsage | null, -): string | null { - if (!usage) { - return null; - } - const remote = usage.remotelyServed + usage.endpointServed; - const total = usage.locallyServed + remote; - if (total === 0) { - return null; - } - if (remote === 0) { - return `${total} ${plural(total, "request")} · all on this computer`; - } - if (usage.locallyServed === 0) { - return `${remote} ${plural(remote, "request")} · all on shared compute`; - } - return `${remote} of ${total} requests on shared compute`; -} - /** * The live "spice is flowing" phrase. * @@ -82,13 +60,20 @@ export function describeRequestOrigin( * this node without distinguishing a local agent from a remote member, so this * may never assert that someone else is using this machine. */ -export function describeActivity( - usage: MeshServingUsage | null, - isSharing: boolean, -): string | null { +export function describeActivity({ + usage, + isSharing, + inboundWork, +}: { + usage: MeshServingUsage | null; + isSharing: boolean; + inboundWork: boolean; +}): string | null { const inflight = usage?.inflight ?? 0; if (inflight > 0) { - return `${inflight} ${plural(inflight, "request")} in flight`; + const requests = `${inflight} ${inflight === 1 ? "request" : "requests"} in flight`; + // Only name a peer as the source when elimination proves it isn't ours. + return inboundWork ? `${requests} · from another member` : requests; } if (!usage) { return null; @@ -102,35 +87,58 @@ export function describeActivity( } export function deriveMeshDetailModel({ + view, snapshot, usage, isSharing, + inboundWork, }: { + view: MeshLiveView | null; snapshot: MeshSnapshot | null; usage: MeshServingUsage | null; isSharing: boolean; + inboundWork: boolean; }): MeshDetailModel { - const devices = snapshot?.devices ?? []; - const sharing = snapshot?.sharingDeviceCount ?? 0; - const members = snapshot?.memberCount ?? 0; - // Members who published nothing. Clamped at 0: a device may report without - // appearing in a stale roster page, and a negative ghost count is nonsense. - const ghostCount = Math.max(0, members - devices.length); - const capacityGb = snapshot?.sharedCapacityGb ?? null; + const connected = view?.connected === true; + const peers = view?.peers ?? []; + + // Capacity from the live mesh: this machine plus serving peers. A consuming + // peer contributes presence but no GB, because it shares none. + const liveCapacity = connected + ? [ + view?.selfCapacityGb ?? 0, + ...peers.map((peer) => peer.capacityGb ?? 0), + ].reduce((total, gb) => total + gb, 0) + : 0; + const capacityGb = connected + ? liveCapacity > 0 + ? liveCapacity + : null + : (snapshot?.sharedCapacityGb ?? null); + + const models = new Set(peers.flatMap((peer) => peer.models)); + if (isSharing) { + for (const model of snapshot?.devices.find((d) => d.isSelf)?.models ?? []) { + models.add(model); + } + } return { + connected, capacityLabel: capacityGb === null ? null : `${capacityGb >= 10 ? Math.round(capacityGb) : Math.round(capacityGb * 10) / 10} GB`, - participationLabel: - members === 0 - ? `${sharing} ${plural(sharing, "device")} sharing` - : `${sharing} of ${members} ${plural(members, "member")} sharing`, - ghostCount, + participationLabel: connected + ? peers.length === 0 + ? "No other devices yet" + : `${peers.length} ${peers.length === 1 ? "peer" : "peers"} connected` + : // Not participating: the relay snapshot is all we have, and it is the + // reason to consider joining. + `${snapshot?.sharingDeviceCount ?? 0} sharing in this community`, busyNow: (usage?.inflight ?? 0) > 0, - activityLabel: describeActivity(usage, isSharing), + activityLabel: describeActivity({ usage, isSharing, inboundWork }), originLabel: describeRequestOrigin(usage), - modelCount: snapshot?.models.length ?? 0, + modelCount: models.size, }; } diff --git a/desktop/src/features/mesh-compute/ui/MeshDetailPopover.tsx b/desktop/src/features/mesh-compute/ui/MeshDetailPopover.tsx index 16f681892..751ece507 100644 --- a/desktop/src/features/mesh-compute/ui/MeshDetailPopover.tsx +++ b/desktop/src/features/mesh-compute/ui/MeshDetailPopover.tsx @@ -1,4 +1,8 @@ -import type { MeshServingUsage, MeshSnapshot } from "@/shared/api/tauriMesh"; +import type { + MeshLiveView, + MeshServingUsage, + MeshSnapshot, +} from "@/shared/api/tauriMesh"; import { Popover, PopoverContent, PopoverTrigger } from "@/shared/ui/popover"; import { POPOVER_SHADOW_STYLE } from "@/shared/ui/popoverSurface"; import { deriveMeshDetailModel } from "../meshDetailModel"; @@ -10,25 +14,38 @@ import { MeshTopologyRadial } from "./MeshTopologyRadial"; * Anchored rather than a centred dialog: this is a glance, not a task. It sits * beside the thing it explains, and dismisses on outside click. * - * Every figure here is one this machine can actually vouch for. In particular - * there is no "N requests served for others" line, because mesh-llm exposes no - * inbound counter — only a soft hint when the pool is live. See - * `meshDetailModel.ts` for the full constraint. + * Every figure here is one this machine can actually vouch for. Peers come from + * live gossip, so a node shown is a node we are connected to — not a relay note + * that may have outlived its writer. Inbound work has no counter, so it is + * inferred by elimination and only named when that inference holds. See + * `meshDetailModel.ts` and `meshActivity.ts`. */ export function MeshDetailPopover({ children, + view, snapshot, usage, isSharing, + inboundWork, onOpenComputeSettings, }: { children: React.ReactNode; + /** Live gossip view — the source for the topology. */ + view: MeshLiveView | null; + /** Relay snapshot, used only when no local runtime exists. */ snapshot: MeshSnapshot | null; usage: MeshServingUsage | null; isSharing: boolean; + inboundWork: boolean; onOpenComputeSettings?: () => void; }) { - const model = deriveMeshDetailModel({ snapshot, usage, isSharing }); + const model = deriveMeshDetailModel({ + view, + snapshot, + usage, + isSharing, + inboundWork, + }); return ( @@ -53,11 +70,14 @@ export function MeshDetailPopover({

- + {model.connected ? ( + + ) : null}
{model.activityLabel ? ( @@ -95,13 +115,11 @@ export function MeshDetailPopover({ ) : null}
- {model.ghostCount > 0 ? ( + {model.connected ? null : (

- {model.ghostCount === 1 - ? "1 member isn't sharing yet." - : `${model.ghostCount} members aren't sharing yet.`} + Turn on sharing to see who's on the mesh.

- ) : null} + )} {onOpenComputeSettings ? (