fix: close cross-process keychain race and namespace dev-build nest (#1409)

Signed-off-by: Will Pfleger <pfleger.will@gmail.com>
Co-authored-by: npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7 <dcfd242e557282d7a1e2cf2e6877522682f1e5c6156dc92ca7d90eaedd3b0f95@sprout-oss.stage.blox.sqprod.co>
This commit is contained in:
Will Pfleger
2026-06-30 18:40:02 -04:00
committed by GitHub
co-authored by npub1mn7jgtj4w2pd0g0zeuhxsa6jy6p0rewxz4kujt98my82ahfmp72sxjexk7
parent 0042c8e106
commit e8adc3383f
6 changed files with 595 additions and 73 deletions
+8 -2
View File
@@ -67,7 +67,9 @@ const overrides = new Map([
// self-contained repos_dir functions and their unit tests live in repos.rs;
// this is the seam that must stay in nest.rs. Approved override; still queued
// to split with the rest of this list.
["src-tauri/src/managed_agents/nest.rs", 1450],
// dev-nest namespace: OnceLock<Option<PathBuf>> + init_nest_dir + constants
// added to plumb the dev/prod discriminator. Load-bearing for the D2 nest fix.
["src-tauri/src/managed_agents/nest.rs", 1501],
// harness-persona-sync: persona-runtime resolution threaded into the spawn
// path here. Load-bearing feature growth; queued to split in the resolver
// unify refactor followup. +26 for resolve_effective_prompt_model_provider
@@ -97,7 +99,7 @@ const overrides = new Map([
["src-tauri/src/migration_tests.rs", 1410],
["src-tauri/src/nostr_convert.rs", 1126],
["src/shared/api/relayClientSession.ts", 1022],
["src-tauri/src/migration.rs", 1449],
["src-tauri/src/migration.rs", 1575],
// onMarkRead + isUnread prop threading (mirrors the onMarkUnread prop
// already here) for the single-toggle mark-read/unread menu item — a small
// overage from load-bearing per-message plumbing, not generic debt growth.
@@ -113,6 +115,10 @@ const overrides = new Map([
// fail-closed regression tests (silent identity rotation on keyring outage).
// A small overage from load-bearing security plumbing on a file already at
// 893 lines, not generic debt growth. Approved override; still queued to split.
// cross-process keychain race fix (D3): interprocess lock + BlobLockGuard +
// uid-keyed lockfile path + behavioral tests add ~303 lines. Load-bearing
// security fix for the lost-update race that stranded agent keys.
["src-tauri/src/secret_store.rs", 1043],
["src-tauri/src/app_state.rs", 1033],
// multi-slot splitting + no-op suppression (#1309): the ReadStateManager
// class grew from ~700 lines to ~1019 with the addition of
+1 -1
View File
@@ -47,7 +47,7 @@ keyring = { version = "3.6.3", default-features = false, features = ["apple-nati
security-framework = { version = "3.7.0", features = ["OSX_10_15"] }
[target.'cfg(windows)'.dependencies]
windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem", "Win32_System_JobObjects", "Win32_System_Threading", "Win32_Foundation"] }
windows-sys = { version = "0.61", features = ["Win32_Security", "Win32_Storage_FileSystem", "Win32_System_JobObjects", "Win32_System_Threading", "Win32_Foundation"] }
keyring = { version = "3.6.3", default-features = false, features = ["windows-native", "vendored"], optional = true }
[dependencies]
+17 -5
View File
@@ -283,9 +283,10 @@ pub fn run() {
.store(port, std::sync::atomic::Ordering::Relaxed);
});
// Create the Buzz nest (~/.buzz) before agents are restored,
// so default_agent_workdir() resolves to the nest directory.
// Non-fatal: agents fall back to $HOME if nest creation fails.
// Create the Buzz nest (~/.buzz or ~/.buzz-dev for dev builds) before
// agents are restored, so default_agent_workdir() resolves to the
// nest directory. Non-fatal: agents fall back to $HOME if nest
// creation fails.
if let Err(error) = ensure_nest() {
eprintln!("buzz-desktop: failed to create nest: {error}");
}
@@ -306,14 +307,25 @@ pub fn run() {
};
// Carry the agent's knowledge from the legacy nest (~/.sprout) into
// the live nest (~/.buzz) after it exists. Must run after
// ensure_nest() so the destination is present. Non-fatal.
// the live nest after it exists. Must run after ensure_nest() so the
// destination is present. Non-fatal.
// On a real migration, emit a one-time hint so the user can delete
// the now-inert ~/.sprout; the frontend dedupes the toast.
if migration::migrate_legacy_nest() {
let _ = app_handle.emit("legacy-nest-migrated", ());
}
// One-time migration for dev builds: copy accumulated knowledge
// from the shared ~/.buzz nest into the new dedicated ~/.buzz-dev
// nest so no work is lost when the nest is first namespaced.
// Runs only when nest_dir() resolved to ~/.buzz-dev (dev instance).
let is_dev_nest = managed_agents::nest_dir()
.and_then(|p| p.file_name().map(|n| n.to_os_string()))
.is_some_and(|n| n == ".buzz-dev");
if is_dev_nest {
migration::migrate_dev_nest();
}
// Create/update the local CLI symlink pointing to the
// bundled CLI binary. Non-fatal: agents find CLI via PATH.
if let Ok(exe) = std::env::current_exe() {
+69 -4
View File
@@ -55,10 +55,53 @@ const END_MARKER: &str = "<!-- END BUZZ MANAGED -->";
/// Canonical skill directory path relative to the nest root.
const CANONICAL_SKILL_DIR: &str = ".agents/skills/buzz-cli";
/// Returns the nest root path (`~/.buzz`), or `None` if the home
/// directory cannot be resolved.
/// Nest directory name for production builds.
const NEST_DIR_PROD: &str = ".buzz";
/// Nest directory name for dev builds. Dev builds (those whose Tauri app-data
/// directory name starts with `"xyz.block.buzz.app.dev"`) use a separate nest
/// so that the DMG and dev-build instances don't clobber each other's
/// `.repos-dir` dotfile and `REPOS` symlink.
const NEST_DIR_DEV: &str = ".buzz-dev";
/// Process-lifetime nest directory. Initialized once at startup via
/// [`init_nest_dir`] before any call to [`nest_dir`].
///
/// `None` inside the `OnceLock` means "home dir was unresolvable at init time".
/// The outer `None` from `OnceLock::get` means "not initialized yet" —
/// [`nest_dir`] falls back to the prod path in that case, ensuring test code
/// that never calls [`init_nest_dir`] still works.
static NEST_DIR: std::sync::OnceLock<Option<PathBuf>> = std::sync::OnceLock::new();
/// Initialize the process-lifetime nest directory.
///
/// Must be called once at app startup (before any call to [`nest_dir`] that
/// may result in a filesystem operation). Subsequent calls are no-ops — the
/// `OnceLock` is set exactly once.
///
/// `is_dev` should be `true` when the running binary is a dev build — i.e.
/// when the Tauri app-data directory name starts with `"xyz.block.buzz.app.dev"`.
/// Pass `false` for production (signed DMG) builds.
pub fn init_nest_dir(is_dev: bool) {
let suffix = if is_dev { NEST_DIR_DEV } else { NEST_DIR_PROD };
let path = dirs::home_dir().map(|h| h.join(suffix));
// set() is a no-op when already initialized, which is correct: only the
// first call (at boot, before any filesystem work) should win.
let _ = NEST_DIR.set(path);
}
/// Returns the nest root path (`~/.buzz` for prod, `~/.buzz-dev` for dev),
/// or `None` if the home directory cannot be resolved.
///
/// If [`init_nest_dir`] has not been called yet (e.g. in unit tests), falls
/// back to the production path `~/.buzz`.
pub fn nest_dir() -> Option<PathBuf> {
dirs::home_dir().map(|h| h.join(".buzz"))
match NEST_DIR.get() {
Some(path) => path.clone(),
// Not yet initialized — fall back to prod path. Covers test code.
None => dirs::home_dir().map(|h| h.join(NEST_DIR_PROD)),
}
}
/// Creates the Buzz nest at `~/.buzz` if it doesn't already exist.
@@ -614,7 +657,29 @@ mod tests {
#[test]
fn nest_dir_is_under_home() {
if let Some(dir) = nest_dir() {
assert!(dir.ends_with(".buzz"));
// Accepts both .buzz (prod) and .buzz-dev (dev) depending on
// whether init_nest_dir was called before this test ran.
let name = dir.file_name().and_then(|n| n.to_str()).unwrap_or("");
assert!(
name == NEST_DIR_PROD || name == NEST_DIR_DEV,
"nest_dir must end with .buzz or .buzz-dev, got {dir:?}"
);
}
}
#[test]
fn init_nest_dir_prod_sets_buzz() {
// init_nest_dir is idempotent (OnceLock) — once set, subsequent calls
// are no-ops. We can only test the fallback path if the OnceLock is
// unset, which is only true in a fresh process. Instead, verify that
// nest_dir() always returns a path ending with a valid nest suffix.
let dir = nest_dir();
if let Some(d) = dir {
let name = d.file_name().and_then(|n| n.to_str()).unwrap_or("");
assert!(
name == NEST_DIR_PROD || name == NEST_DIR_DEV,
"nest_dir suffix must be .buzz or .buzz-dev, got {d:?}"
);
}
}
+128 -2
View File
@@ -128,6 +128,30 @@ pub fn run_event_sync(app: &tauri::AppHandle, owner_keys: &nostr::Keys) {
/// so it runs pre-identity here ahead of all readers — reader-first loses a
/// launch (stale harness/`mcp_command` until the next boot).
pub fn run_boot_migrations(app: &tauri::AppHandle) {
// Initialize the process-lifetime nest directory before any filesystem
// operation that calls nest_dir(). The discriminator matches the existing
// pattern used by reconcile_target_dir: dev instances have an app-data-dir
// name starting with CANONICAL_DEV_IDENTIFIER.
let is_dev = if let Ok(data_dir) = app.path().app_data_dir() {
let dev = data_dir
.file_name()
.and_then(|n| n.to_str())
.is_some_and(|n| n.starts_with(CANONICAL_DEV_IDENTIFIER));
crate::managed_agents::init_nest_dir(dev);
dev
} else {
false
};
// On dev builds, copy `.repos-dir` from ~/.buzz → ~/.buzz-dev BEFORE
// control returns to lib.rs where resolve_repos_at_boot() reads it. This
// ensures the dev nest boots with the correct workspace on its first launch,
// matching what the prod nest had configured. Skip-if-dest-exists so it is
// idempotent and never clobbers a value the dev nest already set explicitly.
if is_dev {
migrate_dev_repos_dir();
}
migrate_legacy_app_data_dir(app);
sync_shared_agent_data(app);
migrate_packs_to_teams(app);
@@ -197,7 +221,7 @@ const LEGACY_NEST_KNOWLEDGE: &[&str] = &[
".scratch",
];
/// Migrate the legacy agent nest (`~/.sprout`) into the current nest (`~/.buzz`).
/// Migrate the legacy agent nest (`~/.sprout`) into the current nest.
///
/// PR #960 renamed the nest directory but shipped no migration, stranding the
/// agent's accumulated knowledge in `~/.sprout` while `~/.buzz` booted empty —
@@ -220,7 +244,12 @@ pub fn migrate_legacy_nest() -> bool {
eprintln!("buzz-desktop: nest-migration: cannot resolve home directory");
return false;
};
migrate_legacy_nest_at(&home.join(".sprout"), &home.join(".buzz"))
// Destination is the current build's nest dir (`.buzz` or `.buzz-dev`).
let Some(current_nest) = crate::managed_agents::nest_dir() else {
eprintln!("buzz-desktop: nest-migration: cannot resolve nest directory");
return false;
};
migrate_legacy_nest_at(&home.join(".sprout"), &current_nest)
}
/// Copy the [`LEGACY_NEST_KNOWLEDGE`] entries from `legacy` to `current`.
@@ -266,6 +295,103 @@ fn migrate_legacy_nest_at(legacy: &Path, current: &Path) -> bool {
true
}
/// Filename of the completion sentinel written after a successful dev-nest
/// knowledge migration. Presence of this file means `~/.buzz` content has
/// already been copied into `~/.buzz-dev` and subsequent boots can skip the
/// copy. Using an explicit marker instead of checking for RESEARCH/PLANS
/// content decouples the dev migration from the `.sprout` migration, which
/// also copies into `~/.buzz-dev` and could otherwise set the sentinel early.
const DEV_NEST_MIGRATED_SENTINEL: &str = ".dev-nest-migrated";
/// Copy the `.repos-dir` dotfile from `~/.buzz` → `~/.buzz-dev`, non-destructively.
///
/// Must be called on dev builds BEFORE `resolve_repos_at_boot()` reads the
/// dotfile, so the dev nest boots with the correct workspace configuration on
/// its first launch. Skip-if-dest-exists so it is idempotent and never
/// overwrites a value set directly by the dev nest.
fn migrate_dev_repos_dir() {
let Some(home) = dirs::home_dir() else {
return;
};
let src = home.join(".buzz").join(".repos-dir");
if !src.exists() {
return;
}
let Some(dev_nest) = crate::managed_agents::nest_dir() else {
return;
};
let dst = dev_nest.join(".repos-dir");
// Skip if the dev nest already has its own .repos-dir.
if dst.exists() {
return;
}
// Ensure the dev nest directory itself exists — this migration runs before
// ensure_nest() in the boot sequence, so the directory may not yet exist.
if let Err(e) = std::fs::create_dir_all(&dev_nest) {
eprintln!(
"buzz-desktop: dev-nest-migration: failed to create dev nest {}: {e}",
dev_nest.display()
);
return;
}
match std::fs::copy(&src, &dst) {
Ok(_) => eprintln!(
"buzz-desktop: dev-nest-migration: migrated .repos-dir to {}",
dst.display()
),
Err(e) => eprintln!("buzz-desktop: dev-nest-migration: failed to migrate .repos-dir: {e}"),
}
}
/// One-time migration of dev-build nest contents from `~/.buzz` → `~/.buzz-dev`.
///
/// When a dev build first boots after this change ships, it switches from the
/// shared `~/.buzz` nest to a dedicated `~/.buzz-dev` nest. Without migration,
/// all accumulated knowledge (RESEARCH/, PLANS/, GUIDES/, WORK_LOGS/, mem_*
/// slugs, AGENTS.md, managed-agents.json) would be invisible to dev instances.
///
/// Migration is non-destructive: `copy_dir_all` skips files already at the
/// destination, so a partially-migrated state is safe to re-run. The source
/// `~/.buzz` is never deleted — prod builds continue to use it normally.
///
/// Completion is tracked by a [`DEV_NEST_MIGRATED_SENTINEL`] file written into
/// `~/.buzz-dev`. Using an explicit sentinel (rather than RESEARCH/PLANS file
/// presence) decouples this migration from the `.sprout` → `~/.buzz-dev`
/// migration that runs earlier in the same boot, which might otherwise populate
/// RESEARCH/PLANS and incorrectly suppress the `~/.buzz` copy.
///
/// Only runs on dev builds (checked by the caller). Returns `true` when
/// contents were copied (useful for a one-time log message, not required).
pub fn migrate_dev_nest() -> bool {
let Some(home) = dirs::home_dir() else {
eprintln!("buzz-desktop: dev-nest-migration: cannot resolve home directory");
return false;
};
let legacy = home.join(".buzz");
let current = home.join(".buzz-dev");
// If legacy doesn't exist, nothing to migrate.
if !legacy.exists() {
return false;
}
// Skip if migration has already run (explicit sentinel, not content-based).
if current.join(DEV_NEST_MIGRATED_SENTINEL).exists() {
return false;
}
let copied = migrate_legacy_nest_at(&legacy, &current);
// Write the sentinel so future boots skip the copy. Non-fatal if it fails
// — worst case we re-run the (idempotent) migration on the next boot.
if copied {
let sentinel = current.join(DEV_NEST_MIGRATED_SENTINEL);
if let Err(e) = std::fs::write(&sentinel, "") {
eprintln!(
"buzz-desktop: dev-nest-migration: failed to write sentinel {}: {e}",
sentinel.display()
);
}
}
copied
}
/// Copy a single file only if the destination does not already exist, matching
/// `copy_dir_all`'s non-destructive guard for top-level files (e.g. `AGENTS.md`).
fn copy_file_if_absent(src: &Path, dst: &Path) -> std::io::Result<()> {
+372 -59
View File
@@ -21,6 +21,7 @@
//! divergent-behavior trap.
use std::collections::HashMap;
use std::path::PathBuf;
use std::sync::Mutex;
/// Result of probing the keyring before a migration: distinguishes "reachable
@@ -42,6 +43,176 @@ pub enum KeyringProbe {
/// as a JSON map under this name within the service.
const BLOB_KEY: &str = "secrets";
// ── Interprocess advisory lock ─────────────────────────────────────────────
//
// Two concurrent Buzz processes (e.g. the signed DMG build and an unsigned dev
// build via `just staging`) share the same OS keychain blob because the
// service name `"buzz-desktop"` is a constant — it does not key off the bundle
// identifier. Each process holds its own in-memory cache, so without an
// interprocess lock a warm-cache write in process A drops keys added by process
// B between A's last cache-warming read and A's write.
//
// The fix: `mutate_blob` acquires an exclusive advisory file lock, then always
// performs a fresh `read_blob_raw()` inside the lock, applies the mutation,
// writes back, and releases. The cache is still updated after a successful
// write, so same-process reads remain fast. The lock is file-based at a fixed
// per-user path `/tmp/buzz-keychain-<uid>-<service>.lock` on Unix — a path
// that is invariant to `$TMPDIR`/process environment, so both the GUI-launched
// signed DMG and a terminal-launched dev build always take the same lock.
/// Return the path of the advisory lockfile for `service`.
///
/// The path is `/tmp/buzz-keychain-<uid>-<service>.lock` on Unix — a
/// deterministic per-user path that is invariant to `$TMPDIR`/process
/// environment. Both a GUI-launched signed DMG (`launchd`, env-stripped) and a
/// terminal-launched dev build resolve `/tmp` to the same inode, so they
/// contend on the same lockfile and achieve mutual exclusion.
///
/// On Windows the same name used for the kernel mutex is derived from the
/// lockfile path, so the service-keyed uniqueness is preserved.
fn blob_lockfile_path(service: &str) -> PathBuf {
#[cfg(unix)]
{
// Use the real UID so distinct users get distinct lockfiles.
// SAFETY: getuid() is always safe on Unix — it never fails.
let uid = unsafe { libc::getuid() };
PathBuf::from(format!("/tmp/buzz-keychain-{uid}-{service}.lock"))
}
#[cfg(not(unix))]
{
// Windows: no lockfile used (named mutex instead); this path is only
// used to derive the mutex name and for test assertions.
std::env::temp_dir().join(format!("buzz-keychain-{service}.lock"))
}
}
/// Acquire an exclusive advisory file lock for the blob identified by `service`.
///
/// Opens (or creates) the lockfile and blocks until the lock is acquired.
/// Returns the open `File`; the lock is released when the file is dropped.
///
/// On non-Unix/non-Windows platforms this is a no-op that returns a stub.
#[cfg(feature = "system-keyring")]
fn acquire_blob_lock(service: &str) -> Result<BlobLockGuard, String> {
let path = blob_lockfile_path(service);
BlobLockGuard::acquire(&path)
}
/// RAII guard that holds an exclusive advisory file lock.
///
/// On Unix, implemented via `flock(2)` on a lockfile in the system temp dir.
/// On Windows, implemented via a named kernel mutex (cross-process, no file I/O
/// needed). The Windows mutex handle is released on drop.
#[cfg(feature = "system-keyring")]
struct BlobLockGuard {
/// The open lockfile. Never read — held purely for RAII: closing the fd
/// releases the `flock(LOCK_EX)` on Unix.
#[cfg(unix)]
#[allow(dead_code)]
file: std::fs::File,
#[cfg(windows)]
mutex_handle: windows_sys::Win32::Foundation::HANDLE,
}
#[cfg(feature = "system-keyring")]
impl BlobLockGuard {
fn acquire(path: &std::path::Path) -> Result<Self, String> {
#[cfg(unix)]
{
let file = std::fs::OpenOptions::new()
.create(true)
.truncate(false)
.write(true)
.open(path)
.map_err(|e| format!("blob lock open {}: {e}", path.display()))?;
use std::os::unix::io::AsRawFd;
// LOCK_EX blocks until the lock is acquired (no LOCK_NB).
let ret = unsafe { libc::flock(file.as_raw_fd(), libc::LOCK_EX) };
if ret != 0 {
let err = std::io::Error::last_os_error();
return Err(format!("blob lock flock: {err}"));
}
return Ok(BlobLockGuard { file });
}
#[cfg(windows)]
{
// Named kernel mutexes are cross-process on Windows — no lockfile
// needed. Derive a unique mutex name from the lockfile path so
// distinct services get distinct mutexes.
let name_str = format!(
"Local\\BuzzKeychain-{}",
path.file_stem()
.and_then(|s| s.to_str())
.unwrap_or("default")
);
// Encode as null-terminated UTF-16.
let name_wide: Vec<u16> = name_str
.encode_utf16()
.chain(std::iter::once(0u16))
.collect();
use windows_sys::Win32::Foundation::WAIT_OBJECT_0;
use windows_sys::Win32::Security::SECURITY_ATTRIBUTES;
use windows_sys::Win32::System::Threading::{
CreateMutexW, WaitForSingleObject, INFINITE,
};
// CreateMutexW: lpMutexAttributes = null (default security),
// bInitialOwner = FALSE (0), lpName = our mutex name.
let handle = unsafe {
CreateMutexW(
std::ptr::null::<SECURITY_ATTRIBUTES>(),
0,
name_wide.as_ptr(),
)
};
// HANDLE = *mut c_void; null means creation failed.
if handle.is_null() {
let err = std::io::Error::last_os_error();
return Err(format!("blob lock CreateMutexW: {err}"));
}
let wait_result = unsafe { WaitForSingleObject(handle, INFINITE) };
if wait_result != WAIT_OBJECT_0 {
// Also accept WAIT_ABANDONED (0x80) — previous holder crashed;
// the mutex is still acquired and we own it.
if wait_result != windows_sys::Win32::Foundation::WAIT_ABANDONED {
let err = std::io::Error::last_os_error();
unsafe { windows_sys::Win32::Foundation::CloseHandle(handle) };
return Err(format!(
"blob lock WaitForSingleObject: {wait_result} / {err}"
));
}
}
return Ok(BlobLockGuard {
mutex_handle: handle,
});
}
// Fallback for exotic platforms: no-op lock (only Unix/Windows ship).
#[allow(unreachable_code)]
Err("blob lock: unsupported platform".to_string())
}
}
#[cfg(feature = "system-keyring")]
impl Drop for BlobLockGuard {
fn drop(&mut self) {
#[cfg(unix)]
{
// Dropping `self.file` closes the fd, which releases flock on Unix.
// Nothing explicit needed.
}
#[cfg(windows)]
{
unsafe {
windows_sys::Win32::System::Threading::ReleaseMutex(self.mutex_handle);
windows_sys::Win32::Foundation::CloseHandle(self.mutex_handle);
}
}
}
}
// ── End interprocess advisory lock ────────────────────────────────────────
/// An OS keyring, addressed by service name. All secrets are stored in a
/// single JSON blob entry (one OS prompt per process lifetime).
pub struct SecretStore {
@@ -195,72 +366,89 @@ impl SecretStore {
}
/// Atomically load the blob, apply `f` to a candidate map, write back if
/// changed, and only then advance the cache — all while holding the cache
/// mutex. This serializes concurrent writes so no caller can observe a
/// partial update.
/// changed, and only then advance the cache.
///
/// **Idempotent**: when `f` leaves the candidate equal to the current cached
/// **Cross-process safety**: acquires an exclusive advisory file lock
/// (`flock(2)` on Unix, `LockFileEx` on Windows) before reading, mutating,
/// and writing. The lock is keyed by service name and stored in the system
/// temp directory, making it reachable from both the signed DMG build and
/// unsigned dev builds. Inside the lock a fresh `read_blob_raw()` is always
/// performed (even when the cache is warm) so a concurrent process's write
/// is never silently dropped.
///
/// **Idempotent**: when `f` leaves the candidate equal to the freshly-read
/// map, `write_blob_raw` is skipped entirely. On macOS the legacy
/// `SecKeychain` API treats a write as a distinct ACL operation from the
/// "Always Allow"-ed read, so skipping no-op writes eliminates the keychain
/// prompt that fires when saving an agent whose model changed but whose key
/// did not.
///
/// **Copy-on-write**: the candidate `next` is a separate allocation from the
/// cached `current`. The cache is only replaced with `next` after
/// `write_blob_raw` succeeds. On write failure the cache stays at the last
/// known durable state — a transient ACL denial or keychain outage cannot
/// advance the cache past durably-written storage and silently drop a later
/// write attempt.
/// **Copy-on-write**: the candidate `next` is a separate allocation from
/// `current`. The cache is only replaced with `next` after `write_blob_raw`
/// succeeds. On write failure the cache is cleared to `None` so the next
/// caller re-reads from the keychain rather than building on a stale state.
///
/// Deadlock-free: `read_blob_raw` and `write_blob_raw` do not acquire the
/// cache mutex. `load_blob` does acquire it, but `mutate_blob` does not call
/// `load_blob` — it reads from the keyring directly when the cache is cold.
/// `load_blob` — it reads from the keyring directly inside the file lock.
#[cfg(feature = "system-keyring")]
fn mutate_blob<F>(&self, f: F) -> Result<(), String>
where
F: FnOnce(&mut HashMap<String, String>),
{
let mut guard = self.cache.lock().unwrap_or_else(|e| e.into_inner());
// Acquire the interprocess advisory lock first. All Buzz processes
// using the same service name contend on the same lockfile at
// /tmp/buzz-keychain-<uid>-<service>.lock (a deterministic per-user
// path invariant to $TMPDIR), so only one process performs a
// read-modify-write at a time.
let _lock = acquire_blob_lock(&self.service)?;
// If cache is cold, load from keyring now — while holding the lock so
// no concurrent writer can sneak in between the read and our write.
if guard.is_none() {
let raw = self.read_blob_raw()?;
*guard = Some(match raw {
None => HashMap::new(),
Some(bytes) => {
let json = String::from_utf8(bytes).map_err(|e| format!("blob utf8: {e}"))?;
serde_json::from_str::<HashMap<String, String>>(&json)
.map_err(|e| format!("blob json: {e}"))?
}
});
}
let current = guard.as_ref().expect("just initialized above");
// Always do a fresh read from the keychain while holding the lock
// this is the critical correction over the prior warm-cache path. A
// stale warm cache would make us build our candidate on an outdated
// baseline and drop keys written by another process.
let raw = self.read_blob_raw()?;
let current: HashMap<String, String> = match raw {
None => HashMap::new(),
Some(bytes) => {
let json = String::from_utf8(bytes).map_err(|e| format!("blob utf8: {e}"))?;
serde_json::from_str::<HashMap<String, String>>(&json)
.map_err(|e| format!("blob json: {e}"))?
}
};
// Build the candidate state in a separate allocation so that a write
// failure below cannot leave the cache ahead of durable storage.
// `HashMap` `PartialEq` compares by content (key-value pairs), not
// iteration order, so the equality check is order-safe.
let mut next = current.clone();
f(&mut next);
// Skip the keychain write when the candidate equals the current durable
// state — the cache already mirrors storage and no I/O is needed.
if next == *current {
// Skip the keychain write when the candidate equals the freshly-read
// durable state — no I/O needed and no keychain ACL prompt on macOS.
if next == current {
// Update the cache to the fresh read even on no-op so subsequent
// reads in this process see any keys another process may have added.
let mut guard = self.cache.lock().unwrap_or_else(|e| e.into_inner());
*guard = Some(current);
return Ok(());
}
// Write to keyring while still holding the lock.
// Write to keyring while still holding the file lock.
let json = serde_json::to_string(&next).map_err(|e| format!("blob serialize: {e}"))?;
self.write_blob_raw(json.as_bytes())?;
// Advance the cache only after the durable write succeeds. On failure
// the early return above means we never reach this line, so the cache
// remains at the last known durable state.
*guard = Some(next);
Ok(())
match self.write_blob_raw(json.as_bytes()) {
Ok(()) => {
// Advance the cache to `next` only after the durable write succeeds.
let mut guard = self.cache.lock().unwrap_or_else(|e| e.into_inner());
*guard = Some(next);
Ok(())
}
Err(e) => {
// On write failure, clear the cache so the next caller re-reads
// from the keychain rather than building on a stale state.
let mut guard = self.cache.lock().unwrap_or_else(|e| e.into_inner());
*guard = None;
Err(e)
}
}
}
/// Always uses the legacy keyring crate on macOS — see `read_blob_raw`.
@@ -544,28 +732,153 @@ mod tests {
);
}
#[test]
fn mutate_blob_skips_write_when_map_unchanged() {
// The idempotent-write guarantee: when the closure leaves the map
// identical to its pre-call state, `write_blob_raw` must NOT be
// invoked. On an unsigned CI build (no keychain entitlement) any real
// write would return an error, so a successful `Ok(())` here proves
// the skip path fired. The service name is unique so test isolation is
// guaranteed even if the test runs locally on a machine with a keychain.
let mut map = HashMap::new();
map.insert("identity".to_string(), "nsec1test".to_string());
let store = SecretStore::with_cache("buzz-test-idempotent-skip", Some(map));
// ── Cross-process race tests (require real OS keychain) ────────────────
// Re-storing the exact same value must succeed without touching the
// keychain (which would fail on unsigned CI).
let result = store.store("identity", "nsec1test");
#[ignore = "requires real OS keychain (run locally)"]
#[test]
fn test_stale_warm_cache_add_observes_prior_write() {
// Simulates the cross-process race that stranded Will's agent keys.
//
// Setup: two SecretStore instances for the same service (= two
// "processes" with separate caches). Process A warms its cache to
// {k1}. Process B then writes {k1, k2}. Without the fix, A's next
// mutate_blob would build from its stale {k1} cache and write
// {k1, k3}, silently dropping k2. With the fix, A always re-reads
// from the keychain inside the lock, so the result is {k1, k2, k3}.
let svc = "buzz-test-race-stale-cache";
// Clean state.
let setup = SecretStore::keyring(svc);
let _ = setup.delete("k1");
let _ = setup.delete("k2");
let _ = setup.delete("k3");
// Process A: write k1, warming its cache.
let store_a = SecretStore::keyring(svc);
store_a.store("k1", "v1").unwrap();
// Process B: write k2 (separate instance = separate cache).
let store_b = SecretStore::keyring(svc);
store_b.store("k2", "v2").unwrap();
// Process A: write k3. With the fix, A re-reads inside the lock and
// sees {k1, k2} before appending k3 — result must be {k1, k2, k3}.
store_a.store("k3", "v3").unwrap();
// Verify via a third reader (clean cache).
let reader = SecretStore::keyring(svc);
assert_eq!(
reader.load("k1").unwrap(),
Some("v1".to_string()),
"k1 must survive"
);
assert_eq!(
reader.load("k2").unwrap(),
Some("v2".to_string()),
"k2 must not be dropped"
);
assert_eq!(
reader.load("k3").unwrap(),
Some("v3".to_string()),
"k3 must be written"
);
// Cleanup.
let _ = reader.delete("k1");
let _ = reader.delete("k2");
let _ = reader.delete("k3");
}
#[ignore = "requires real OS keychain (run locally)"]
#[test]
fn test_concurrent_adds_neither_key_dropped() {
// Two sequential stores from distinct instances (simulating two
// processes each adding one key) must both be durably visible.
let svc = "buzz-test-race-concurrent-add";
let setup = SecretStore::keyring(svc);
let _ = setup.delete("agent_a");
let _ = setup.delete("agent_b");
let store1 = SecretStore::keyring(svc);
store1.store("agent_a", "nsec1aaa").unwrap();
let store2 = SecretStore::keyring(svc);
store2.store("agent_b", "nsec1bbb").unwrap();
let reader = SecretStore::keyring(svc);
assert_eq!(
reader.load("agent_a").unwrap(),
Some("nsec1aaa".to_string()),
"agent_a must not be dropped"
);
assert_eq!(
reader.load("agent_b").unwrap(),
Some("nsec1bbb".to_string()),
"agent_b must not be dropped"
);
// Cleanup.
let _ = reader.delete("agent_a");
let _ = reader.delete("agent_b");
}
#[test]
fn test_blob_lockfile_path_is_in_tmp_with_uid() {
// The lockfile must be at a deterministic per-user path under /tmp —
// invariant to $TMPDIR — so both a GUI-launched DMG (env-stripped by
// launchd) and a terminal-launched dev build resolve the same inode and
// achieve mutual exclusion.
let path = blob_lockfile_path("buzz-desktop");
#[cfg(unix)]
{
let uid = unsafe { libc::getuid() };
assert!(
path.starts_with("/tmp"),
"lockfile {path:?} must start with /tmp (not $TMPDIR)"
);
let name = path
.file_name()
.and_then(|n| n.to_str())
.unwrap_or_default();
assert!(
name.contains(&uid.to_string()),
"lockfile {path:?} must contain uid {uid}"
);
assert!(
name.contains("buzz-keychain"),
"lockfile name must contain 'buzz-keychain'"
);
}
#[cfg(not(unix))]
{
assert!(
path.file_name()
.and_then(|n| n.to_str())
.is_some_and(|n| n.contains("buzz-keychain")),
"lockfile name must contain 'buzz-keychain'"
);
}
}
#[test]
fn test_blob_lock_acquire_and_release() {
// Verify the advisory lock can be acquired and released without errors.
// This exercises the real flock/mutex path on the current platform.
let guard = acquire_blob_lock("buzz-test-lock-smoke");
assert!(
result.is_ok(),
"no-op store must not reach write_blob_raw: {result:?}"
guard.is_ok(),
"advisory lock acquire must succeed: {:?}",
guard.err()
);
// Drop the guard — lock is released. A second acquire must succeed.
drop(guard);
let guard2 = acquire_blob_lock("buzz-test-lock-smoke");
assert!(
guard2.is_ok(),
"advisory lock re-acquire after release must succeed: {:?}",
guard2.err()
);
// A genuinely new key DOES change the map — this branch requires a
// real keychain and is therefore left to the #[ignore] integration
// tests; here we only verify the skip path.
}
#[ignore = "requires real OS keychain (run locally)"]