diff --git a/artifacts/CHALLENGE_DETAILS.md b/artifacts/CHALLENGE_DETAILS.md new file mode 100644 index 000000000000..747ff45f2437 --- /dev/null +++ b/artifacts/CHALLENGE_DETAILS.md @@ -0,0 +1,62 @@ +# Committed agent-rename recovery + +Upstream report: [zeroclaw-labs/zeroclaw#10373](https://github.com/zeroclaw-labs/zeroclaw/issues/10373). + +Original issue pin: `c1e79a774b8d0a8539481b08838a3c818c97a6d0`. + +Branch base: `fb116d612` (`origin/master`), one commit after that pin. Both patches still apply on this base. + +## Contract + +One runtime-owned recovery record is shared by the CLI, the gateway map-key handlers, and RPC. It is armed when a rename commits (and when a committed rename is discovered with residue or an unreadable follower). It is cleared only after workspace, memory, cron, ACP, and session followers have been read and converged. While it is open, the retired alias cannot be created again, including after residue is deleted out of band. + +Knowledge-graph ownership from unmerged work is out of scope. Followers are the ones on this pin: default per-alias workspace, memory, cron, ACP sessions, and session attribution. + +## Why a naive patch fails + +Copying the existing gateway residue scan into the CLI is not enough: + +- That scan fails open. An unreadable cron database is treated as "no residue", so the rename is reported as not configured and recovery is not armed. The tests require a retryable failure and a refused recreate of the old alias. +- Residue alone cannot tell a stranded rename from an alias that was recreated on purpose. After an out-of-band wipe, a residue scan allows `agents create` of the retired alias. The tests require that create to keep failing until a later rename converges and clears recovery. +- A gateway-only journal is invisible to the CLI, and a CLI-only journal is invisible to the gateway. The cross-surface test arms recovery through the gateway handler and then requires the CLI to refuse the old alias and later finish the same recovery. +- Clearing recovery on the first follower error, or moving a custom workspace path, fails the blocked-workspace and custom-workspace cases. +- HTTP and RPC still return a committed rename when a follower lags, matching the existing gateway tests. The CLI must exit non-zero for that same lag, without the "is not configured" wording. + +The hidden tests call the `zeroclaw` binary, the public gateway handlers, and the public RPC dispatcher. They do not import the recovery module, so a solution that only adds an unwired helper still fails. + +## Patches + +`solution.patch` changes 7 production files: + +- `crates/zeroclaw-runtime/src/agent_rename_recovery.rs` +- `crates/zeroclaw-runtime/src/lib.rs` +- `crates/zeroclaw-runtime/src/rpc/dispatch.rs` +- `crates/zeroclaw-runtime/locales/en/cli.ftl` +- `crates/zeroclaw-gateway/src/api_config.rs` +- `crates/zeroclaw-gateway/src/agent_owned_state.rs` +- `src/alias_cli/mod.rs` + +`test.patch` changes 5 files and does not overlap those paths: + +- `Cargo.toml` +- `test.sh` +- `tests/committed_agent_rename.rs` +- `crates/zeroclaw-gateway/src/lib.rs` +- `crates/zeroclaw-gateway/src/committed_rename_recovery_tests.rs` + +## Harness + +`test.sh --output_path FILE {base|new}` runs, without `set -e`: + +- `cargo test --offline --test committed_agent_rename -- --test-threads=1` +- `cargo test --offline -p zeroclaw-gateway --lib committed_rename -- --test-threads=1` + +It writes JUnit from the cargo `test ... ok|FAILED` lines. Exit status is non-zero when any cargo invocation fails. `base` (tests only) is expected to be non-zero. `new` (tests plus solution) is expected to be zero. + +Build the image with the repository root as context after `test.patch` is applied and before `solution.patch`: + +```sh +docker build -f artifacts/Dockerfile . +``` + +The image is `rust:1.96-bookworm`, `WORKDIR /app`, fetches the locked crates, compiles the harness tests with `--offline --no-run`, and ends with `CMD ["bash"]`. diff --git a/artifacts/Dockerfile b/artifacts/Dockerfile new file mode 100644 index 000000000000..4faa5310588e --- /dev/null +++ b/artifacts/Dockerfile @@ -0,0 +1,15 @@ +FROM rust:1.96-bookworm + +RUN apt-get update \ + && apt-get install -y --no-install-recommends pkg-config libssl-dev \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app + +COPY . . + +RUN cargo fetch --locked \ + && cargo test --offline --no-run --test committed_agent_rename \ + && cargo test --offline --no-run -p zeroclaw-gateway --lib + +CMD ["bash"] diff --git a/artifacts/VERIFY.md b/artifacts/VERIFY.md new file mode 100644 index 000000000000..0b20964fcece --- /dev/null +++ b/artifacts/VERIFY.md @@ -0,0 +1,71 @@ +# Harness verification + +Original issue pin: `c1e79a774b8d0a8539481b08838a3c818c97a6d0`. + +Resynced onto `origin/master` at `fb116d612` (`feat(log): add entry-count rotation and multi-segment log queries (#10214)`). That commit edits `crates/zeroclaw-runtime/src/rpc/dispatch.rs` only around the log-query handlers, not the rename handlers. On this base: + +```sh +git apply --check artifacts/test.patch +git apply --check artifacts/solution.patch +``` + +Both succeeded. The fail-to-pass runs below were executed on the original pin before this rebase. + +Path intersection of `solution.patch` and `test.patch` is empty. + +Solution files (7): + +- `crates/zeroclaw-runtime/src/agent_rename_recovery.rs` +- `crates/zeroclaw-runtime/src/lib.rs` +- `crates/zeroclaw-runtime/src/rpc/dispatch.rs` +- `crates/zeroclaw-runtime/locales/en/cli.ftl` +- `crates/zeroclaw-gateway/src/api_config.rs` +- `crates/zeroclaw-gateway/src/agent_owned_state.rs` +- `src/alias_cli/mod.rs` + +Test files (5): + +- `Cargo.toml` +- `test.sh` +- `tests/committed_agent_rename.rs` +- `crates/zeroclaw-gateway/src/lib.rs` +- `crates/zeroclaw-gateway/src/committed_rename_recovery_tests.rs` + +## Commands + +Tests only (solution reverted, `test.patch` contents present): + +```sh +./test.sh --output_path /tmp/junit-base.xml base +``` + +Exit code: `1`. JUnit: `tests="12" failures="12"`. Every CLI, gateway, and RPC case failed its assertion. Representative base results: + +- CLI resume of a committed rename reported `invalid new alias: alias agent_b already exists` instead of converging followers. +- CLI create after an unreadable cron store printed `created agents.agent_a`. +- Gateway create after an unreadable cron store returned `{"created":true,...}`. +- An unrelated `agent_a -> agent_b` with no residue returned HTTP 400 `alias agent_b already exists` rather than 404 `is not configured`. + +Tests plus solution (`git apply artifacts/solution.patch`): + +```sh +./test.sh --output_path /tmp/junit-new.xml new +``` + +Exit code: `0`. JUnit: `tests="12" failures="0"`. + +`base` is non-zero. `new` is zero. + +## Other checks + +```sh +cargo fmt -- --check +cargo clippy -p zeroclaw-runtime --lib -- -D warnings +cargo clippy -p zeroclaw-gateway --lib --tests --no-deps -- -D warnings +cargo clippy --bin zeroclaw --test committed_agent_rename --no-deps -- -D warnings +bash -n test.sh +``` + +All exited 0. A workspace-wide `cargo clippy -p zeroclaw-gateway --lib --tests -- -D warnings` stopped in pre-existing `zeroclaw-channels` `clippy::drop_non_drop` findings and did not report diagnostics in the recovery files. + +`docker build` was not run. The image recipe is `artifacts/Dockerfile` (`rust:1.96-bookworm`, `WORKDIR /app`, `cargo fetch --locked`, offline `--no-run` for `--test committed_agent_rename` and `-p zeroclaw-gateway --lib`, `CMD ["bash"]`). The fail-to-pass commands above are the ones that image is meant to run. diff --git a/artifacts/solution.patch b/artifacts/solution.patch new file mode 100644 index 000000000000..13d2f330adbc --- /dev/null +++ b/artifacts/solution.patch @@ -0,0 +1,1439 @@ +diff --git a/crates/zeroclaw-gateway/src/agent_owned_state.rs b/crates/zeroclaw-gateway/src/agent_owned_state.rs +index 8bb767838..5b09b8ac5 100644 +--- a/crates/zeroclaw-gateway/src/agent_owned_state.rs ++++ b/crates/zeroclaw-gateway/src/agent_owned_state.rs +@@ -172,84 +172,3 @@ pub async fn cascade_owned_state( + + report + } +- +-/// What the agent-rename owned-state cascade re-pointed +-#[derive(Debug, Default, Clone, serde::Serialize)] +-pub struct RenameStateReport { +- pub memory_rows: usize, +- pub cron_jobs: usize, +- pub acp_sessions: usize, +- pub sessions_repointed: usize, +- /// Surfaced failures. Non-empty means part of the cascade did NOT complete — +- /// those rows were not silently treated as re-pointed. +- pub warnings: Vec, +-} +- +-pub async fn cascade_rename_agent( +- config: &Config, +- mem: &Arc, +- session_backend: Option<&Arc>, +- from: &str, +- to: &str, +-) -> RenameStateReport { +- let mut warnings: Vec = Vec::new(); +- +- let memory_rows = match mem.rename_agent(from, to).await { +- Ok(n) => n, +- Err(e) => { +- warnings.push(format!("memory rename: {e}")); +- 0 +- } +- }; +- +- let cron_jobs = match zeroclaw_runtime::cron::rename_jobs_by_agent(config, from, to) { +- Ok(n) => n, +- Err(e) => { +- warnings.push(format!("cron rename: {e}")); +- 0 +- } +- }; +- +- let acp_sessions = match AcpSessionStore::new(&config.data_dir) { +- Ok(store) => match store.rename_sessions_by_agent(from, to) { +- Ok(n) => n, +- Err(e) => { +- warnings.push(format!("acp rename: {e}")); +- 0 +- } +- }, +- Err(e) => { +- warnings.push(format!("acp store open: {e}")); +- 0 +- } +- }; +- +- let sessions_repointed = match session_backend { +- Some(b) => match b.rename_agent_attribution(from, to) { +- Ok(n) => n, +- Err(e) => { +- warnings.push(format!("session attribution rename: {e}")); +- 0 +- } +- }, +- None => 0, +- }; +- +- if !warnings.is_empty() { +- ::zeroclaw_log::record!( +- WARN, +- ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note) +- .with_outcome(::zeroclaw_log::EventOutcome::Unknown) +- .with_attrs(::serde_json::json!({"from": from, "to": to, "warnings": warnings})), +- "rename owned-state cascade completed with warnings (some state may not have been re-pointed)" +- ); +- } +- +- RenameStateReport { +- memory_rows, +- cron_jobs, +- acp_sessions, +- sessions_repointed, +- warnings, +- } +-} +diff --git a/crates/zeroclaw-gateway/src/api_config.rs b/crates/zeroclaw-gateway/src/api_config.rs +index d6e7a65ae..31b4403c1 100644 +--- a/crates/zeroclaw-gateway/src/api_config.rs ++++ b/crates/zeroclaw-gateway/src/api_config.rs +@@ -1360,6 +1360,25 @@ pub async fn handle_map_key( + let path = q.path.clone(); + let key = q.key.clone(); + ++ if path == "agents" { ++ match zeroclaw_runtime::agent_rename_recovery::alias_reuse_blocked(&working.data_dir, &key) ++ .await ++ { ++ Ok(true) => { ++ let alias = key.clone(); ++ return recovery_error_response( ++ &path, ++ &key, ++ zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError::UnsafeReuse { ++ alias, ++ }, ++ ); ++ } ++ Ok(false) => {} ++ Err(err) => return recovery_error_response(&path, &key, err), ++ } ++ } ++ + // Create through the shared guarded boundary so the reserved-agent rule (the + // `default` runtime fallback) is enforced once for every surface. Reserved -> + // 400 (validation_failed), symmetric with the rename guard; an unknown +@@ -1628,7 +1647,9 @@ pub async fn handle_rename_map_key( + + match zeroclaw_config::alias_refs::alias_kind_for_map_path(&body.path) { + Some(zeroclaw_config::alias_refs::AliasKind::Agent) => { +- rename_agent_cascade(&state, working, &body, _cfg_guard).await ++ // The cascade keeps config snapshots across awaits. Box it so that ++ // state stays off this task's stack. ++ Box::pin(rename_agent_cascade(&state, working, &body, _cfg_guard)).await + } + Some(kind) => rename_config_cascade(&state, working, &kind, &body, &_cfg_guard).await, + None => { +@@ -1697,85 +1718,35 @@ async fn rename_config_cascade( + .into_response() + } + +-async fn move_renamed_workspace( +- old_ws: &std::path::Path, +- new_ws: &std::path::Path, +-) -> Option { +- if old_ws == new_ws || !old_ws.exists() { +- return None; +- } +- if let Some(parent) = new_ws.parent() { +- let _ = tokio::fs::create_dir_all(parent).await; +- } +- match tokio::fs::rename(old_ws, new_ws).await { +- Ok(()) => None, +- Err(err) => { +- ::zeroclaw_log::record!( +- WARN, +- ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note) +- .with_outcome(::zeroclaw_log::EventOutcome::Unknown) +- .with_attrs(::serde_json::json!({ +- "old": old_ws.display().to_string(), +- "new": new_ws.display().to_string(), +- "err": err.to_string() +- })), +- "agent rename: workspace move failed" +- ); +- Some(format!( +- "workspace move {} -> {} failed: {err}", +- old_ws.display(), +- new_ws.display() +- )) +- } +- } +-} +- +-async fn rename_residue_exists( +- state: &AppState, +- working: &zeroclaw_config::schema::Config, ++fn recovery_error_response( ++ path: &str, + from: &str, +-) -> bool { +- // Workspace: the default per-alias dir for `from`. A custom/alias-independent +- // path is not moved by the cascade, so it is not residue. +- if working.agent_workspace_dir(from).exists() { +- return true; +- } +- +- // Short-lived clone for the DB-backed stores - never hold the lock across an +- // `.await`. +- let cfg = state.config.read().clone(); +- +- // Cron jobs still owned by `from`. +- if zeroclaw_runtime::cron::list_jobs_by_agent(&cfg, from) +- .map(|jobs| !jobs.is_empty()) +- .unwrap_or(false) +- { +- return true; +- } +- +- // ACP sessions (live OR killed) still owned by `from`. +- if let Ok(store) = zeroclaw_infra::acp_session_store::AcpSessionStore::new(&cfg.data_dir) +- && store +- .list_sessions_by_agent(from) +- .map(|s| !s.is_empty()) +- .unwrap_or(false) +- { +- return true; +- } +- +- // Memory rows still attributed to `from`. +- if state.mem.count_agent(from).await.unwrap_or(0) > 0 { +- return true; +- } ++ err: zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError, ++) -> Response { ++ use zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError; ++ let (code, msg) = match err { ++ RenameRecoveryError::NotConfigured { .. } => ( ++ ConfigApiCode::PathNotFound, ++ format!("agents.{from} is not configured"), ++ ), ++ RenameRecoveryError::UnsafeReuse { alias } => ( ++ ConfigApiCode::ValidationFailed, ++ format!( ++ "alias `{alias}` is stranded by an unfinished agent rename and cannot be reused yet" ++ ), ++ ), ++ other => (ConfigApiCode::InternalError, other.to_string()), ++ }; ++ error_response(ConfigApiError::new(code, msg).with_path(format!("{path}.{from}"))) ++} + +- // Session-metadata attribution still pointing at `from`. +- if let Some(backend) = state.session_backend.as_ref() +- && backend.count_agent_attribution(from).unwrap_or(0) > 0 +- { +- return true; ++fn follower_stores<'a>( ++ state: &'a AppState, ++) -> zeroclaw_runtime::agent_rename_recovery::FollowerStores<'a> { ++ zeroclaw_runtime::agent_rename_recovery::FollowerStores { ++ memory: Some(state.mem.as_ref()), ++ session_backend: state.session_backend.as_deref(), + } +- +- false + } + + async fn rename_agent_cascade( +@@ -1785,80 +1756,94 @@ async fn rename_agent_cascade( + guard: ConfigWriteGuard, + ) -> Response { + use zeroclaw_config::alias_refs::{self, AliasKind}; ++ use zeroclaw_runtime::agent_rename_recovery::{self, RenameDisposition, RenameRecoveryError}; + let (from, to) = (&body.from, &body.to); ++ let stores = follower_stores(state); + +- // Capture the OLD workspace path while the entry still lives under `from` +- // (custom paths are read off the entry, which is about to move). ++ // Capture the old workspace while `from` still exists. A custom path is ++ // read off that entry; after the commit it would fall back to the default. + let old_ws = working.agent_workspace_dir(from); ++ let disposition = ++ match agent_rename_recovery::resolve_committed_rename(&working, from, to, &stores).await { ++ Ok(disposition) => disposition, ++ Err(RenameRecoveryError::Incomplete { warnings }) => { ++ return rename_partial_response(body, from, to, 0, false, 0, 0, 0, 0, warnings); ++ } ++ Err(err) => return recovery_error_response(&body.path, from, err), ++ }; + +- let committed_to = working.agent(from).is_none() && working.agent(to).is_some(); +- let dirty_count = if committed_to && rename_residue_exists(state, &working, from).await { +- 0 +- } else { ++ let dirty_count = if matches!(disposition, RenameDisposition::Fresh { .. }) { + match alias_refs::rename_with_cascade(&mut working, &AliasKind::Agent, from, to) { + Ok(report) => { + for path in &report.dirty_paths { + working.mark_dirty(path); + } + let dirty_count = report.dirty_paths.len(); +- if let Err(e) = persist_and_swap(state, working, &guard).await { +- return error_response(e); ++ if let Err(err) = persist_and_swap(state, working, &guard).await { ++ return error_response(err); ++ } ++ let committed = state.config.read().clone(); ++ if let Err(err) = ++ agent_rename_recovery::arm_committed_rename(&committed, from, to, &old_ws).await ++ { ++ return recovery_error_response(&body.path, from, err); + } + dirty_count + } +- Err(e) => return rename_error_response(&body.path, from, e), ++ Err(err) => return rename_error_response(&body.path, from, err), + } ++ } else { ++ 0 + }; +- // Config is committed (saved + swapped, or already committed by a prior +- // crashed run). Release before the post-commit side effects below: +- // workspace move and the memory/cron/ACP/session-backend cascade can be +- // slow or wedge, and holding the lock across them would stall every +- // other gateway config write process-wide. ++ // Config is committed. Release before follower I/O so a slow converge ++ // cannot stall every other gateway config write. + drop(guard); + + let cfg = state.config.read().clone(); +- // The NEW workspace path off the committed config (the rewritten `to`). +- let new_ws = cfg.agent_workspace_dir(to); +- +- // Move the workspace dir. For the default per-alias location this is +- // `/agents//workspace` → `…//workspace`. A custom +- // workspace path is alias-independent, so `old_ws == new_ws` and we skip. +- let ws_existed = old_ws != new_ws && old_ws.exists(); +- let move_warning = move_renamed_workspace(&old_ws, &new_ws).await; +- let workspace_moved = ws_existed && move_warning.is_none(); +- let mut warnings: Vec = Vec::new(); +- warnings.extend(move_warning); ++ let stores = follower_stores(state); ++ match agent_rename_recovery::converge_committed_rename(&cfg, from, to, &stores).await { ++ Ok(report) => rename_partial_response( ++ body, ++ from, ++ to, ++ dirty_count, ++ report.workspace_moved, ++ report.memory_rows, ++ report.cron_jobs, ++ report.acp_sessions, ++ report.sessions_repointed, ++ Vec::new(), ++ ), ++ Err(RenameRecoveryError::Incomplete { warnings }) => { ++ // Config is `to` and the recovery record stays armed, so a retry ++ // (or the CLI) can finish. Surface the lag instead of hiding it. ++ rename_partial_response(body, from, to, dirty_count, false, 0, 0, 0, 0, warnings) ++ } ++ Err(err) => recovery_error_response(&body.path, from, err), ++ } ++} + +- // Re-point owned DB state (memory/cron/acp/session). Best-effort + reported. +- let owned = crate::agent_owned_state::cascade_rename_agent( +- &cfg, +- &state.mem, +- state.session_backend.as_ref(), +- from, +- to, +- ) +- .await; +- // Combine the workspace-move warning (if any) with the owned-store warnings +- // so every partial failure reaches the caller, not just the server log. +- warnings.extend(owned.warnings); +- +- // The config rename committed. A non-empty `warnings` means a post-persist +- // side-effect did not follow (config is `to`, some follower lags at `from`, +- // re-runnable) - escalate to WARN so that degraded outcome is visible +- // operationally instead of buried at INFO. ++fn rename_partial_response( ++ body: &RenameMapKeyBody, ++ from: &str, ++ to: &str, ++ dirty_count: usize, ++ workspace_moved: bool, ++ memory_rows: usize, ++ cron_jobs: usize, ++ acp_sessions: usize, ++ sessions_repointed: usize, ++ warnings: Vec, ++) -> Response { + if warnings.is_empty() { +- ::zeroclaw_log::record!(INFO, ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note).with_attrs(::serde_json::json!({"from": from, "to": to, "memory": owned.memory_rows, "cron": owned.cron_jobs, "acp": owned.acp_sessions, "sessions": owned.sessions_repointed, "workspace_moved": workspace_moved, "dirty_paths": dirty_count})), "agent renamed with owned-state cascade"); ++ ::zeroclaw_log::record!(INFO, ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note).with_attrs(::serde_json::json!({"from": from, "to": to, "memory": memory_rows, "cron": cron_jobs, "acp": acp_sessions, "sessions": sessions_repointed, "workspace_moved": workspace_moved, "dirty_paths": dirty_count})), "agent renamed with owned-state cascade"); + } else { +- ::zeroclaw_log::record!(WARN, ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note).with_attrs(::serde_json::json!({"from": from, "to": to, "memory": owned.memory_rows, "cron": owned.cron_jobs, "acp": owned.acp_sessions, "sessions": owned.sessions_repointed, "workspace_moved": workspace_moved, "dirty_paths": dirty_count, "warnings": warnings})), "agent rename persisted but a post-persist side-effect did not follow; re-issue the rename to converge"); ++ ::zeroclaw_log::record!(WARN, ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note).with_attrs(::serde_json::json!({"from": from, "to": to, "memory": memory_rows, "cron": cron_jobs, "acp": acp_sessions, "sessions": sessions_repointed, "workspace_moved": workspace_moved, "dirty_paths": dirty_count, "warnings": warnings})), "agent rename persisted but a post-persist side-effect did not follow; re-issue the rename to converge"); + } +- +- // Persisted rename. `warnings` carries any post-persist side-effect that did +- // not follow, so the split can be remediated rather than reported as a clean +- // success (207-style partial success). + axum::Json(RenameMapKeyResponse { + path: body.path.clone(), +- from: from.clone(), +- to: to.clone(), ++ from: from.to_string(), ++ to: to.to_string(), + renamed: true, + warnings, + }) +@@ -3397,29 +3382,51 @@ mod tests { + #[tokio::test] + async fn renamed_workspace_move_failure_is_surfaced() { + // A failed workspace move during rename must surface a warning (so the +- // caller learns config/DB moved to `to` while the workspace is stranded +- // at `from`), not be swallowed as a clean success. ++ // caller learns config moved to `to` while the workspace is stranded ++ // at `from`), not be swallowed as a clean success with an empty warning ++ // list. + let tmp = tempfile::tempdir().unwrap(); +- let old_ws = tmp.path().join("from-ws"); ++ let mut config = zeroclaw_config::schema::Config { ++ config_path: tmp.path().join("config.toml"), ++ data_dir: tmp.path().join("data"), ++ ..Default::default() ++ }; ++ std::fs::create_dir_all(&config.data_dir).unwrap(); ++ config.agents.insert( ++ "from".to_string(), ++ zeroclaw_config::schema::AliasedAgentConfig { ++ risk_profile: "default".into(), ++ ..Default::default() ++ }, ++ ); ++ config.risk_profiles.entry("default".into()).or_default(); ++ config.runtime_profiles.entry("default".into()).or_default(); ++ let old_ws = config.agent_workspace_dir("from"); + std::fs::create_dir_all(&old_ws).unwrap(); +- // Force the move to fail: new_ws's parent is a FILE, so create_dir_all +- // and rename both fail. +- let blocker = tmp.path().join("blocker"); ++ // Force the move to fail: the destination alias directory is a file, ++ // so creating its workspace parent cannot succeed. ++ let blocker = tmp.path().join("agents").join("to"); ++ if let Some(parent) = blocker.parent() { ++ std::fs::create_dir_all(parent).unwrap(); ++ } + std::fs::write(&blocker, b"x").unwrap(); +- let new_ws = blocker.join("to-ws"); + +- let warning = move_renamed_workspace(&old_ws, &new_ws).await; ++ let state = crate::api::test_state(config.clone()); ++ let body = RenameMapKeyBody { ++ path: "agents".to_string(), ++ from: "from".to_string(), ++ to: "to".to_string(), ++ }; ++ let guard = Arc::clone(&state.config_write_lock).lock_owned().await; ++ let resp = rename_agent_cascade(&state, config, &body, guard).await; ++ let (status, json) = response_json(resp).await; ++ assert!(status.is_success(), "config commit still succeeds"); ++ let warnings = json["warnings"].to_string(); + assert!( +- warning.is_some(), +- "a failed workspace move must surface a warning" ++ warnings.contains("workspace move"), ++ "a failed workspace move must surface a warning, got {warnings}" + ); +- assert!(warning.unwrap().contains("workspace move")); + assert!(old_ws.exists(), "source dir stays put when the move fails"); +- +- // Nothing-to-move paths return None (no spurious warning). +- assert!(move_renamed_workspace(&old_ws, &old_ws).await.is_none()); +- let missing = tmp.path().join("does-not-exist"); +- assert!(move_renamed_workspace(&missing, &new_ws).await.is_none()); + } + + #[tokio::test] +diff --git a/crates/zeroclaw-runtime/locales/en/cli.ftl b/crates/zeroclaw-runtime/locales/en/cli.ftl +index fb525fe5a..179578873 100644 +--- a/crates/zeroclaw-runtime/locales/en/cli.ftl ++++ b/crates/zeroclaw-runtime/locales/en/cli.ftl +@@ -1106,6 +1106,10 @@ cli-alias-deleted = deleted {$section}.{$alias} (scrubbed {$count} reference(s)) + cli-alias-delete-refused-header = refused: {$count} hard reference(s) block the delete: + cli-alias-delete-refused-hint = delete refused — resolve the hard references first + cli-alias-not-configured = {$path} is not configured ++cli-alias-rename-recovery-unreadable = committed agent rename could not read {$store}: {$detail} ++cli-alias-rename-recovery-incomplete = committed agent rename did not finish; retry the same rename to converge ++cli-alias-rename-recovery-reuse = alias `{$alias}` is stranded by an unfinished agent rename and cannot be reused yet ++cli-alias-rename-recovery-persist = committed agent rename recovery could not be recorded: {$detail} + cli-alias-delete-failed = delete failed: {$error} + cli-alias-delete-reserved-default = the `default` agent is reserved and cannot be deleted + cli-alias-create-reserved-default = the `default` agent is reserved and cannot be created +diff --git a/crates/zeroclaw-runtime/src/agent_rename_recovery.rs b/crates/zeroclaw-runtime/src/agent_rename_recovery.rs +new file mode 100644 +index 000000000..aa7c8d531 +--- /dev/null ++++ b/crates/zeroclaw-runtime/src/agent_rename_recovery.rs +@@ -0,0 +1,519 @@ ++//! Committed agent-rename recovery shared by the CLI, gateway, and RPC. ++//! ++//! A rename persists the configuration alias change before workspace, memory, ++//! cron, ACP, and session followers finish. This module is the one record of ++//! that in-progress rename: callers arm it when the config commit succeeds, ++//! resume from it (or from leftover follower residue) instead of treating the ++//! retired alias as unknown, and delete it only after every configured ++//! follower has been read and converged. While the record is open the retired ++//! alias cannot be created again. ++ ++use std::path::{Path, PathBuf}; ++ ++use serde::{Deserialize, Serialize}; ++use zeroclaw_api::memory_traits::Memory; ++use zeroclaw_config::schema::Config; ++use zeroclaw_infra::acp_session_store::AcpSessionStore; ++use zeroclaw_infra::session_backend::SessionBackend; ++ ++const JOURNAL_FILE: &str = "agent_rename_recovery.json"; ++const UNSUPPORTED_MEMORY_RENAME: &str = "rename_agent not supported by this memory backend"; ++ ++/// Followers the caller has open. `None` means that store is not configured ++/// on this surface, not that a read failed. ++pub struct FollowerStores<'a> { ++ pub memory: Option<&'a dyn Memory>, ++ pub session_backend: Option<&'a dyn SessionBackend>, ++} ++ ++/// What the caller should do with live configuration before converging. ++#[derive(Debug)] ++pub enum RenameDisposition { ++ /// `from` is still configured. Cascade and persist config, arm recovery, ++ /// then converge. ++ Fresh { old_workspace: PathBuf }, ++ /// Config already names `to`. Do not rewrite config; converge followers. ++ Resume, ++} ++ ++/// Counts from a converge that cleared the recovery record. ++#[derive(Debug, Default)] ++pub struct ConvergeReport { ++ pub memory_rows: usize, ++ pub cron_jobs: usize, ++ pub acp_sessions: usize, ++ pub sessions_repointed: usize, ++ pub workspace_moved: bool, ++} ++ ++#[derive(Debug)] ++pub enum RenameRecoveryError { ++ NotConfigured { path: String }, ++ Unreadable { store: String, detail: String }, ++ UnsafeReuse { alias: String }, ++ Incomplete { warnings: Vec }, ++ Persist { detail: String }, ++} ++ ++impl std::fmt::Display for RenameRecoveryError { ++ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { ++ match self { ++ Self::NotConfigured { path } => write!(f, "{path} is not configured"), ++ Self::Unreadable { store, detail } => { ++ write!( ++ f, ++ "committed rename recovery could not read {store}: {detail}" ++ ) ++ } ++ Self::UnsafeReuse { alias } => write!( ++ f, ++ "alias `{alias}` is stranded by an unfinished agent rename and cannot be reused yet" ++ ), ++ Self::Incomplete { warnings } => write!( ++ f, ++ "committed agent rename did not finish ({})", ++ warnings.join("; ") ++ ), ++ Self::Persist { detail } => { ++ write!( ++ f, ++ "committed rename recovery could not be recorded: {detail}" ++ ) ++ } ++ } ++ } ++} ++ ++impl std::error::Error for RenameRecoveryError {} ++ ++#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] ++struct RecoveryRecord { ++ from: String, ++ to: String, ++ old_workspace: String, ++ move_workspace: bool, ++} ++ ++#[derive(Debug, Default, Serialize, Deserialize)] ++struct Journal { ++ records: Vec, ++} ++ ++fn journal_path(data_dir: &Path) -> PathBuf { ++ data_dir.join(JOURNAL_FILE) ++} ++ ++fn unreadable(store: &str, detail: impl ToString) -> RenameRecoveryError { ++ RenameRecoveryError::Unreadable { ++ store: store.to_string(), ++ detail: detail.to_string(), ++ } ++} ++ ++async fn load_journal(data_dir: &Path) -> Result { ++ let path = journal_path(data_dir); ++ match tokio::fs::read(&path).await { ++ Ok(bytes) => serde_json::from_slice(&bytes).map_err(|err| unreadable("recovery", err)), ++ Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(Journal::default()), ++ Err(err) => Err(unreadable("recovery", err)), ++ } ++} ++ ++async fn store_journal(data_dir: &Path, journal: &Journal) -> Result<(), RenameRecoveryError> { ++ if let Err(err) = tokio::fs::create_dir_all(data_dir).await { ++ return Err(RenameRecoveryError::Persist { ++ detail: err.to_string(), ++ }); ++ } ++ let path = journal_path(data_dir); ++ let tmp = path.with_extension("json.tmp"); ++ let bytes = serde_json::to_vec_pretty(journal).map_err(|err| RenameRecoveryError::Persist { ++ detail: err.to_string(), ++ })?; ++ if let Err(err) = tokio::fs::write(&tmp, &bytes).await { ++ return Err(RenameRecoveryError::Persist { ++ detail: err.to_string(), ++ }); ++ } ++ tokio::fs::rename(&tmp, &path) ++ .await ++ .map_err(|err| RenameRecoveryError::Persist { ++ detail: err.to_string(), ++ }) ++} ++ ++fn upsert(journal: &mut Journal, record: RecoveryRecord) { ++ if let Some(existing) = journal ++ .records ++ .iter_mut() ++ .find(|row| row.from == record.from) ++ { ++ *existing = record; ++ } else { ++ journal.records.push(record); ++ } ++} ++ ++/// `true` when `alias` is the retired source of an open committed rename. ++pub async fn alias_reuse_blocked( ++ data_dir: &Path, ++ alias: &str, ++) -> Result { ++ let journal = load_journal(data_dir).await?; ++ Ok(journal.records.iter().any(|row| row.from == alias)) ++} ++ ++/// Remember a committed `from` → `to` rename until followers converge. ++pub async fn arm_committed_rename( ++ config: &Config, ++ from: &str, ++ to: &str, ++ old_workspace: &Path, ++) -> Result<(), RenameRecoveryError> { ++ let new_workspace = config.agent_workspace_dir(to); ++ let record = RecoveryRecord { ++ from: from.to_string(), ++ to: to.to_string(), ++ old_workspace: old_workspace.display().to_string(), ++ move_workspace: old_workspace != new_workspace.as_path(), ++ }; ++ let mut journal = load_journal(&config.data_dir).await?; ++ upsert(&mut journal, record); ++ store_journal(&config.data_dir, &journal).await ++} ++ ++async fn clear_record(config: &Config, from: &str, to: &str) -> Result<(), RenameRecoveryError> { ++ let mut journal = load_journal(&config.data_dir).await?; ++ let before = journal.records.len(); ++ journal ++ .records ++ .retain(|row| !(row.from == from && row.to == to)); ++ if journal.records.len() == before { ++ return Ok(()); ++ } ++ if journal.records.is_empty() { ++ let path = journal_path(&config.data_dir); ++ match tokio::fs::remove_file(&path).await { ++ Ok(()) => Ok(()), ++ Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(()), ++ Err(err) => Err(RenameRecoveryError::Persist { ++ detail: err.to_string(), ++ }), ++ } ++ } else { ++ store_journal(&config.data_dir, &journal).await ++ } ++} ++ ++/// Residue and readability of the followers a committed rename must move. ++/// ++/// An unreadable configured store is an error. It is not "no residue". ++async fn probe_followers( ++ config: &Config, ++ from: &str, ++ stores: &FollowerStores<'_>, ++) -> Result { ++ let mut residue = config.agent_workspace_dir(from).exists(); ++ ++ match crate::cron::list_jobs_by_agent(config, from) { ++ Ok(jobs) if !jobs.is_empty() => residue = true, ++ Ok(_) => {} ++ Err(err) => return Err(unreadable("cron", err)), ++ } ++ ++ if let Some(memory) = stores.memory { ++ match memory.count_agent(from).await { ++ Ok(0) => {} ++ Ok(_) => residue = true, ++ Err(err) => return Err(unreadable("memory", err)), ++ } ++ } ++ ++ let acp_db = config.data_dir.join("sessions").join("acp-sessions.db"); ++ if acp_db.exists() { ++ match AcpSessionStore::new(&config.data_dir) { ++ Ok(store) => match store.list_sessions_by_agent(from) { ++ Ok(rows) if !rows.is_empty() => residue = true, ++ Ok(_) => {} ++ Err(err) => return Err(unreadable("acp", err)), ++ }, ++ Err(err) => return Err(unreadable("acp", err)), ++ } ++ } ++ ++ if let Some(backend) = stores.session_backend { ++ match backend.count_agent_attribution(from) { ++ Ok(0) => {} ++ Ok(_) => residue = true, ++ Err(err) => return Err(unreadable("sessions", err)), ++ } ++ } ++ ++ Ok(residue) ++} ++ ++fn fresh(config: &Config, from: &str) -> RenameDisposition { ++ RenameDisposition::Fresh { ++ old_workspace: config.agent_workspace_dir(from), ++ } ++} ++ ++/// Decide whether `from` → `to` is a new rename, a resume, or a hard failure. ++pub async fn resolve_committed_rename( ++ config: &Config, ++ from: &str, ++ to: &str, ++ stores: &FollowerStores<'_>, ++) -> Result { ++ let journal = load_journal(&config.data_dir).await?; ++ if journal ++ .records ++ .iter() ++ .any(|row| row.from == to && row.from != from) ++ { ++ return Err(RenameRecoveryError::UnsafeReuse { ++ alias: to.to_string(), ++ }); ++ } ++ ++ if let Some(record) = journal.records.iter().find(|row| row.from == from).cloned() { ++ return resolve_open_record(config, from, to, &record); ++ } ++ ++ if config.agent(from).is_some() { ++ return Ok(fresh(config, from)); ++ } ++ if config.agent(to).is_none() { ++ return Err(RenameRecoveryError::NotConfigured { ++ path: format!("agents.{from}"), ++ }); ++ } ++ ++ // Config already committed `to` and there is no journal yet. Residue or an ++ // unreadable store still belongs to this rename; a clean readable scan does ++ // not. ++ match probe_followers(config, from, stores).await { ++ Ok(false) => Err(RenameRecoveryError::NotConfigured { ++ path: format!("agents.{from}"), ++ }), ++ Ok(true) => { ++ let old_workspace = config.agent_workspace_dir(from); ++ arm_committed_rename(config, from, to, &old_workspace).await?; ++ Ok(RenameDisposition::Resume) ++ } ++ Err(err) => { ++ let old_workspace = config.agent_workspace_dir(from); ++ arm_committed_rename(config, from, to, &old_workspace).await?; ++ Err(err) ++ } ++ } ++} ++ ++fn resolve_open_record( ++ config: &Config, ++ from: &str, ++ to: &str, ++ record: &RecoveryRecord, ++) -> Result { ++ let from_live = config.agent(from).is_some(); ++ let to_live = config.agent(to).is_some(); ++ let recorded_to_live = config.agent(&record.to).is_some(); ++ ++ if record.to != to { ++ // The config commit has not landed, so a different target is still a ++ // fresh rename and replaces the uncommitted record. ++ if from_live && !recorded_to_live && !to_live { ++ return Ok(fresh(config, from)); ++ } ++ return Err(RenameRecoveryError::UnsafeReuse { ++ alias: from.to_string(), ++ }); ++ } ++ ++ if from_live && to_live { ++ return Err(RenameRecoveryError::UnsafeReuse { ++ alias: from.to_string(), ++ }); ++ } ++ if from_live && !to_live { ++ return Ok(RenameDisposition::Fresh { ++ old_workspace: PathBuf::from(&record.old_workspace), ++ }); ++ } ++ if !to_live { ++ return Err(unreadable( ++ "recovery", ++ "committed rename target is missing from live configuration", ++ )); ++ } ++ Ok(RenameDisposition::Resume) ++} ++ ++fn memory_skips_agents(err: &anyhow::Error) -> bool { ++ err.to_string().contains(UNSUPPORTED_MEMORY_RENAME) ++} ++ ++async fn converge_memory( ++ stores: &FollowerStores<'_>, ++ from: &str, ++ to: &str, ++ warnings: &mut Vec, ++) -> usize { ++ let Some(memory) = stores.memory else { ++ return 0; ++ }; ++ let pending = match memory.count_agent(from).await { ++ Ok(count) => count, ++ Err(err) => { ++ warnings.push(format!("memory count: {err}")); ++ return 0; ++ } ++ }; ++ // Nothing attributed to `from`. Skip the backend call: a zero-row rename is ++ // not always a no-op (sqlite drops an empty destination alias row). ++ if pending == 0 { ++ return 0; ++ } ++ match memory.rename_agent(from, to).await { ++ Ok(rows) => rows, ++ Err(err) if memory_skips_agents(&err) => 0, ++ Err(err) => { ++ warnings.push(format!("memory rename: {err}")); ++ 0 ++ } ++ } ++} ++ ++fn converge_cron(config: &Config, from: &str, to: &str, warnings: &mut Vec) -> usize { ++ match crate::cron::rename_jobs_by_agent(config, from, to) { ++ Ok(rows) => rows, ++ Err(err) => { ++ warnings.push(format!("cron rename: {err}")); ++ 0 ++ } ++ } ++} ++ ++fn converge_acp(config: &Config, from: &str, to: &str, warnings: &mut Vec) -> usize { ++ match AcpSessionStore::new(&config.data_dir) { ++ Ok(store) => match store.rename_sessions_by_agent(from, to) { ++ Ok(rows) => rows, ++ Err(err) => { ++ warnings.push(format!("acp rename: {err}")); ++ 0 ++ } ++ }, ++ Err(err) => { ++ warnings.push(format!("acp store open: {err}")); ++ 0 ++ } ++ } ++} ++ ++fn converge_sessions( ++ stores: &FollowerStores<'_>, ++ from: &str, ++ to: &str, ++ warnings: &mut Vec, ++) -> usize { ++ let Some(backend) = stores.session_backend else { ++ return 0; ++ }; ++ match backend.rename_agent_attribution(from, to) { ++ Ok(rows) => rows, ++ Err(err) => { ++ warnings.push(format!("session attribution rename: {err}")); ++ 0 ++ } ++ } ++} ++ ++async fn relocate_workspace(old_workspace: &Path, new_workspace: &Path) -> Result { ++ if !old_workspace.exists() { ++ return Ok(false); ++ } ++ if let Some(parent) = new_workspace.parent() ++ && let Err(err) = tokio::fs::create_dir_all(parent).await ++ { ++ return Err(format!( ++ "workspace move {} -> {} failed: {err}", ++ old_workspace.display(), ++ new_workspace.display() ++ )); ++ } ++ match tokio::fs::rename(old_workspace, new_workspace).await { ++ Ok(()) => Ok(true), ++ Err(err) => Err(format!( ++ "workspace move {} -> {} failed: {err}", ++ old_workspace.display(), ++ new_workspace.display() ++ )), ++ } ++} ++ ++/// Move the default workspace and re-point owned followers. Idempotent. ++/// ++/// The recovery record is removed only when every configured follower was ++/// readable and no old-alias residue remains. ++pub async fn converge_committed_rename( ++ config: &Config, ++ from: &str, ++ to: &str, ++ stores: &FollowerStores<'_>, ++) -> Result { ++ let journal = load_journal(&config.data_dir).await?; ++ let record = journal ++ .records ++ .iter() ++ .find(|row| row.from == from && row.to == to) ++ .cloned(); ++ let new_workspace = config.agent_workspace_dir(to); ++ let (old_workspace, move_workspace) = if let Some(record) = &record { ++ (PathBuf::from(&record.old_workspace), record.move_workspace) ++ } else { ++ let old_workspace = config.agent_workspace_dir(from); ++ let move_workspace = old_workspace != new_workspace; ++ (old_workspace, move_workspace) ++ }; ++ ++ let mut warnings = Vec::new(); ++ let workspace_moved = if move_workspace { ++ match relocate_workspace(&old_workspace, &new_workspace).await { ++ Ok(moved) => moved, ++ Err(err) => { ++ warnings.push(err); ++ false ++ } ++ } ++ } else { ++ false ++ }; ++ ++ let memory_rows = converge_memory(stores, from, to, &mut warnings).await; ++ let cron_jobs = converge_cron(config, from, to, &mut warnings); ++ let acp_sessions = converge_acp(config, from, to, &mut warnings); ++ let sessions_repointed = converge_sessions(stores, from, to, &mut warnings); ++ ++ if warnings.is_empty() { ++ match probe_followers(config, from, stores).await { ++ Ok(false) => {} ++ Ok(true) => { ++ warnings.push("owned state still attributes rows to the retired alias".to_string()); ++ } ++ Err(err) => return Err(err), ++ } ++ } ++ ++ if !warnings.is_empty() { ++ return Err(RenameRecoveryError::Incomplete { warnings }); ++ } ++ ++ clear_record(config, from, to).await?; ++ Ok(ConvergeReport { ++ memory_rows, ++ cron_jobs, ++ acp_sessions, ++ sessions_repointed, ++ workspace_moved, ++ }) ++} +diff --git a/crates/zeroclaw-runtime/src/lib.rs b/crates/zeroclaw-runtime/src/lib.rs +index bcf02ca54..fc2198507 100644 +--- a/crates/zeroclaw-runtime/src/lib.rs ++++ b/crates/zeroclaw-runtime/src/lib.rs +@@ -14,6 +14,7 @@ pub mod migration; + pub mod util; + + pub mod agent; ++pub mod agent_rename_recovery; + pub mod approval; + pub mod browse; + pub mod calendar; +diff --git a/crates/zeroclaw-runtime/src/rpc/dispatch.rs b/crates/zeroclaw-runtime/src/rpc/dispatch.rs +index ecfa16d2b..24abe82e9 100644 +--- a/crates/zeroclaw-runtime/src/rpc/dispatch.rs ++++ b/crates/zeroclaw-runtime/src/rpc/dispatch.rs +@@ -622,23 +622,19 @@ fn rename_error_to_rpc( + rpc_err(code, format!("{path}.{from}: {err}")) + } + +-async fn move_renamed_agent_workspace( +- old_workspace: &std::path::Path, +- new_workspace: &std::path::Path, +-) -> Option { +- if old_workspace == new_workspace || !old_workspace.exists() { +- return None; +- } +- if let Some(parent) = new_workspace.parent() { +- let _ = tokio::fs::create_dir_all(parent).await; +- } +- match tokio::fs::rename(old_workspace, new_workspace).await { +- Ok(()) => None, +- Err(err) => Some(format!( +- "workspace move {} -> {} failed: {err}", +- old_workspace.display(), +- new_workspace.display() +- )), ++fn recovery_error_to_rpc( ++ path: &str, ++ from: &str, ++ err: crate::agent_rename_recovery::RenameRecoveryError, ++) -> JsonRpcError { ++ use crate::agent_rename_recovery::RenameRecoveryError; ++ match err { ++ RenameRecoveryError::NotConfigured { .. } => rpc_err( ++ INVALID_PARAMS, ++ format!("{path}.{from}: alias not found: {from}"), ++ ), ++ RenameRecoveryError::UnsafeReuse { .. } => rpc_err(INVALID_PARAMS, err.to_string()), ++ other => rpc_err(INTERNAL_ERROR, other.to_string()), + } + } + +@@ -881,41 +877,6 @@ impl RpcDispatcher { + Ok(()) + } + +- async fn agent_rename_residue_exists( +- &self, +- config: &zeroclaw_config::schema::Config, +- from: &str, +- ) -> bool { +- if config.agent_workspace_dir(from).exists() { +- return true; +- } +- if crate::cron::list_jobs_by_agent(config, from) +- .map(|jobs| !jobs.is_empty()) +- .unwrap_or(false) +- { +- return true; +- } +- if let Some(store) = self.ctx.acp_session_store.as_ref() +- && store +- .list_sessions_by_agent(from) +- .map(|sessions| !sessions.is_empty()) +- .unwrap_or(false) +- { +- return true; +- } +- if let Some(mem) = self.ctx.memory.as_ref() +- && mem.count_agent(from).await.unwrap_or(0) > 0 +- { +- return true; +- } +- if let Some(backend) = self.ctx.session_backend.as_ref() +- && backend.count_agent_attribution(from).unwrap_or(0) > 0 +- { +- return true; +- } +- false +- } +- + /// Read frames from transport, dispatch, repeat. + pub async fn run(&mut self, transport: &mut (dyn RpcTransport + Send)) { + while let Some(line) = transport.next_frame().await { +@@ -4826,6 +4787,22 @@ impl RpcDispatcher { + + async fn handle_config_map_key_create(&self, params: &Value) -> RpcResult { + let req: ConfigMapKeyCreateParams = parse_params(params)?; ++ if req.path == "agents" { ++ let data_dir = self.ctx.config.read().data_dir.clone(); ++ match crate::agent_rename_recovery::alias_reuse_blocked(&data_dir, &req.key).await { ++ Ok(true) => { ++ return Err(rpc_err( ++ INVALID_PARAMS, ++ format!( ++ "alias `{}` is stranded by an unfinished agent rename and cannot be reused yet", ++ req.key ++ ), ++ )); ++ } ++ Ok(false) => {} ++ Err(err) => return Err(recovery_error_to_rpc(&req.path, &req.key, err)), ++ } ++ } + let config_write_guard = Arc::clone(&self.ctx.config_write_lock).lock_owned().await; + let create = |config: &mut Config| -> Result { + // Shared guarded boundary: enforces the reserved-agent rule (the +@@ -5061,16 +5038,26 @@ impl RpcDispatcher { + } + + let mut working = self.ctx.config.read().clone(); +- let old_workspace = is_agent.then(|| working.agent_workspace_dir(&req.from)); +- // If a prior call saved config as `to` but crashed before side effects, +- // re-running `from -> to` should converge lagging owned state instead +- // of failing because `from` is no longer a config key. +- let resume_committed_to = is_agent +- && working.agent(&req.from).is_none() +- && working.agent(&req.to).is_some() +- && self.agent_rename_residue_exists(&working, &req.from).await; +- +- if !resume_committed_to { ++ let old_workspace = working.agent_workspace_dir(&req.from); ++ let resume_agent = if is_agent { ++ let stores = crate::agent_rename_recovery::FollowerStores { ++ memory: self.ctx.memory.as_deref(), ++ session_backend: self.ctx.session_backend.as_deref(), ++ }; ++ match crate::agent_rename_recovery::resolve_committed_rename( ++ &working, &req.from, &req.to, &stores, ++ ) ++ .await ++ { ++ Ok(crate::agent_rename_recovery::RenameDisposition::Resume) => true, ++ Ok(crate::agent_rename_recovery::RenameDisposition::Fresh { .. }) => false, ++ Err(err) => return Err(recovery_error_to_rpc(&req.path, &req.from, err)), ++ } ++ } else { ++ false ++ }; ++ ++ if !resume_agent { + let report = zeroclaw_config::alias_refs::rename_with_cascade( + &mut working, + &kind, +@@ -5095,6 +5082,17 @@ impl RpcDispatcher { + self.save_and_swap_config(working.clone(), &config_write_guard) + .await?; + } ++ if is_agent { ++ let committed = self.ctx.config.read().clone(); ++ crate::agent_rename_recovery::arm_committed_rename( ++ &committed, ++ &req.from, ++ &req.to, ++ &old_workspace, ++ ) ++ .await ++ .map_err(|err| recovery_error_to_rpc(&req.path, &req.from, err))?; ++ } + } + // Config is committed (saved + swapped, or already committed by a + // prior crashed run). Release before the post-commit side effects +@@ -5102,15 +5100,25 @@ impl RpcDispatcher { + // cascade can be slow or wedge, and holding the lock across them + // would stall every config-mutating RPC daemon-wide. + drop(config_write_guard); +- let new_workspace = is_agent.then(|| working.agent_workspace_dir(&req.to)); + + let mut warnings = Vec::new(); +- if let (Some(old_workspace), Some(new_workspace)) = (old_workspace, new_workspace) { +- warnings.extend(move_renamed_agent_workspace(&old_workspace, &new_workspace).await); +- warnings.extend( +- self.rename_agent_owned_state(&working, &req.from, &req.to) +- .await, +- ); ++ if is_agent { ++ let committed = self.ctx.config.read().clone(); ++ let stores = crate::agent_rename_recovery::FollowerStores { ++ memory: self.ctx.memory.as_deref(), ++ session_backend: self.ctx.session_backend.as_deref(), ++ }; ++ match crate::agent_rename_recovery::converge_committed_rename( ++ &committed, &req.from, &req.to, &stores, ++ ) ++ .await ++ { ++ Ok(_) => {} ++ Err(crate::agent_rename_recovery::RenameRecoveryError::Incomplete { ++ warnings: lag, ++ }) => warnings = lag, ++ Err(err) => return Err(recovery_error_to_rpc(&req.path, &req.from, err)), ++ } + } + + to_result(ConfigMapKeyRenameResult { +@@ -5123,64 +5131,6 @@ impl RpcDispatcher { + }) + } + +- async fn rename_agent_owned_state( +- &self, +- config: &zeroclaw_config::schema::Config, +- from: &str, +- to: &str, +- ) -> Vec { +- let mut warnings = Vec::new(); +- let mut memory_rows = 0usize; +- let mut cron_jobs = 0usize; +- let mut acp_sessions = 0usize; +- let mut sessions_repointed = 0usize; +- +- if let Some(mem) = &self.ctx.memory { +- match mem.rename_agent(from, to).await { +- Ok(n) => memory_rows = n, +- Err(e) => warnings.push(format!("memory rename: {e}")), +- } +- } +- +- match crate::cron::rename_jobs_by_agent(config, from, to) { +- Ok(n) => cron_jobs = n, +- Err(e) => warnings.push(format!("cron rename: {e}")), +- } +- +- match &self.ctx.acp_session_store { +- Some(store) => match store.rename_sessions_by_agent(from, to) { +- Ok(n) => acp_sessions = n, +- Err(e) => warnings.push(format!("acp rename: {e}")), +- }, +- None => warnings.push("acp store unavailable".to_string()), +- } +- +- if let Some(backend) = &self.ctx.session_backend { +- match backend.rename_agent_attribution(from, to) { +- Ok(n) => sessions_repointed = n, +- Err(e) => warnings.push(format!("session attribution rename: {e}")), +- } +- } +- +- ::zeroclaw_log::record!( +- INFO, +- ::zeroclaw_log::Event::new(module_path!(), ::zeroclaw_log::Action::Note).with_attrs( +- ::serde_json::json!({ +- "from": from, +- "to": to, +- "memory": memory_rows, +- "cron": cron_jobs, +- "acp": acp_sessions, +- "sessions": sessions_repointed, +- "warnings": warnings.clone(), +- }) +- ), +- "agent renamed with RPC owned-state cascade" +- ); +- +- warnings +- } +- + fn handle_config_templates(&self) -> RpcResult { + use zeroclaw_config::schema::Config; + let templates: Vec = Config::map_key_sections() +diff --git a/src/alias_cli/mod.rs b/src/alias_cli/mod.rs +index 4d69030b2..7bc7b1e0f 100644 +--- a/src/alias_cli/mod.rs ++++ b/src/alias_cli/mod.rs +@@ -359,20 +359,11 @@ pub async fn handle_agents(cmd: AgentsCommands, config: &mut Config) -> Result<( + match cmd { + AgentsCommands::List => list_section(config, "agents"), + AgentsCommands::Create { alias } => { ++ ensure_agent_alias_reusable(config, &alias).await?; + create_entry(config, "agents", &alias)?; + save(config).await + } +- AgentsCommands::Rename { from, to } => { +- // Capture the workspace path while the `from` entry still exists +- // (custom paths are read off the entry, which the rename moves). +- let old_ws = config.agent_workspace_dir(&from); +- rename_config(config, &AliasKind::Agent, &from, &to)?; +- // Persist the config rename before the irreversible owned-state side +- // effects (workspace move + DB re-point), so a later failure can't +- // leave the config and owned state split. +- save(config).await?; +- agent_rename_owned_state(config, &from, &to, &old_ws).await +- } ++ AgentsCommands::Rename { from, to } => rename_agent(config, &from, &to).await, + AgentsCommands::Delete { + alias, + dry_run, +@@ -556,40 +547,83 @@ async fn agent_delete_owned_state( + } + + #[cfg(all(feature = "gateway", feature = "agent-runtime"))] +-async fn agent_rename_owned_state( +- config: &Config, +- from: &str, +- to: &str, +- old_ws: &std::path::Path, +-) -> Result<()> { +- // Move the workspace dir (default per-alias location only; a custom path is +- // alias-independent → old_ws == new_ws → skip). +- let new_ws = config.agent_workspace_dir(to); +- if old_ws != new_ws && old_ws.exists() { +- if let Some(parent) = new_ws.parent() { +- tokio::fs::create_dir_all(parent).await.ok(); +- } +- if let Err(e) = tokio::fs::rename(old_ws, &new_ws).await { +- let es = e.to_string(); +- eprintln!( +- "{}", +- mta( +- "cli-alias-warn-workspace-move", +- &[("error", es.as_str())], +- "warning: workspace move failed: {$error}" +- ) +- ); +- } ++fn recovery_cli_error( ++ err: zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError, ++) -> anyhow::Error { ++ use zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError; ++ match err { ++ RenameRecoveryError::NotConfigured { path } => anyhow::Error::msg(mta( ++ "cli-alias-not-configured", ++ &[("path", path.as_str())], ++ "{$path} is not configured", ++ )), ++ RenameRecoveryError::UnsafeReuse { alias } => anyhow::Error::msg(mta( ++ "cli-alias-rename-recovery-reuse", ++ &[("alias", alias.as_str())], ++ "alias `{$alias}` is stranded by an unfinished agent rename and cannot be reused yet", ++ )), ++ RenameRecoveryError::Unreadable { store, detail } => anyhow::Error::msg(mta( ++ "cli-alias-rename-recovery-unreadable", ++ &[("store", store.as_str()), ("detail", detail.as_str())], ++ "committed agent rename could not read {$store}: {$detail}", ++ )), ++ RenameRecoveryError::Persist { detail } => anyhow::Error::msg(mta( ++ "cli-alias-rename-recovery-persist", ++ &[("detail", detail.as_str())], ++ "committed agent rename recovery could not be recorded: {$detail}", ++ )), ++ RenameRecoveryError::Incomplete { .. } => anyhow::Error::msg(mt( ++ "cli-alias-rename-recovery-incomplete", ++ "committed agent rename did not finish; retry the same rename to converge", ++ )), + } ++} ++ ++#[cfg(all(feature = "gateway", feature = "agent-runtime"))] ++async fn ensure_agent_alias_reusable(config: &Config, alias: &str) -> Result<()> { ++ let blocked = ++ zeroclaw_runtime::agent_rename_recovery::alias_reuse_blocked(&config.data_dir, alias) ++ .await ++ .map_err(recovery_cli_error)?; ++ if blocked { ++ return Err(recovery_cli_error( ++ zeroclaw_runtime::agent_rename_recovery::RenameRecoveryError::UnsafeReuse { ++ alias: alias.to_string(), ++ }, ++ )); ++ } ++ Ok(()) ++} ++ ++#[cfg(not(all(feature = "gateway", feature = "agent-runtime")))] ++async fn ensure_agent_alias_reusable(_config: &Config, _alias: &str) -> Result<()> { ++ Ok(()) ++} ++ ++#[cfg(all(feature = "gateway", feature = "agent-runtime"))] ++async fn rename_agent(config: &mut Config, from: &str, to: &str) -> Result<()> { ++ use zeroclaw_runtime::agent_rename_recovery::{self, FollowerStores, RenameDisposition}; + let (mem, session_backend) = build_owned_state_handles(config)?; +- let report = crate::gateway::agent_owned_state::cascade_rename_agent( +- config, +- &mem, +- session_backend.as_ref(), +- from, +- to, +- ) +- .await; ++ let stores = FollowerStores { ++ memory: Some(mem.as_ref()), ++ session_backend: session_backend.as_ref().map(|backend| backend.as_ref()), ++ }; ++ // Captured before the config entry moves, so a custom workspace path is ++ // the one recovery will converge. ++ let old_workspace = config.agent_workspace_dir(from); ++ let disposition = agent_rename_recovery::resolve_committed_rename(config, from, to, &stores) ++ .await ++ .map_err(recovery_cli_error)?; ++ if matches!(disposition, RenameDisposition::Fresh { .. }) { ++ rename_config(config, &AliasKind::Agent, from, to)?; ++ save(config).await?; ++ agent_rename_recovery::arm_committed_rename(config, from, to, &old_workspace) ++ .await ++ .map_err(recovery_cli_error)?; ++ } ++ let report = agent_rename_recovery::converge_committed_rename(config, from, to, &stores) ++ .await ++ .map_err(recovery_cli_error)?; + let memory = report.memory_rows.to_string(); + let cron = report.cron_jobs.to_string(); + let acp = report.acp_sessions.to_string(); +@@ -607,26 +641,13 @@ async fn agent_rename_owned_state( + "owned-state re-pointed: memory {$memory} · cron {$cron} · acp {$acp} · sessions {$sessions}" + ) + ); +- for w in &report.warnings { +- eprintln!( +- "{}", +- mta( +- "cli-alias-warn", +- &[("warning", w.as_str())], +- "warning: {$warning}" +- ) +- ); +- } + Ok(()) + } + + #[cfg(not(all(feature = "gateway", feature = "agent-runtime")))] +-async fn agent_rename_owned_state( +- _config: &Config, +- _from: &str, +- _to: &str, +- _old_ws: &std::path::Path, +-) -> Result<()> { ++async fn rename_agent(config: &mut Config, from: &str, to: &str) -> Result<()> { ++ rename_config(config, &AliasKind::Agent, from, to)?; ++ save(config).await?; + warn_agent_owned_state(); + Ok(()) + } diff --git a/artifacts/task_prompt.txt b/artifacts/task_prompt.txt new file mode 100644 index 000000000000..4c1286fbe81e --- /dev/null +++ b/artifacts/task_prompt.txt @@ -0,0 +1,19 @@ +Committed agent renames are shared by the CLI (`zeroclaw agents rename` and `zeroclaw agents create`), the gateway config map-key rename and create handlers, and RPC `config/map-key-rename` and `config/map-key-create`. + +The gateway and RPC already resume a rename whose live config has moved from the old alias to the new one when old-alias residue remains. The CLI always rewrites config, so once that commit has landed it reports that the old alias is not configured. Residue checks also treat an unreadable follower as if it were empty. + +One runtime-owned recovery contract must be shared by all three surfaces: + +- Resolve recovery from live configuration and the canonical durable stores: the default per-alias workspace, memory, cron, ACP sessions, and session attribution. +- Persist that recovery across the post-commit window, so a retry can resume after the old alias is already gone from config. +- Resume instead of reporting the alias missing when recovery is still outstanding, or when config already names the new alias and old-alias residue remains in any of those followers. +- If a configured durable store or the recovery record cannot be read, fail so the same rename can be retried. Do not treat that as empty, and do not clear recovery. +- Converge the default workspace and owned memory, cron, ACP, and session state idempotently. Attempt every follower. A partial failure stays retryable. +- Clear recovery only after every follower has been checked and has converged. +- While recovery is open, refuse to create or reuse the retired alias, even if its residue was removed outside the rename. A different alias may still be created. After a successful converge, reuse is allowed and the recreated agent does not inherit the old rows. +- A custom or otherwise alias-independent workspace path is not residue. Do not move or delete it. +- A from-to request with no recovery record, no residue, and readable stores stays not configured: gateway HTTP 404 with "is not configured", CLI text "is not configured", and RPC "alias not found". Do not arm recovery for that case. + +Knowledge-graph ownership is out of scope. + +The gateway and RPC may still report the rename as committed when a follower lags, but they must keep recovery armed and must not describe an unreadable store as a missing alias. The CLI must exit non-zero until convergence finishes, and that failure must not say the alias is not configured. diff --git a/artifacts/test.patch b/artifacts/test.patch new file mode 100644 index 000000000000..ce30882884e3 --- /dev/null +++ b/artifacts/test.patch @@ -0,0 +1,1035 @@ +diff --git a/Cargo.toml b/Cargo.toml +index 28ac6f1c8..a4a551e4f 100644 +--- a/Cargo.toml ++++ b/Cargo.toml +@@ -477,6 +477,10 @@ path = "tests/plugin_channel_runtime_e2e.rs" + name = "channel_egress_e2e" + path = "tests/channel_egress_e2e.rs" + ++[[test]] ++name = "committed_agent_rename" ++path = "tests/committed_agent_rename.rs" ++ + [[bench]] + name = "agent_benchmarks" + harness = false +diff --git a/crates/zeroclaw-gateway/src/committed_rename_recovery_tests.rs b/crates/zeroclaw-gateway/src/committed_rename_recovery_tests.rs +new file mode 100644 +index 000000000..faa472c16 +--- /dev/null ++++ b/crates/zeroclaw-gateway/src/committed_rename_recovery_tests.rs +@@ -0,0 +1,477 @@ ++//! Gateway and RPC coverage for committed agent-rename recovery. ++//! ++//! These tests call the public handlers and the public RPC dispatcher. They ++//! do not reach into the recovery module. ++ ++use std::sync::Arc; ++ ++use axum::extract::{Json, Query, State}; ++use axum::http::{HeaderMap, StatusCode}; ++use axum::response::Response; ++use http_body_util::BodyExt; ++use zeroclaw_config::alias_refs::{self, AliasKind}; ++use zeroclaw_config::schema::{AliasedAgentConfig, Config}; ++use zeroclaw_runtime::rpc::context::RpcContext; ++use zeroclaw_runtime::rpc::dispatch::RpcDispatcher; ++use zeroclaw_runtime::rpc::session::SessionStore; ++use zeroclaw_runtime::rpc::transport::RpcTransport; ++ ++use crate::api::test_state; ++use crate::api_config::{MapKeyQuery, RenameMapKeyBody, handle_map_key, handle_rename_map_key}; ++ ++fn fixture(dir: &std::path::Path, alias: &str) -> Config { ++ let mut config = Config { ++ config_path: dir.join("config.toml"), ++ data_dir: dir.join("data"), ++ ..Config::default() ++ }; ++ std::fs::create_dir_all(&config.data_dir).unwrap(); ++ config.agents.insert( ++ alias.to_string(), ++ AliasedAgentConfig { ++ risk_profile: "default".into(), ++ ..AliasedAgentConfig::default() ++ }, ++ ); ++ config ++ .risk_profiles ++ .entry("default".into()) ++ .or_default() ++ .allowed_commands = vec!["echo".into()]; ++ config.runtime_profiles.entry("default".into()).or_default(); ++ config ++} ++ ++fn commit_alias(config: &mut Config, from: &str, to: &str) { ++ alias_refs::rename_with_cascade(config, &AliasKind::Agent, from, to).unwrap(); ++} ++ ++async fn response_json(response: Response) -> (StatusCode, serde_json::Value) { ++ let status = response.status(); ++ let bytes = response ++ .into_body() ++ .collect() ++ .await ++ .expect("response body") ++ .to_bytes(); ++ let json = serde_json::from_slice(&bytes).unwrap_or_else( ++ |_| serde_json::json!({ "raw": String::from_utf8_lossy(&bytes).to_string() }), ++ ); ++ (status, json) ++} ++ ++async fn rename(state: &crate::AppState, from: &str, to: &str) -> (StatusCode, serde_json::Value) { ++ let response = Box::pin(handle_rename_map_key( ++ State(state.clone()), ++ HeaderMap::new(), ++ Json(RenameMapKeyBody { ++ path: "agents".into(), ++ from: from.into(), ++ to: to.into(), ++ }), ++ )) ++ .await; ++ response_json(response).await ++} ++ ++async fn create_agent(state: &crate::AppState, alias: &str) -> (StatusCode, serde_json::Value) { ++ let response = Box::pin(handle_map_key( ++ State(state.clone()), ++ HeaderMap::new(), ++ Query(MapKeyQuery { ++ path: "agents".into(), ++ key: alias.into(), ++ }), ++ )) ++ .await; ++ response_json(response).await ++} ++ ++fn message_of(json: &serde_json::Value) -> String { ++ json["message"].as_str().unwrap_or("").to_string() ++} ++ ++struct PipeTransport { ++ frames: std::collections::VecDeque, ++ writer: tokio::sync::mpsc::Sender, ++} ++ ++#[async_trait::async_trait] ++impl RpcTransport for PipeTransport { ++ fn writer(&self) -> tokio::sync::mpsc::Sender { ++ self.writer.clone() ++ } ++ ++ async fn next_frame(&mut self) -> Option { ++ self.frames.pop_front() ++ } ++ ++ fn peer_label(&self) -> String { ++ "committed-rename-test".into() ++ } ++} ++ ++fn rpc_frame(id: u64, method: &str, params: serde_json::Value) -> String { ++ serde_json::json!({ ++ "jsonrpc": "2.0", ++ "id": id, ++ "method": method, ++ "params": params, ++ }) ++ .to_string() ++} ++ ++async fn rpc_roundtrip( ++ config: Config, ++ calls: Vec<(u64, String, serde_json::Value)>, ++) -> Vec { ++ let sessions = Arc::new(SessionStore::new( ++ 8, ++ Arc::new(zeroclaw_infra::session_queue::SessionActorQueue::new( ++ 8, 30, 600, ++ )), ++ )); ++ let ctx = RpcContext::for_live_test(config, sessions); ++ let (tx, mut rx) = tokio::sync::mpsc::channel(32); ++ let mut dispatcher = RpcDispatcher::new(ctx, tx.clone(), "committed-rename-test".into()); ++ let mut frames = std::collections::VecDeque::new(); ++ frames.push_back(rpc_frame( ++ 1, ++ "initialize", ++ serde_json::json!({ "protocol_version": 1 }), ++ )); ++ for (id, method, params) in calls { ++ frames.push_back(rpc_frame(id, &method, params)); ++ } ++ let mut transport = PipeTransport { frames, writer: tx }; ++ dispatcher.run(&mut transport).await; ++ let mut out = Vec::new(); ++ while let Ok(line) = rx.try_recv() { ++ if let Ok(value) = serde_json::from_str::(&line) { ++ out.push(value); ++ } ++ } ++ out ++} ++ ++fn rpc_by_id(messages: &[serde_json::Value], id: u64) -> serde_json::Value { ++ messages ++ .iter() ++ .find(|message| message["id"] == id) ++ .cloned() ++ .unwrap_or_else(|| panic!("missing rpc response {id}: {messages:?}")) ++} ++ ++#[tokio::test] ++async fn committed_rename_unreadable_cron_fails_closed_and_blocks_reuse() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path(), "agent_a"); ++ commit_alias(&mut config, "agent_a", "agent_b"); ++ let cron_db = config.data_dir.join("cron").join("jobs.db"); ++ std::fs::create_dir_all(&cron_db).unwrap(); ++ ++ let state = test_state(config); ++ let (status, json) = rename(&state, "agent_a", "agent_b").await; ++ let message = message_of(&json); ++ assert_ne!( ++ status, ++ StatusCode::NOT_FOUND, ++ "an unreadable cron store must not look like a missing alias: {status} {json}" ++ ); ++ assert!( ++ !message.contains("is not configured") && !message.contains("alias not found"), ++ "unreadable recovery must stay retryable, got {message}" ++ ); ++ ++ let (create_status, create_json) = create_agent(&state, "agent_a").await; ++ assert!( ++ !create_status.is_success(), ++ "the retired alias stays blocked while recovery is outstanding: {create_json}" ++ ); ++} ++ ++#[tokio::test] ++async fn committed_rename_blocked_workspace_refuses_alias_reuse_until_converged() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let config = fixture(tmp.path(), "agent_a"); ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ std::fs::write(old_ws.join("marker"), b"keep").unwrap(); ++ let blocker = tmp.path().join("agents").join("agent_b"); ++ std::fs::create_dir_all(blocker.parent().unwrap()).unwrap(); ++ std::fs::write(&blocker, b"x").unwrap(); ++ ++ let state = test_state(config); ++ let (status, json) = rename(&state, "agent_a", "agent_b").await; ++ assert!( ++ status.is_success(), ++ "the config commit is kept when only the workspace move lags: {json}" ++ ); ++ assert!( ++ old_ws.join("marker").is_file(), ++ "failed workspace move leaves the marker in place" ++ ); ++ ++ let (blocked_status, blocked_json) = create_agent(&state, "agent_a").await; ++ assert!( ++ !blocked_status.is_success(), ++ "open recovery refuses the retired alias: {blocked_json}" ++ ); ++ let (other_status, other_json) = create_agent(&state, "agent_c").await; ++ assert!( ++ other_status.is_success(), ++ "a different alias is not stranded: {other_json}" ++ ); ++ ++ std::fs::remove_file(&blocker).unwrap(); ++ std::fs::remove_dir_all(&old_ws).unwrap(); ++ let (still_status, still_json) = create_agent(&state, "agent_a").await; ++ assert!( ++ !still_status.is_success(), ++ "wiping residue out of band does not clear recovery: {still_json}" ++ ); ++ ++ let (retry_status, retry_json) = rename(&state, "agent_a", "agent_b").await; ++ assert!( ++ retry_status.is_success(), ++ "retry converges once followers are readable: {retry_json}" ++ ); ++ let warnings = retry_json["warnings"] ++ .as_array() ++ .cloned() ++ .unwrap_or_default(); ++ assert!( ++ warnings.is_empty(), ++ "a clean retry clears recovery, got {retry_json}" ++ ); ++ let (freed_status, freed_json) = create_agent(&state, "agent_a").await; ++ assert!( ++ freed_status.is_success(), ++ "reuse is allowed after convergence: {freed_json}" ++ ); ++} ++ ++#[tokio::test] ++async fn committed_rename_sqlite_followers_converge_and_then_allow_reuse() { ++ use zeroclaw_memory::Memory; ++ ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path(), "agent_a"); ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ std::fs::write(old_ws.join("marker"), b"ws").unwrap(); ++ zeroclaw_runtime::cron::add_job(&config, "agent_a", "* * * * *", "echo hi").unwrap(); ++ ++ let memory = Arc::new( ++ zeroclaw_memory::SqliteMemory::new("agent_a", &config.data_dir).expect("sqlite memory"), ++ ); ++ memory.ensure_agent_uuid("agent_a").await.unwrap(); ++ let sessions = zeroclaw_infra::make_session_backend(&config.data_dir, "sqlite").unwrap(); ++ sessions ++ .set_session_agent_alias("sess-1", "agent_a") ++ .unwrap(); ++ { ++ let acp = ++ zeroclaw_infra::acp_session_store::AcpSessionStore::new(&config.data_dir).unwrap(); ++ acp.create_session("acp-1", "agent_a", old_ws.to_str().unwrap_or("/tmp")) ++ .unwrap(); ++ } ++ ++ commit_alias(&mut config, "agent_a", "agent_b"); ++ let mut state = test_state(config.clone()); ++ state.mem = memory.clone(); ++ state.session_backend = Some(sessions.clone()); ++ ++ let (status, json) = rename(&state, "agent_a", "agent_b").await; ++ assert!(status.is_success(), "committed residue resumes: {json}"); ++ let warnings = json["warnings"].as_array().cloned().unwrap_or_default(); ++ assert!( ++ warnings.is_empty(), ++ "followers converge without leftover warnings: {json}" ++ ); ++ assert!( ++ state ++ .config ++ .read() ++ .agent_workspace_dir("agent_b") ++ .join("marker") ++ .is_file() ++ ); ++ assert!(!old_ws.exists()); ++ assert_eq!(memory.count_agent("agent_a").await.unwrap(), 0); ++ assert_eq!(memory.count_agent("agent_b").await.unwrap(), 1); ++ assert_eq!( ++ zeroclaw_runtime::cron::list_jobs_by_agent(&config, "agent_b") ++ .unwrap() ++ .len(), ++ 1 ++ ); ++ assert!( ++ zeroclaw_runtime::cron::list_jobs_by_agent(&config, "agent_a") ++ .unwrap() ++ .is_empty() ++ ); ++ let acp = zeroclaw_infra::acp_session_store::AcpSessionStore::new(&config.data_dir).unwrap(); ++ assert_eq!(acp.list_sessions_by_agent("agent_b").unwrap().len(), 1); ++ assert!(acp.list_sessions_by_agent("agent_a").unwrap().is_empty()); ++ assert_eq!(sessions.count_agent_attribution("agent_b").unwrap(), 1); ++ assert_eq!(sessions.count_agent_attribution("agent_a").unwrap(), 0); ++ ++ let (again_status, again_json) = rename(&state, "agent_a", "agent_b").await; ++ assert_eq!(again_status, StatusCode::NOT_FOUND, "{again_json}"); ++ assert!( ++ message_of(&again_json).contains("is not configured"), ++ "{again_json}" ++ ); ++ ++ let (created_status, created_json) = create_agent(&state, "agent_a").await; ++ assert!(created_status.is_success(), "{created_json}"); ++ assert_eq!(memory.count_agent("agent_a").await.unwrap(), 0); ++ assert_eq!(memory.count_agent("agent_b").await.unwrap(), 1); ++} ++ ++#[tokio::test] ++async fn committed_rename_unrelated_target_stays_not_configured() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let config = fixture(tmp.path(), "agent_b"); ++ let state = test_state(config); ++ let (status, json) = rename(&state, "agent_a", "agent_b").await; ++ assert_eq!(status, StatusCode::NOT_FOUND, "{json}"); ++ assert!(message_of(&json).contains("is not configured"), "{json}"); ++} ++ ++#[tokio::test] ++async fn committed_rename_rpc_unreadable_cron_is_not_alias_not_found() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path(), "agent_a"); ++ commit_alias(&mut config, "agent_a", "agent_b"); ++ config.save().await.unwrap(); ++ std::fs::create_dir_all(config.data_dir.join("cron").join("jobs.db")).unwrap(); ++ let saved = config.clone(); ++ ++ let messages = rpc_roundtrip( ++ config, ++ vec![( ++ 2, ++ "config/map-key-rename".into(), ++ serde_json::json!({"path": "agents", "from": "agent_a", "to": "agent_b"}), ++ )], ++ ) ++ .await; ++ let rename_msg = rpc_by_id(&messages, 2); ++ let err = rename_msg["error"]["message"].as_str().unwrap_or(""); ++ assert!( ++ rename_msg.get("error").is_some(), ++ "unreadable cron fails the rpc rename: {rename_msg}" ++ ); ++ assert!( ++ !err.contains("alias not found") && !err.contains("is not configured"), ++ "rpc must not report the alias missing when cron cannot be read: {err}" ++ ); ++ ++ let create_messages = rpc_roundtrip( ++ saved, ++ vec![( ++ 2, ++ "config/map-key-create".into(), ++ serde_json::json!({"path": "agents", "key": "agent_a"}), ++ )], ++ ) ++ .await; ++ let create_msg = rpc_by_id(&create_messages, 2); ++ assert!( ++ create_msg.get("error").is_some(), ++ "rpc create of the retired alias stays refused: {create_msg}" ++ ); ++} ++ ++#[tokio::test] ++async fn committed_rename_rpc_resumes_workspace_then_reports_alias_not_found() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path(), "agent_a"); ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ std::fs::write(old_ws.join("marker"), b"rpc").unwrap(); ++ let new_ws = { ++ let mut preview = config.clone(); ++ commit_alias(&mut preview, "agent_a", "agent_b"); ++ preview.agent_workspace_dir("agent_b") ++ }; ++ commit_alias(&mut config, "agent_a", "agent_b"); ++ config.save().await.unwrap(); ++ ++ let messages = rpc_roundtrip( ++ config.clone(), ++ vec![( ++ 2, ++ "config/map-key-rename".into(), ++ serde_json::json!({"path": "agents", "from": "agent_a", "to": "agent_b"}), ++ )], ++ ) ++ .await; ++ let first = rpc_by_id(&messages, 2); ++ assert!( ++ first.get("result").is_some(), ++ "workspace residue resumes over rpc: {first}" ++ ); ++ assert!( ++ new_ws.join("marker").is_file(), ++ "marker moved to the new workspace" ++ ); ++ assert!(!old_ws.exists()); ++ ++ let second_messages = rpc_roundtrip( ++ config, ++ vec![( ++ 2, ++ "config/map-key-rename".into(), ++ serde_json::json!({"path": "agents", "from": "agent_a", "to": "agent_b"}), ++ )], ++ ) ++ .await; ++ let second = rpc_by_id(&second_messages, 2); ++ let err = second["error"]["message"].as_str().unwrap_or(""); ++ assert!( ++ err.contains("alias not found"), ++ "a finished rename reports the alias missing on the next call: {second}" ++ ); ++} ++ ++#[tokio::test] ++async fn committed_rename_rpc_blocked_workspace_refuses_create() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let config = fixture(tmp.path(), "agent_a"); ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ let blocker = tmp.path().join("agents").join("agent_b"); ++ std::fs::create_dir_all(blocker.parent().unwrap()).unwrap(); ++ std::fs::write(&blocker, b"x").unwrap(); ++ config.save().await.unwrap(); ++ ++ let messages = rpc_roundtrip( ++ config.clone(), ++ vec![ ++ ( ++ 2, ++ "config/map-key-rename".into(), ++ serde_json::json!({"path": "agents", "from": "agent_a", "to": "agent_b"}), ++ ), ++ ( ++ 3, ++ "config/map-key-create".into(), ++ serde_json::json!({"path": "agents", "key": "agent_a"}), ++ ), ++ ], ++ ) ++ .await; ++ let renamed = rpc_by_id(&messages, 2); ++ assert!( ++ renamed.get("result").is_some(), ++ "blocked workspace still commits the rename: {renamed}" ++ ); ++ let created = rpc_by_id(&messages, 3); ++ assert!( ++ created.get("error").is_some(), ++ "rpc refuses to recreate the retired alias while recovery is open: {created}" ++ ); ++} +diff --git a/crates/zeroclaw-gateway/src/lib.rs b/crates/zeroclaw-gateway/src/lib.rs +index d76c0d8ee..e7d79c86d 100644 +--- a/crates/zeroclaw-gateway/src/lib.rs ++++ b/crates/zeroclaw-gateway/src/lib.rs +@@ -57,6 +57,9 @@ pub mod ws; + pub mod ws_approval; + pub mod ws_sop_runs; + ++#[cfg(test)] ++mod committed_rename_recovery_tests; ++ + use anyhow::{Context, Result}; + #[cfg(any( + feature = "channel-email", +diff --git a/test.sh b/test.sh +new file mode 100755 +index 000000000..d3c0fcbd7 +--- /dev/null ++++ b/test.sh +@@ -0,0 +1,94 @@ ++#!/usr/bin/env bash ++# Fail-to-pass harness for committed agent-rename recovery. ++# `base` is expected to exit non-zero. `new` is expected to exit zero. ++ ++output_path="" ++mode="" ++while [[ $# -gt 0 ]]; do ++ case "$1" in ++ --output_path) ++ output_path="${2:-}" ++ shift 2 ++ ;; ++ base|new) ++ mode="$1" ++ shift ++ ;; ++ *) ++ echo "unknown argument: $1" >&2 ++ exit 2 ++ ;; ++ esac ++done ++ ++if [[ -z "$output_path" || -z "$mode" ]]; then ++ echo "usage: test.sh --output_path FILE {base|new}" >&2 ++ exit 2 ++fi ++ ++mkdir -p "$(dirname "$output_path")" ++ ++logs="" ++failed=0 ++run_cargo() { ++ local label="$1" ++ shift ++ local log ++ log="$(mktemp)" ++ "$@" >"$log" 2>&1 ++ local status=$? ++ cat "$log" ++ logs="${logs}"$'\n'"===== ${label} exit ${status} ====="$'\n'"$(cat "$log")" ++ if [[ $status -ne 0 ]]; then ++ failed=1 ++ fi ++ rm -f "$log" ++} ++ ++run_cargo "cli" cargo test --offline --test committed_agent_rename -- --test-threads=1 ++run_cargo "gateway" cargo test --offline -p zeroclaw-gateway --lib committed_rename -- --test-threads=1 ++ ++log_file="$(mktemp)" ++printf '%s\n' "$logs" >"$log_file" ++python3 - "$output_path" "$mode" "$failed" "$log_file" <<'PY' ++import html ++import re ++import sys ++from pathlib import Path ++ ++output_path, mode, failed, log_path = sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4] ++log = Path(log_path).read_text(encoding="utf-8", errors="replace") ++cases = [] ++seen = set() ++pattern = re.compile(r"^test (?P\S+) \.\.\. (?Pok|FAILED|ignored)", re.M) ++for match in pattern.finditer(log): ++ name = match.group("name") ++ status = match.group("status") ++ key = (name, status, match.start()) ++ if key in seen: ++ continue ++ seen.add(key) ++ cases.append((name, status)) ++ ++failures = sum(1 for _, status in cases if status == "FAILED") ++if not cases and failed != "0": ++ cases.append(("cargo", "FAILED")) ++ failures = 1 ++ ++parts = [ ++ '', ++ f'', ++] ++for name, status in cases: ++ escaped = html.escape(name) ++ if status == "FAILED": ++ parts.append(f' ') ++ else: ++ parts.append(f' ') ++parts.append("") ++with open(output_path, "w", encoding="utf-8") as handle: ++ handle.write("\n".join(parts) + "\n") ++PY ++rm -f "$log_file" ++ ++exit "$failed" +diff --git a/tests/committed_agent_rename.rs b/tests/committed_agent_rename.rs +new file mode 100644 +index 000000000..daffc4d08 +--- /dev/null ++++ b/tests/committed_agent_rename.rs +@@ -0,0 +1,417 @@ ++//! CLI and cross-surface coverage for committed agent-rename recovery. ++//! ++//! The binary and the public gateway handlers are the only surfaces these ++//! tests drive. ++ ++use std::path::{Path, PathBuf}; ++use std::process::{Command, Output}; ++use std::sync::Arc; ++use std::time::Duration; ++ ++use parking_lot::RwLock; ++use zeroclaw::gateway::{self, AppState}; ++use zeroclaw_api::attribution::Attributable; ++use zeroclaw_config::alias_refs::{self, AliasKind}; ++use zeroclaw_config::schema::{AliasedAgentConfig, Config}; ++use zeroclaw_memory::{Memory, SqliteMemory}; ++use zeroclaw_providers::ModelProvider; ++ ++#[derive(Default)] ++struct MockModelProvider; ++ ++#[async_trait::async_trait] ++impl ModelProvider for MockModelProvider { ++ async fn chat_with_system( ++ &self, ++ _system_prompt: Option<&str>, ++ _message: &str, ++ _model: &str, ++ _temperature: Option, ++ ) -> anyhow::Result { ++ Ok("ok".into()) ++ } ++} ++ ++impl Attributable for MockModelProvider { ++ fn role(&self) -> zeroclaw_api::attribution::Role { ++ zeroclaw_api::attribution::Role::Provider(zeroclaw_api::attribution::ProviderKind::Model( ++ zeroclaw_api::attribution::ModelProviderKind::Custom, ++ )) ++ } ++ ++ fn alias(&self) -> &str { ++ "MockModelProvider" ++ } ++} ++ ++fn gateway_state(config: Config) -> AppState { ++ let memory: Arc = ++ Arc::new(zeroclaw_memory::NoneMemory::new("none")); ++ AppState { ++ config: Arc::new(RwLock::new(config)), ++ config_write_lock: Arc::new(tokio::sync::Mutex::new(())), ++ model_provider: Arc::new(MockModelProvider), ++ model: "test-model".into(), ++ temperature: None, ++ mem: memory.clone(), ++ memory_strategy: Arc::new( ++ zeroclaw_runtime::agent::memory_strategy::DefaultMemoryStrategy::with_config( ++ memory, ++ zeroclaw_config::schema::MemoryConfig::default(), ++ PathBuf::new(), ++ ), ++ ), ++ auto_save: false, ++ pairing: Arc::new(zeroclaw_runtime::security::PairingGuard::new( ++ false, ++ &[], ++ zeroclaw_config::pairing::PairingCodePolicy::default(), ++ )), ++ trust_forwarded_headers: false, ++ rate_limiter: Arc::new(gateway::GatewayRateLimiter::new(100, 100, 100)), ++ auth_limiter: Arc::new(gateway::auth_rate_limit::AuthRateLimiter::new()), ++ idempotency_store: Arc::new(gateway::IdempotencyStore::new( ++ Duration::from_secs(300), ++ 1000, ++ )), ++ #[cfg(feature = "channel-whatsapp-cloud")] ++ whatsapp: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-whatsapp-cloud")] ++ whatsapp_app_secret: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-linq")] ++ linq: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-linq")] ++ linq_signing_secrets: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-nextcloud")] ++ nextcloud_talk: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-nextcloud")] ++ nextcloud_talk_webhook_secret: std::collections::HashMap::new(), ++ #[cfg(feature = "channel-email")] ++ gmail_push: None, ++ observer: Arc::new(zeroclaw_runtime::observability::NoopObserver), ++ tools_registry: Arc::new(Vec::new()), ++ tools_registry_by_agent: Arc::new(std::collections::HashMap::new()), ++ cost_tracker: None, ++ event_tx: tokio::sync::broadcast::channel(16).0, ++ event_buffer: Arc::new(gateway::sse::EventBuffer::new(16)), ++ shutdown_tx: tokio::sync::watch::channel(false).0, ++ reload_tx: None, ++ node_registry: Arc::new(gateway::nodes::NodeRegistry::new(16)), ++ mdns_peer_registry: gateway::nodes::mdns::MdnsPeerRegistry::default(), ++ path_prefix: String::new(), ++ web_dist_dir: None, ++ session_backend: None, ++ session_queue: Arc::new(gateway::session_queue::SessionActorQueue::new(8, 30, 600)), ++ device_registry: None, ++ pending_pairings: None, ++ canvas_store: zeroclaw_runtime::tools::CanvasStore::new(), ++ #[cfg(feature = "webauthn")] ++ webauthn: None, ++ cancel_tokens: Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())), ++ pending_reload: Arc::new(std::sync::atomic::AtomicBool::new(false)), ++ tui_registry: None, ++ sop_engine: None, ++ sop_audit: None, ++ } ++} ++ ++fn fixture(dir: &Path) -> Config { ++ let mut config = Config { ++ config_path: dir.join("config.toml"), ++ data_dir: dir.join("data"), ++ ..Config::default() ++ }; ++ std::fs::create_dir_all(&config.data_dir).unwrap(); ++ config.agents.insert( ++ "agent_a".into(), ++ AliasedAgentConfig { ++ risk_profile: "default".into(), ++ ..AliasedAgentConfig::default() ++ }, ++ ); ++ config ++ .risk_profiles ++ .entry("default".into()) ++ .or_default() ++ .allowed_commands = vec!["echo".into()]; ++ config.runtime_profiles.entry("default".into()).or_default(); ++ config ++} ++ ++fn run_agents(config_dir: &Path, args: &[&str]) -> Output { ++ let mut command = Command::new(env!("CARGO_BIN_EXE_zeroclaw")); ++ command ++ .env("ZEROCLAW_CONFIG_DIR", config_dir) ++ .env_remove("ZEROCLAW_DATA_DIR") ++ .env_remove("ZEROCLAW_WORKSPACE") ++ .env("RUST_LOG", "off") ++ .arg("--config-dir") ++ .arg(config_dir) ++ .arg("agents") ++ .args(args); ++ command.output().expect("spawn zeroclaw") ++} ++ ++fn text_of(output: &Output) -> String { ++ format!( ++ "{}{}", ++ String::from_utf8_lossy(&output.stdout), ++ String::from_utf8_lossy(&output.stderr) ++ ) ++} ++ ++async fn seed_followers(config: &Config) -> PathBuf { ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ std::fs::write(old_ws.join("marker"), b"owned").unwrap(); ++ zeroclaw_runtime::cron::add_job(config, "agent_a", "* * * * *", "echo hi").unwrap(); ++ { ++ let memory = SqliteMemory::new("agent_a", &config.data_dir).unwrap(); ++ memory.ensure_agent_uuid("agent_a").await.unwrap(); ++ } ++ { ++ let sessions = zeroclaw_infra::make_session_backend(&config.data_dir, "sqlite").unwrap(); ++ sessions ++ .set_session_agent_alias("sess-1", "agent_a") ++ .unwrap(); ++ } ++ { ++ let acp = ++ zeroclaw_infra::acp_session_store::AcpSessionStore::new(&config.data_dir).unwrap(); ++ acp.create_session("acp-1", "agent_a", old_ws.to_string_lossy().as_ref()) ++ .unwrap(); ++ } ++ old_ws ++} ++ ++async fn assert_followers_on(config: &Config, alias: &str, present: bool) { ++ let memory = SqliteMemory::new("probe", &config.data_dir).unwrap(); ++ let memory_count = memory.count_agent(alias).await.unwrap(); ++ let cron_count = zeroclaw_runtime::cron::list_jobs_by_agent(config, alias) ++ .unwrap() ++ .len(); ++ let acp = zeroclaw_infra::acp_session_store::AcpSessionStore::new(&config.data_dir).unwrap(); ++ let acp_count = acp.list_sessions_by_agent(alias).unwrap().len(); ++ let sessions = zeroclaw_infra::make_session_backend(&config.data_dir, "sqlite").unwrap(); ++ let session_count = sessions.count_agent_attribution(alias).unwrap(); ++ if present { ++ assert_eq!(memory_count, 1, "memory rows for {alias}"); ++ assert_eq!(cron_count, 1, "cron jobs for {alias}"); ++ assert_eq!(acp_count, 1, "acp sessions for {alias}"); ++ assert_eq!(session_count, 1, "session attribution for {alias}"); ++ } else { ++ assert_eq!(memory_count, 0, "memory rows for {alias}"); ++ assert_eq!(cron_count, 0, "cron jobs for {alias}"); ++ assert_eq!(acp_count, 0, "acp sessions for {alias}"); ++ assert_eq!(session_count, 0, "session attribution for {alias}"); ++ } ++} ++ ++#[tokio::test] ++async fn committed_rename_cli_resumes_followers_then_allows_reuse() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path()); ++ let old_ws = seed_followers(&config).await; ++ let new_ws = { ++ let mut preview = config.clone(); ++ alias_refs::rename_with_cascade(&mut preview, &AliasKind::Agent, "agent_a", "agent_b") ++ .unwrap(); ++ preview.agent_workspace_dir("agent_b") ++ }; ++ alias_refs::rename_with_cascade(&mut config, &AliasKind::Agent, "agent_a", "agent_b").unwrap(); ++ config.save().await.unwrap(); ++ ++ let first = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ let first_text = text_of(&first); ++ assert!( ++ first.status.success(), ++ "a committed rename with follower residue must resume, not report the alias missing: {first_text}" ++ ); ++ assert!(new_ws.join("marker").is_file(), "workspace marker moved"); ++ assert!(!old_ws.exists(), "old workspace removed"); ++ assert_followers_on(&config, "agent_b", true).await; ++ assert_followers_on(&config, "agent_a", false).await; ++ ++ let second = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ let second_text = text_of(&second); ++ assert!( ++ !second.status.success(), ++ "recovery is cleared after convergence: {second_text}" ++ ); ++ assert!( ++ second_text.contains("is not configured"), ++ "a finished rename is not configured anymore: {second_text}" ++ ); ++ ++ let created = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ created.status.success(), ++ "reuse is allowed after convergence: {}", ++ text_of(&created) ++ ); ++ assert_followers_on(&config, "agent_a", false).await; ++ assert_followers_on(&config, "agent_b", true).await; ++} ++ ++#[tokio::test] ++async fn committed_rename_cli_unreadable_cron_stays_retryable_and_blocks_reuse() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path()); ++ alias_refs::rename_with_cascade(&mut config, &AliasKind::Agent, "agent_a", "agent_b").unwrap(); ++ config.save().await.unwrap(); ++ std::fs::create_dir_all(config.data_dir.join("cron").join("jobs.db")).unwrap(); ++ ++ let renamed = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ let renamed_text = text_of(&renamed); ++ assert!(!renamed.status.success(), "{renamed_text}"); ++ assert!( ++ !renamed_text.contains("is not configured") && !renamed_text.contains("alias not found"), ++ "an unreadable cron store is not an absent alias: {renamed_text}" ++ ); ++ ++ let created = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ !created.status.success(), ++ "the retired alias cannot be recreated while recovery is outstanding: {}", ++ text_of(&created) ++ ); ++} ++ ++#[tokio::test] ++async fn committed_rename_cli_blocked_workspace_survives_out_of_band_wipe() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let config = fixture(tmp.path()); ++ let old_ws = seed_followers(&config).await; ++ config.save().await.unwrap(); ++ let blocker = tmp.path().join("agents").join("agent_b"); ++ std::fs::create_dir_all(blocker.parent().unwrap()).unwrap(); ++ std::fs::write(&blocker, b"x").unwrap(); ++ ++ let first = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ let first_text = text_of(&first); ++ assert!( ++ !first.status.success(), ++ "a failed follower converge must be retryable: {first_text}" ++ ); ++ assert!( ++ !first_text.contains("is not configured"), ++ "the source alias was configured when the rename started: {first_text}" ++ ); ++ assert!( ++ old_ws.join("marker").is_file(), ++ "marker stays when the move fails" ++ ); ++ ++ let blocked = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ !blocked.status.success(), ++ "open recovery blocks the retired alias: {}", ++ text_of(&blocked) ++ ); ++ let other = run_agents(tmp.path(), &["create", "agent_c"]); ++ assert!( ++ other.status.success(), ++ "another alias can still be created: {}", ++ text_of(&other) ++ ); ++ ++ std::fs::remove_file(&blocker).unwrap(); ++ std::fs::remove_dir_all(&old_ws).unwrap(); ++ let still = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ !still.status.success(), ++ "deleting residue outside the rename does not finish recovery: {}", ++ text_of(&still) ++ ); ++ ++ let second = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ assert!( ++ second.status.success(), ++ "retry converges and clears recovery: {}", ++ text_of(&second) ++ ); ++ let freed = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ freed.status.success(), ++ "reuse works after a clean converge: {}", ++ text_of(&freed) ++ ); ++ assert_followers_on(&config, "agent_a", false).await; ++ assert_followers_on(&config, "agent_b", true).await; ++} ++ ++#[tokio::test] ++async fn committed_rename_custom_workspace_is_not_residue() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let mut config = fixture(tmp.path()); ++ let custom = tmp.path().join("custom-ws"); ++ std::fs::create_dir_all(&custom).unwrap(); ++ std::fs::write(custom.join("keep"), b"stay").unwrap(); ++ config.agents.get_mut("agent_a").unwrap().workspace.path = Some(custom.clone()); ++ alias_refs::rename_with_cascade(&mut config, &AliasKind::Agent, "agent_a", "agent_b").unwrap(); ++ config.save().await.unwrap(); ++ ++ let renamed = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ let renamed_text = text_of(&renamed); ++ assert!(!renamed.status.success(), "{renamed_text}"); ++ assert!( ++ renamed_text.contains("is not configured"), ++ "an alias-independent workspace is not stranded rename residue: {renamed_text}" ++ ); ++ assert!( ++ custom.join("keep").is_file(), ++ "custom workspace was not moved" ++ ); ++ assert!(custom.exists(), "custom workspace was not deleted"); ++} ++ ++#[tokio::test] ++async fn committed_rename_gateway_recovery_is_visible_to_the_cli() { ++ let tmp = tempfile::tempdir().unwrap(); ++ let config = fixture(tmp.path()); ++ let old_ws = config.agent_workspace_dir("agent_a"); ++ std::fs::create_dir_all(&old_ws).unwrap(); ++ std::fs::write(old_ws.join("marker"), b"cross").unwrap(); ++ config.save().await.unwrap(); ++ let blocker = tmp.path().join("agents").join("agent_b"); ++ std::fs::create_dir_all(blocker.parent().unwrap()).unwrap(); ++ std::fs::write(&blocker, b"x").unwrap(); ++ ++ let state = gateway_state(config); ++ let response = gateway::api_config::handle_rename_map_key( ++ axum::extract::State(state), ++ axum::http::HeaderMap::new(), ++ axum::Json(gateway::api_config::RenameMapKeyBody { ++ path: "agents".into(), ++ from: "agent_a".into(), ++ to: "agent_b".into(), ++ }), ++ ) ++ .await; ++ assert!( ++ response.status().is_success(), ++ "gateway keeps the committed rename when the workspace move fails" ++ ); ++ ++ let blocked = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ !blocked.status.success(), ++ "cli create sees recovery armed by the gateway: {}", ++ text_of(&blocked) ++ ); ++ ++ std::fs::remove_file(&blocker).unwrap(); ++ std::fs::remove_dir_all(&old_ws).unwrap(); ++ let resumed = run_agents(tmp.path(), &["rename", "agent_a", "agent_b"]); ++ assert!( ++ resumed.status.success(), ++ "cli converges recovery the gateway armed: {}", ++ text_of(&resumed) ++ ); ++ let freed = run_agents(tmp.path(), &["create", "agent_a"]); ++ assert!( ++ freed.status.success(), ++ "cli can reuse the alias after the shared recovery converges: {}", ++ text_of(&freed) ++ ); ++}