//! Startup auto-migration. Six idempotent phases: applied repo, //! proposed repo, meta repo, container repoint, root→h-root rename, //! and manager tool-groups backfill. //! Kill-switch: `HIVE_SKIP_META_MIGRATION=1`. Full migration sequence //! and phase details: `docs/approvals.md::Migration from the pre-tag`. use std::path::Path; use std::sync::Arc; use std::time::Duration; use anyhow::{Context, Result}; use tokio::process::Command; use crate::coordinator::Coordinator; use crate::lifecycle::{self, AGENT_PREFIX, MANAGER_CONTAINER, MANAGER_NAME}; use crate::meta; use crate::tool_groups; const KILL_SWITCH: &str = "HIVE_SKIP_META_MIGRATION"; /// Per-shellout timeouts for the blocking startup migration. `run` is /// awaited *before* the daemon starts serving (main.rs), so any child /// process that wedges here freezes the whole daemon — admin socket + /// dashboard included — with no diagnostics: a git/container shellout was /// observed blocked for 86min under a concurrent `nixos-rebuild`. Every /// shellout now runs under a timeout that kills the child on elapse, so a /// stuck migration degrades to a logged warning instead of a hung boot. /// Git ops are quick; `nixos-container update` can legitimately trigger a /// nix build, so it gets a much longer budget. const GIT_TIMEOUT: Duration = Duration::from_mins(2); const CONTAINER_TIMEOUT: Duration = Duration::from_mins(10); /// Substring that identifies the *current* agent flake boilerplate. /// Bumped whenever the template changes so the startup migration /// re-renders existing agents onto the new shape. Today the marker /// is the `flakeInputs` module-arg forwarding line — older templates /// (raw `import ./agent.nix`) get rewritten on next hive-c0re start. const MODULE_FLAKE_MARKER: &str = "_module.args.flakeInputs"; pub async fn run(coord: &Arc) -> Result<()> { if std::env::var(KILL_SWITCH).is_ok() { tracing::info!("migration: {KILL_SWITCH} set — skipping"); return Ok(()); } // Stale meta index lock: a previous hive-c0re crash mid-`git add` // can leave `.git/index.lock` behind, which blocks every // subsequent meta op until somebody `rm`s it manually. We just // booted so nothing of ours is holding it; safe to clear. let meta_lock = crate::paths::meta_git_index_lock(); if meta_lock.exists() { match std::fs::remove_file(&meta_lock) { Ok(()) => tracing::warn!("cleared stale meta/.git/index.lock"), Err(e) => tracing::warn!(error = ?e, "clear stale meta lock failed"), } } let names = enumerate_agents().await; tracing::info!(count = names.len(), "migration: scanning"); // Phase 0: move harness-owned files out of state/ into harness/. // Idempotent — rename is a no-op if the source doesn't exist and // the destination already does. tracing::debug!("migration: phase 0 (harness files)"); for name in &names { migrate_harness_files(name); } // Phase 1 + 2: per-agent applied + proposed. tracing::debug!("migration: phase 1+2 (applied + proposed repos)"); for name in &names { tracing::debug!(%name, "migration: applied+proposed"); if let Err(e) = migrate_applied_repo(name.as_str()).await { tracing::warn!(%name, error = ?e, "migration: applied repo rewrite failed"); } let proposed_dir = Coordinator::agent_proposed_dir(name); let proposed = lifecycle::setup_proposed(&proposed_dir, name.as_str()); match tokio::time::timeout(GIT_TIMEOUT, proposed).await { Ok(Err(e)) => tracing::warn!(%name, error = ?e, "migration: setup_proposed failed"), Err(_) => { tracing::warn!(%name, timeout = ?GIT_TIMEOUT, "migration: setup_proposed timed out — skipping"); } Ok(Ok(())) => {} } } // Phase 3: meta repo. tracing::debug!("migration: phase 3 (meta sync_agents)"); let agents = lifecycle::agents_for_meta_listing() .await .unwrap_or_default(); match tokio::time::timeout(GIT_TIMEOUT, meta::sync_agents(&coord.hive_env(), &agents)).await { Ok(Err(e)) => tracing::warn!(error = ?e, "migration: meta sync_agents failed"), Err(_) => { tracing::warn!(timeout = ?GIT_TIMEOUT, "migration: meta sync_agents timed out — skipping"); } Ok(Ok(())) => {} } // Phase 4: container repoint, guarded by marker. if crate::paths::meta_migration_marker().exists() { tracing::debug!("migration: phase 4 marker present, skipping repoint"); return Ok(()); } tracing::debug!("migration: phase 4 (container repoint)"); let mut all_ok = true; for name in &names { // Mark Rebuilding so the crash watcher skips this container // during the brief stop+start window the nixos-container // update activation triggers. Without this, crash_watch // would fire ContainerCrash for every agent here and the // manager would spuriously try to recover them. let guard = coord.transient_guard(name.as_str(), crate::coordinator::TransientKind::Rebuilding); let result = repoint_container(name.as_str()).await; drop(guard); if let Err(e) = result { tracing::warn!(%name, error = ?e, "migration: container repoint failed"); all_ok = false; } } if all_ok && !names.is_empty() && let Err(e) = std::fs::write(crate::paths::meta_migration_marker(), b"done\n") { tracing::warn!(error = ?e, "migration: write repoint marker failed"); } // Phase 5: rename `root` nixos-container to `h-root` for naming // consistency with sub-agents. Guarded by marker; skipped on // fresh installs (conf file absent) and after first successful run. rename_manager_container(coord).await; // Phase 6: ensure ruth has explicit tool groups so removing the // role-based fallback (Role::Manager → MANAGER_DEFAULT) doesn't // silently strip her privileged tools on next rebuild. backfill_manager_tool_groups(&names); Ok(()) } /// Move harness-owned sqlite/config files out of the agent-visible state dir /// and into the sibling harness dir. Best-effort: logs warnings but never /// fails. Idempotent — each file is only moved if present at the old path /// and absent at the new path. fn migrate_harness_files(name: &hive_host_sock::Ident) { const HARNESS_FILES: &[&str] = &[ "hyperhive-events.sqlite", "hyperhive-turn-stats.sqlite", "hyperhive-model", ]; let state_dir = Coordinator::agent_notes_dir(name); let harness_dir = Coordinator::agent_harness_dir(name); if let Err(e) = std::fs::create_dir_all(&harness_dir) { tracing::warn!(%name, error = ?e, "migration: create harness dir failed"); return; } for file in HARNESS_FILES { let src = state_dir.join(file); let dst = harness_dir.join(file); if !src.exists() || dst.exists() { continue; } match std::fs::rename(&src, &dst) { Ok(()) => tracing::info!(%name, %file, "migration: moved to harness dir"), Err(e) => { tracing::warn!(%name, %file, error = ?e, "migration: move to harness dir failed"); } } } } /// Phase 5: rename the `root` nixos-container to `h-root` so the /// manager container name is consistent with the `h-` prefix used by /// all sub-agents. Idempotent and marker-guarded. Steps: /// /// 1. Check `/etc/nixos-containers/root.conf` exists (old name present). /// 2. Stop the `root` container. /// 3. Copy `root.conf` → `h-root.conf`. /// 4. Move `/var/lib/nixos-containers/root/` → `h-root/` (if present). /// 5. `systemctl daemon-reload` so systemd sees the new unit name. /// 6. `nixos-container start h-root`. /// 7. Write the done marker. /// /// Best-effort: logs warnings on failure. A failed rename leaves both /// conf files present; on the next hive-c0re start the marker is /// absent so the phase retries. async fn rename_manager_container(coord: &Arc) { if crate::paths::hroot_rename_marker().exists() { return; } let old_conf = std::path::PathBuf::from("/etc/nixos-containers/root.conf"); let new_conf = std::path::PathBuf::from("/etc/nixos-containers/h-root.conf"); if !old_conf.exists() { // Fresh install — root container was never created under the old name. let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n"); return; } if new_conf.exists() { // Already renamed (but marker was lost — write it and return). tracing::info!("migration phase 5: h-root.conf already present, marking done"); let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n"); return; } tracing::info!("migration phase 5: renaming root container to h-root"); let _guard = coord.transient_guard(MANAGER_NAME, crate::coordinator::TransientKind::Rebuilding); // Stop the old container. Abort if stop fails — continuing with a // running `root` and then starting `h-root` risks two manager // instances racing for the same broker / state files. match Command::new("nixos-container") .args(["stop", "root"]) .status() .await { Ok(s) if s.success() => {} Ok(s) => { tracing::warn!(status = %s, "migration phase 5: nixos-container stop root failed — aborting"); return; } Err(e) => { tracing::warn!(error = ?e, "migration phase 5: nixos-container stop root failed — aborting"); return; } } // Copy conf file. if let Err(e) = std::fs::copy(&old_conf, &new_conf) { tracing::warn!(error = ?e, "migration phase 5: copy root.conf failed — aborting"); return; } // Move rootfs if it exists (may be absent for ephemeral containers). let old_rootfs = std::path::PathBuf::from("/var/lib/nixos-containers/root"); let new_rootfs = std::path::PathBuf::from("/var/lib/nixos-containers/h-root"); if old_rootfs.exists() && !new_rootfs.exists() && let Err(e) = std::fs::rename(&old_rootfs, &new_rootfs) { tracing::warn!(error = ?e, "migration phase 5: rename rootfs failed (non-fatal)"); } // Daemon reload so systemd picks up the new container@h-root unit. if let Err(e) = Command::new("systemctl") .args(["daemon-reload"]) .status() .await { tracing::warn!(error = ?e, "migration phase 5: systemctl daemon-reload failed"); } // Start the renamed container. if let Err(e) = Command::new("nixos-container") .args(["start", "h-root"]) .status() .await { tracing::warn!(error = ?e, "migration phase 5: nixos-container start h-root failed"); return; } tracing::info!("migration phase 5: root container renamed to h-root"); let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n"); // Clean up the old conf file so `nixos-container list` doesn't show // a stale stopped `root` entry. Best-effort; a failure here is // harmless — h-root is already running and the marker is written. if let Err(e) = std::fs::remove_file(&old_conf) { tracing::warn!(error = ?e, "migration phase 5: remove old root.conf failed (non-fatal)"); } } async fn enumerate_agents() -> Vec { let containers = lifecycle::list().await.unwrap_or_default(); containers .into_iter() .filter_map(|c| { let name = if c == MANAGER_CONTAINER { MANAGER_NAME } else { c.strip_prefix(AGENT_PREFIX)? }; hive_host_sock::Ident::parse(name).ok() }) .collect() } async fn migrate_applied_repo(name: &str) -> Result<()> { let dir = crate::paths::applied_dir(name); if !dir.join(".git").exists() { return Ok(()); } let flake_path = dir.join("flake.nix"); let cur = std::fs::read_to_string(&flake_path).unwrap_or_default(); if cur.contains(MODULE_FLAKE_MARKER) { return Ok(()); } let want = lifecycle::initial_flake_nix(); std::fs::write(&flake_path, want).with_context(|| format!("write {}", flake_path.display()))?; raw_git( &dir, &[ "-c", "user.name=c0re", "-c", "user.email=c0re@hyperhive.local", "add", "flake.nix", ], ) .await?; raw_git( &dir, &[ "-c", "user.name=c0re", "-c", "user.email=c0re@hyperhive.local", "commit", "-m", "migration: module-only flake", ], ) .await?; // Relocate deployed/0 to the migration commit so // setup_applied's existence check passes. raw_git(&dir, &["tag", "-f", "deployed/0", "HEAD"]).await?; tracing::info!(%name, "migration: applied repo migrated to module-only flake"); Ok(()) } async fn repoint_container(name: &str) -> Result<()> { let container = lifecycle::container_name(name); let flake_ref = format!("{}#{name}", crate::paths::meta_root().display()); let mut cmd = Command::new("nixos-container"); cmd.args(["update", &container, "--flake", &flake_ref]); let out = output_with_timeout( cmd, CONTAINER_TIMEOUT, &format!("nixos-container update {container}"), ) .await?; if !out.status.success() { anyhow::bail!( "nixos-container update {container} exited {}: {}", out.status, String::from_utf8_lossy(&out.stderr).trim() ); } tracing::info!(%name, %container, "migration: container repointed at meta"); Ok(()) } /// Phase 6: if ruth is a deployed agent and has no explicit entry in /// `tool-groups.json`, set her groups to `MANAGER_DEFAULT` (all groups). /// Idempotent — skips when entry already present. Prevents a silent tool /// downgrade when upgrading from a build that relied on the manager-flavor /// fallback in `effective_tool_groups()`. fn backfill_manager_tool_groups(names: &[hive_host_sock::Ident]) { if !names.iter().any(|n| n.as_str() == MANAGER_NAME) { return; // ruth not deployed — nothing to backfill } let existing = tool_groups::groups_for(MANAGER_NAME); if !existing.is_empty() { tracing::debug!("migration: ruth already has explicit tool groups — skipping backfill"); return; } let all_groups: Vec = hive_sh4re::ToolGroup::MANAGER_DEFAULT .iter() .map(|g| g.as_str().to_owned()) .collect(); match tool_groups::set_groups(MANAGER_NAME, &all_groups) { Ok(()) => tracing::info!( "migration: backfilled ruth's tool groups to MANAGER_DEFAULT (all groups)" ), Err(e) => tracing::warn!( error = ?e, "migration: failed to backfill ruth's tool groups — she may lose privileged tools on next rebuild" ), } } /// Run a command to completion under a timeout, capturing its output. On /// timeout the child is killed (`kill_on_drop`) and an error is returned, /// so a wedged shellout can never freeze startup migration. `what` is a /// human label surfaced in the timeout error + `with_context`. async fn output_with_timeout( mut cmd: Command, timeout: Duration, what: &str, ) -> Result { cmd.kill_on_drop(true); match tokio::time::timeout(timeout, cmd.output()).await { Ok(r) => r.with_context(|| format!("run {what}")), Err(_) => anyhow::bail!("{what} timed out after {timeout:?} (child killed)"), } } async fn raw_git(dir: &Path, args: &[&str]) -> Result<()> { let mut cmd = lifecycle::git_command(); cmd.current_dir(dir).args(args); let label = format!("git {} in {}", args.join(" "), dir.display()); let out = output_with_timeout(cmd, GIT_TIMEOUT, &label).await?; if !out.status.success() { anyhow::bail!( "git {} failed: {}", args.join(" "), String::from_utf8_lossy(&out.stderr).trim() ); } Ok(()) }