//! `nixos-container` lifecycle + per-agent config flake generation. use std::path::Path; use anyhow::{Context, Result, bail}; use hive_sh4re::priv_proto::BindMount; use tokio::process::Command; use crate::coordinator::{AgentPaths, HiveEnv}; /// Sub-agent container prefix. `nixos-container` caps the total container name /// at 11 chars (it gets encoded into network interface names), so the agent /// name itself can be at most `MAX_AGENT_NAME` chars. pub const AGENT_PREFIX: &str = "h-"; pub const MAX_AGENT_NAME: usize = 9; /// Logical name of the manager agent (broker recipient, state-dir key, /// meta flake attribute). All persistent state lives under `ruth/`. pub const MANAGER_NAME: &str = "ruth"; /// Container name of the manager. Uses the same `h-` prefix as sub-agents /// so `nixos-container list` output is uniform and the list filter is /// a single `starts_with(AGENT_PREFIX)` check. Logical name → container /// name: `ruth` → `h-ruth`. pub const MANAGER_CONTAINER: &str = "h-ruth"; /// Mount point of the per-agent runtime directory inside the container. pub const CONTAINER_RUNTIME_MOUNT: &str = "/run/hive"; /// Where the per-agent Claude credentials dir mounts inside the /// container. The harness service runs as a non-root unix user /// whose home is `/home//`, so the mount path varies per /// agent — `container_claude_mount(name)` returns /// `/home//.claude` for every agent including the manager. /// `claude` inside the container reads /// `$HOME/.claude` and the service environment sets `HOME` to the /// same path, so the OAuth session survives container restarts. #[must_use] pub fn container_claude_mount(name: &str) -> String { format!("/home/{name}/.claude") } /// Mount point of the shared directory accessible to all agents. /// All agents can read/write here; agents should only put things they're /// willing to lose (other agents may delete them). pub const CONTAINER_SHARED_MOUNT: &str = "/shared"; const GIT_NAME: &str = "c0re"; const GIT_EMAIL: &str = "c0re@hyperhive.local"; /// Sub-agent web UI port range. Deterministic from the agent's name (FNV-1a /// hash mod range size), so the dashboard can compute the same port without /// asking hive-c0re. const WEB_PORT_BASE: u16 = 8100; const WEB_PORT_RANGE: u16 = 900; /// FNV-1a hash of a string — shared by `agent_web_port` and /// `agent_network_ip` so the derivation rule is identical. fn fnv1a(s: &str) -> u32 { let mut hash: u32 = 2_166_136_261; for b in s.bytes() { hash ^= u32::from(b); hash = hash.wrapping_mul(16_777_619); } hash } /// Per-agent web UI port — `WEB_PORT_BASE + FNV-1a(name) % /// WEB_PORT_RANGE` for every agent including the manager. The port /// allocation rule reads the same for every name; collisions are /// possible (birthday paradox at ~30 agents) and the operator /// resolves them by renaming an agent (different hash → different /// port). Stable across hosts, restarts, and dashboard renders — /// no state-file dance. #[must_use] pub fn agent_web_port(name: &str) -> u16 { // Modulo of a u32 by a u16's value is guaranteed < u16::MAX, so try_from never fails. WEB_PORT_BASE + u16::try_from(fnv1a(name) % u32::from(WEB_PORT_RANGE)).unwrap_or(0) } /// Deterministic IPv4 address for an agent inside an isolated subnet. /// /// Parses `subnet_cidr` as `/` (e.g. /// `"10.42.0.0/24"`), then computes: /// /// ```text /// host_count = 2^(32 - prefix_len) /// usable = host_count - 3 // skip .0 (network), .1 (gateway), .255 (broadcast) /// offset = FNV-1a(name) % usable + 2 // .2 is the first agent slot /// agent_ip = network_base_u32 + offset /// ``` /// /// Returns `None` when `subnet_cidr` can't be parsed (invalid format, /// prefix out of range, etc.) so callers can fall back gracefully. /// Collisions are possible (birthday paradox) and the operator resolves /// them by renaming an agent, same as for port collisions. #[must_use] pub fn agent_network_ip(name: &str, subnet_cidr: &str) -> Option { let (ip_str, prefix_str) = subnet_cidr.split_once('/')?; let prefix_len: u32 = prefix_str.parse().ok()?; if prefix_len > 30 { // /31 and /32 have no room for agents; /30 has 1 usable slot. // /0 (the other extreme) is handled further down: host_count // overflows checked_shl(32) → 0 → usable = 0 → None. return None; } // Parse dotted-decimal IPv4. let octets: Vec = ip_str .split('.') .map(|o| o.parse::().ok()) .collect::>>()?; if octets.len() != 4 { return None; } let base_u32 = u32::from_be_bytes([octets[0], octets[1], octets[2], octets[3]]); // Mask off host bits to get the true network address. let mask = if prefix_len == 0 { 0u32 } else { !0u32 << (32 - prefix_len) }; let network_base = base_u32 & mask; let host_count: u32 = 1u32.checked_shl(32 - prefix_len).unwrap_or(0); // `.0` = network, `.1` = bridge gateway, last = broadcast → 3 reserved. let usable = host_count.saturating_sub(3); if usable == 0 { return None; } let offset = fnv1a(name) % usable + 2; // +2: skip .0 and .1 let ip_u32 = network_base + offset; let [a, b, c, d] = ip_u32.to_be_bytes(); Some(format!("{a}.{b}.{c}.{d}")) } #[must_use] pub fn container_name(name: &str) -> String { format!("{AGENT_PREFIX}{name}") } /// Read the agent user's `(uid, gid)` from the container's nixos-managed /// `/etc/passwd`. Returns `None` when the container hasn't been built /// yet, the passwd file is unparseable, or the agent user is missing /// (e.g. legacy container that still runs as root). /// /// Used by `forge` + `matrix` after writing per-agent state files so /// the bind-mounted host file ends up readable by the agent user /// without waiting for the next container activation to run the chown /// fixup. /// /// Notes: /// - Reads the *container-local* passwd at /// `/var/lib/nixos-containers//etc/passwd`, not the host's. /// The container's user-namespace shares uids with the host (no /// `PrivateUsers`), so the uid is directly usable in host-side /// `chown(2)`. /// - Best-effort: caller treats `None` as "skip the chown". #[must_use] pub fn agent_uid_gid(agent_name: &str) -> Option<(u32, u32)> { let container = container_name(agent_name); let passwd_path = format!("/var/lib/nixos-containers/{container}/etc/passwd"); let content = std::fs::read_to_string(&passwd_path).ok()?; for line in content.lines() { let mut parts = line.split(':'); let user = parts.next()?; if user != agent_name { continue; } let _ = parts.next()?; // x (password placeholder) let uid: u32 = parts.next()?.parse().ok()?; let gid: u32 = parts.next()?.parse().ok()?; return Some((uid, gid)); } None } /// Best-effort `chown(path, agent_uid, agent_gid)`. Resolves the agent's /// uid/gid via [`agent_uid_gid`] and shells out to `std::os::unix::fs::chown`. /// Silently no-ops when the container isn't built yet (`None` from /// [`agent_uid_gid`]) and logs at debug on chown syscall failure — the /// activation script in `harness-base.nix` is the steady-state safety /// net. Used by per-agent state writers in `forge` + `matrix` so the /// agent can read the file without waiting for the next container /// rebuild. pub fn chown_to_agent(name: &str, path: &Path, subsystem: &str) { let Some((uid, gid)) = agent_uid_gid(name) else { return; }; if let Err(e) = std::os::unix::fs::chown(path, Some(uid), Some(gid)) { tracing::debug!(%name, %subsystem, path = %path.display(), error = %e, "chown to agent failed"); } } fn validate(name: &str) -> Result<()> { if name.is_empty() { bail!("agent name must not be empty"); } if name.len() > MAX_AGENT_NAME { bail!( "agent name '{name}' is too long ({} chars); max {MAX_AGENT_NAME}", name.len() ); } Ok(()) } /// First name (≠ `self_name`) currently running whose hashed port /// matches this agent's. The harness inside the colliding container /// would otherwise loop on `AddrInUse` forever; we surface the /// conflict here so spawn / rebuild fails loudly with an actionable /// message instead. async fn port_collision(self_name: &str) -> Option { let port = agent_web_port(self_name); let raw = list().await.unwrap_or_default(); for c in raw { let Some(other) = c.strip_prefix(AGENT_PREFIX) else { continue; }; if other == self_name { continue; } if agent_web_port(other) == port && is_running(other).await { return Some(other.to_owned()); } } None } pub async fn spawn(name: &str, hive: &HiveEnv, paths: &AgentPaths) -> Result<()> { validate(name)?; if let Some(other) = port_collision(name).await { bail!( "port {} is already taken by '{other}' — rename one of them and retry", agent_web_port(name) ); } setup_proposed(&paths.proposed_dir, name).await?; setup_applied(&paths.applied_dir, Some(&paths.proposed_dir), name).await?; ensure_claude_dir(&paths.claude_dir)?; ensure_state_dir(&paths.notes_dir)?; // Meta flake gets the new agent's input + nixosConfiguration // before `nixos-container create` so the `--flake meta#` // ref resolves. let agents = agents_after_spawn(name).await?; crate::meta::sync_agents(hive, &agents).await?; let container = container_name(name); priv_run("create", name).await?; set_nspawn_flags( &container, &paths.agent_dir, &paths.claude_dir, &paths.notes_dir, ) .await?; set_resource_limits(&container, &hive.agent_cpu_quota, &hive.agent_memory_max).await?; systemd_daemon_reload().await?; priv_run("start", name).await } /// Build the `AgentSpec` list for the meta flake from `nixos-container /// list` + a hypothetical extra name not yet in the list (for spawn /// where the new agent's container doesn't exist yet). Pass empty /// `name_to_add` from rebuild paths where the agent is already in the /// container list. /// /// Propagates errors from `list()` rather than swallowing them. /// Using `.unwrap_or_default()` here would silently produce an empty /// agent list when `nixos-container list` fails (priv helper down, race), /// which `sync_agents` would then commit to meta — dropping every agent /// from `flake.nix`. Callers that can tolerate failures (e.g. migration) /// handle the `Err` themselves with `.unwrap_or_default()`. async fn agents_for_meta(name_to_add: Option<&str>) -> Result> { let containers = list().await?; let mut out: Vec = containers .into_iter() .filter_map(|c| { let name = c.strip_prefix(AGENT_PREFIX)?.to_owned(); Some(crate::meta::AgentSpec { is_manager: name == MANAGER_NAME, port: agent_web_port(&name), name, }) }) .collect(); if let Some(extra) = name_to_add && !out.iter().any(|a| a.name == extra) { out.push(crate::meta::AgentSpec { is_manager: extra == MANAGER_NAME, port: agent_web_port(extra), name: extra.to_owned(), }); } out.sort_by(|a, b| a.name.cmp(&b.name)); Ok(out) } async fn agents_after_spawn(name: &str) -> Result> { agents_for_meta(Some(name)).await } /// Like `agents_for_meta_listing` but with an extra agent added (for a /// container that doesn't exist yet). Used by the first-spawn path in /// `actions::run_apply_commit` to register the new agent in meta before /// `prepare_deploy` tries to update its input lock. pub async fn agents_for_meta_listing_with(extra: &str) -> Result> { agents_for_meta(Some(extra)).await } /// Public enumeration of currently-existing agents (whatever /// `nixos-container list` says), sorted, no extras. For callers /// outside this module that need to reseed meta after lifecycle /// changes — destroy, startup reconciliation, etc. pub async fn agents_for_meta_listing() -> Result> { agents_for_meta(None).await } /// True when the named container already exists (appears in /// `nixos-container list`). Used by the apply-commit path to decide /// between first-spawn (`nixos-container create`) and normal rebuild /// (`nixos-container update`). pub async fn container_exists(name: &str) -> bool { let container = container_name(name); list() .await .unwrap_or_default() .iter() .any(|c| c == &container) } pub async fn kill(name: &str) -> Result<()> { validate(name)?; priv_run("stop", name).await } pub async fn start(name: &str) -> Result<()> { validate(name)?; priv_run("start", name).await } /// Stop + start without regenerating any config. For "kick the container" /// without touching the flake or nspawn flags. pub async fn restart(name: &str) -> Result<()> { kill(name).await?; start(name).await } /// True when the container's systemd unit is active. Used by the dashboard /// to gate stop/restart buttons. pub async fn is_running(name: &str) -> bool { let container = container_name(name); let unit = format!("container@{container}.service"); Command::new("systemctl") .args(["is-active", "--quiet", &unit]) .status() .await .is_ok_and(|s| s.success()) } /// Fully tear down a sub-agent's container: stop + remove via `nixos-container /// destroy`, then clean our own systemd drop-in. Leaves it to the caller to /// wipe `/var/lib/hyperhive/...` state and the per-agent runtime dir. pub async fn destroy(name: &str) -> Result<()> { validate(name)?; let container = container_name(name); // nixos-container destroy handles stop + removal of /var/lib/nixos-containers/ // and /etc/nixos-containers/.conf. Tolerate "no such container". if let Err(e) = priv_run("destroy", name).await { tracing::warn!(error = ?e, "nixos-container destroy returned an error; continuing cleanup"); } // Remove the systemd resource-limits drop-in via hive-priv. if let Err(e) = crate::priv_client::remove_service_dropin(&container).await { tracing::warn!(error = ?e, "remove service drop-in failed (non-fatal)"); } Ok(()) } pub async fn rebuild( name: &str, hive: &HiveEnv, paths: &AgentPaths, on_step: &(dyn Fn(&str) + Send + Sync), on_build_log_id: &(dyn Fn(i64) + Send + Sync), ) -> Result<()> { // Sync the meta flake (idempotent — no-op when the rendered // flake matches disk) so a manual rebuild from the dashboard // can also recover from a divergent meta repo (e.g. an agent // got added directly via `nixos-container create` outside // hive-c0re). let agents = agents_for_meta(None).await?; crate::meta::sync_agents(hive, &agents).await?; // Then bump just this agent's input — picks up whatever // `applied//main` currently points at (deployed/). // Commits the lock if it changed. crate::meta::lock_update_for_rebuild(name).await?; rebuild_no_meta(name, hive, paths, on_step, on_build_log_id).await } /// Container-level rebuild without touching the meta repo. Callers /// that own the meta side themselves (`actions::run_apply_commit` /// drives meta through the two-phase prepare/finalize/abort flow) /// use this directly. Public `rebuild` wraps it with idempotent meta /// sync + lock-bump-and-commit. /// /// `on_step` is called at each phase boundary with a short human-readable /// label so callers can surface progress (e.g. update the rebuild-queue /// step shown in the dashboard). Pass `&|_| ()` when progress reporting /// is not needed. /// /// `on_build_log_id` is called with the build-log row id immediately after /// the `nixos-container update` log row opens, before the actual update /// command starts. Callers can use this to link the queue entry to the log /// for live streaming. Pass `&|_| ()` when not needed. pub async fn rebuild_no_meta( name: &str, hive: &HiveEnv, paths: &AgentPaths, on_step: &(dyn Fn(&str) + Send + Sync), on_build_log_id: &(dyn Fn(i64) + Send + Sync), ) -> Result<()> { validate(name)?; if let Some(other) = port_collision(name).await { bail!( "port {} is already taken by '{other}' — rename one of them and retry", agent_web_port(name) ); } setup_applied(&paths.applied_dir, None, name).await?; ensure_claude_dir(&paths.claude_dir)?; ensure_state_dir(&paths.notes_dir)?; let container = container_name(name); let flake_ref = format!("{}#{name}", crate::meta::meta_dir().display()); if container_exists(name).await { // Rebuild strategy: stop-before-update + pre-build. // See `docs/coordinator.md::Container lifecycle`. let was_running = is_running(name).await; set_nspawn_flags( &container, &paths.agent_dir, &paths.claude_dir, &paths.notes_dir, ) .await?; set_resource_limits(&container, &hive.agent_cpu_quota, &hive.agent_memory_max).await?; systemd_daemon_reload().await?; if was_running { on_step("nix build"); prebuild_toplevel(name, &flake_ref).await?; on_step("nixos-container stop"); priv_run("stop", name).await?; } on_step("nixos-container update"); let update_result = priv_run_inner("update", name, Some(on_build_log_id)).await; if let Err(ref update_err) = update_result { // The update failed (e.g. nix build error). If the agent was // running before we stopped it, try to bring it back up on the // previous successful configuration so it doesn't stay dead. // The start failure is logged but not promoted to an error — // we always propagate the original update error (below). if was_running { tracing::warn!( %name, error = %update_err, "nixos-container update failed; attempting restart on old config" ); on_step("nixos-container start (recovery)"); if let Err(e) = priv_run("start", name).await { tracing::warn!(%name, error = %e, "recovery start after failed update also failed"); } } } update_result?; if was_running { // Cold-start fallback on activation errors. // See `docs/coordinator.md::Cold-start fallback`. on_step("nixos-container start"); if let Err(start_err) = priv_run("start", name).await { tracing::warn!( container = %container, error = %start_err, "start after rebuild failed (possible activation error); \ retrying via stop + kill + start" ); priv_run("stop", name).await.unwrap_or_else(|e| { tracing::warn!( container = %container, error = %e, "stop before cold-start retry failed (ignored)" ); }); priv_run("kill", name).await.unwrap_or_else(|e| { tracing::warn!( container = %container, error = %e, "kill before cold-start retry failed (ignored)" ); }); priv_run("start", name).await.map_err(|e| { anyhow::anyhow!( "cold-start fallback also failed: {e:#} \ (original start error: {start_err:#})" ) }) } else { Ok(()) } } else { Ok(()) } } else { // Spawn path: create is atomic, no prebuild needed. // See `docs/coordinator.md::Spawn path`. on_step("nixos-container create"); priv_run("create", name).await?; set_nspawn_flags( &container, &paths.agent_dir, &paths.claude_dir, &paths.notes_dir, ) .await?; set_resource_limits(&container, &hive.agent_cpu_quota, &hive.agent_memory_max).await?; systemd_daemon_reload().await?; on_step("nixos-container start"); priv_run("start", name).await } } /// Pre-build `system.build.toplevel` against `meta#` so the /// subsequent `nixos-container update` finds the result cached and /// skips straight to the profile-swap. Store-warming only — container /// is untouched. See `docs/coordinator.md::Rebuild path` for why /// the prebuild happens before stop, and `docs/coordinator.md::Prebuild /// attr path` for why the explicit nixosConfigurations attr is required. async fn prebuild_toplevel(name: &str, flake_ref: &str) -> Result<()> { use tokio::io::{AsyncBufReadExt, BufReader}; // Split `#` so we can re-emit with the explicit // `nixosConfigurations.` segment. The flake_ref shape is // constructed by `rebuild_no_meta` and always contains exactly one // `#`; `split_once` returning None here would be a programmer // error we'd want to surface loudly rather than paper over. let (flake_root, fragment) = flake_ref .split_once('#') .with_context(|| format!("flake_ref {flake_ref:?} missing '#' fragment"))?; // Sanity-check the fragment matches the agent name we were // passed — guards against future calls that pass a divergent // pair (no current callsite does, but the pair is redundant // and worth checking once). if fragment != name { anyhow::bail!("prebuild_toplevel: flake_ref fragment '{fragment}' ≠ agent name '{name}'"); } let attr = format!("{flake_root}#nixosConfigurations.{name}.config.system.build.toplevel"); let args = vec![ "--extra-experimental-features", "nix-command flakes", "build", "--no-link", "--print-out-paths", &attr, ]; let cmdline = format!("nix {}", args.join(" ")); tracing::info!(%name, %cmdline, "prebuild: warming system toplevel"); // Open a build_logs row for this attempt (best-effort — None when // the global handle hasn't been installed, e.g. early startup // or standalone tests). Lines pumped from stdout/stderr append // into the row; `finish` lands the terminal status before we bail. let logs = crate::build_logs::global(); let log_id = logs.as_ref().and_then(|h| { h.start(name, "prebuild", &cmdline) .map_err(|e| { tracing::warn!(error = ?e, "build_logs: start failed (prebuild log dropped)"); }) .ok() }); let mut child = Command::new("nix") .args(&args) .stdout(std::process::Stdio::piped()) .stderr(std::process::Stdio::piped()) .spawn() .with_context(|| format!("spawn {cmdline}"))?; let stdout = child.stdout.take().expect("piped stdout"); let stderr = child.stderr.take().expect("piped stderr"); let stdout_cmdline = cmdline.clone(); let stdout_logs = logs.clone(); let pump_stdout = tokio::spawn(async move { let mut lines = BufReader::new(stdout).lines(); while let Ok(Some(line)) = lines.next_line().await { tracing::info!(target: "nix-prebuild", cmdline = %stdout_cmdline, "{line}"); if let (Some(h), Some(id)) = (&stdout_logs, log_id) { h.append_stdout(id, &line); } } }); let stderr_cmdline = cmdline.clone(); let stderr_logs = logs.clone(); let pump_stderr = tokio::spawn(async move { let mut lines = BufReader::new(stderr).lines(); while let Ok(Some(line)) = lines.next_line().await { tracing::warn!(target: "nix-prebuild", cmdline = %stderr_cmdline, "{line}"); if let (Some(h), Some(id)) = (&stderr_logs, log_id) { h.append_stderr(id, &line); } } }); let status = child .wait() .await .with_context(|| format!("wait {cmdline}"))?; let _ = pump_stdout.await; let _ = pump_stderr.await; let ok = status.success(); if let (Some(h), Some(id)) = (&logs, log_id) { h.finish( id, if ok { crate::build_logs::BuildStatus::Ok } else { crate::build_logs::BuildStatus::Fail }, ); } if !ok { match log_id { Some(id) => bail!("prebuild {cmdline} failed ({status}); see build log #{id}"), None => bail!("prebuild {cmdline} failed ({status})"), } } Ok(()) } pub async fn list() -> Result> { let stdout = crate::priv_client::list_containers().await?; Ok(stdout .lines() .map(str::trim) .filter(|line| line.starts_with(AGENT_PREFIX)) .map(str::to_owned) .collect()) } /// Initialize the manager-editable proposed repo. Seeds two tracked /// files: `agent.nix` (the module the manager edits) and `flake.nix` /// (the boilerplate that lets the meta flake import this repo as an /// input — meta locks at a specific sha and reads /// `nixosModules.default`, so `flake.nix` must be in the commit). The /// manager shouldn't edit `flake.nix` (the prompt says so) but it's /// visible so they can introspect. /// /// Touched by hive-c0re only on first spawn — never again — so the /// manager can't be surprised by hive-c0re commits or working-tree /// resets. pub async fn setup_proposed(proposed_dir: &Path, name: &str) -> Result<()> { let fresh = !proposed_dir.join(".git").exists(); if fresh { std::fs::create_dir_all(proposed_dir) .with_context(|| format!("create {}", proposed_dir.display()))?; let agent_path = proposed_dir.join("agent.nix"); if !agent_path.exists() { std::fs::write(&agent_path, initial_agent_nix(name)) .with_context(|| format!("write {}", agent_path.display()))?; } let flake_path = proposed_dir.join("flake.nix"); if !flake_path.exists() { std::fs::write(&flake_path, initial_flake_nix()) .with_context(|| format!("write {}", flake_path.display()))?; } git(proposed_dir, &["init", "--initial-branch=main"]).await?; git(proposed_dir, &["add", "agent.nix", "flake.nix"]).await?; git_commit(proposed_dir, "hive-c0re init").await?; } // Idempotently wire the `applied` remote — purely for the // manager's ergonomics. The URL is the path inside the manager // container (`/applied//.git`), where the RO bind in // `set_nspawn_flags` makes it real. hive-c0re itself never // dereferences this remote; the host-side fetch in // `request_apply_commit` uses absolute host paths. ensure_applied_remote(proposed_dir, name).await } async fn ensure_applied_remote(proposed_dir: &Path, name: &str) -> Result<()> { let want = format!("/applied/{name}/.git"); let existing = git_command() .current_dir(proposed_dir) .args(["remote", "get-url", "applied"]) .output() .await .with_context(|| format!("git remote get-url applied in {}", proposed_dir.display()))?; if existing.status.success() { let current = String::from_utf8_lossy(&existing.stdout).trim().to_owned(); if current == want { return Ok(()); } // URL drifted (path scheme changed, etc.) — re-point it. return git(proposed_dir, &["remote", "set-url", "applied", &want]).await; } git(proposed_dir, &["remote", "add", "applied", &want]).await } /// Set up the applied repo. First-spawn only: init the repo, pull /// proposed's initial commit in via `git fetch`, tag it `deployed/0`. /// This is the *only* time hive-c0re reads from `proposed` for an /// agent — subsequent proposals are fetched at `request_apply_commit` /// time and tagged `proposal/` (see `actions::approve` for the /// tag state machine). /// /// `proposed_dir` is `None` on rebuild paths where the repo already /// exists — we just verify it's the right shape and bail otherwise. /// Unlike the pre-overhaul code path, `flake.nix` is no longer /// regenerated at the host level: it's tracked in proposed (seeded by /// `setup_proposed`) and rides along on every fetch. pub async fn setup_applied( applied_dir: &Path, proposed_dir: Option<&Path>, name: &str, ) -> Result<()> { std::fs::create_dir_all(applied_dir) .with_context(|| format!("create {}", applied_dir.display()))?; if !applied_dir.join(".git").exists() { let Some(proposed) = proposed_dir else { bail!( "applied repo at {} is missing its .git directory; \ cannot rebuild without a proposed source to seed from. \ destroy --purge and re-spawn this agent.", applied_dir.display() ); }; git(applied_dir, &["init", "--initial-branch=main"]).await?; let proposed_str = proposed.display().to_string(); // Seed the applied repo at the root (template) commit of proposed, // not at `main`. This ensures `deployed/0` is the template baseline // so the first ApplyCommit diff shows the manager's real changes // rather than an empty diff (which happens when the manager has // already committed their config and proposed/main == proposal/). let root_sha = git_root_commit(proposed).await?; git( applied_dir, // --update-head-ok lets us fetch into refs/heads/main while // HEAD still points there. git's default safeguard refuses // to avoid index/working-tree desync, but the working tree // is empty (we just `init`'d) and we read-tree-reset right // after, so the safeguard is moot here. &[ "fetch", "--no-tags", "--update-head-ok", &proposed_str, &format!("{root_sha}:refs/heads/main"), ], ) .await?; git_read_tree_reset(applied_dir, "refs/heads/main").await?; git_tag(applied_dir, "deployed/0", "refs/heads/main").await?; } else if git_rev_parse(applied_dir, "refs/tags/deployed/0") .await .is_err() { // Pre-overhaul applied repo — no deployed/* tag scheme, // flake.nix may be untracked, agent.nix possibly authored by // hive-c0re directly. The startup auto-migration fixes this // in place; if it didn't run (or got skipped), surface a // clear error. bail!( "applied repo at {} predates the meta-flake layout. \ Restart hive-c0re to let the auto-migration run, or \ destroy --purge {name} and re-spawn.", applied_dir.display() ); } Ok(()) } /// Create the per-agent Claude credentials dir if missing. Mode 0755 — hive-core /// needs read+execute to list the directory so `claude_has_session` can detect a /// valid session; credential files inside (`.credentials.json` etc.) are 0600 so /// secrets stay private regardless of the directory mode. Idempotent: existing /// dirs are left untouched (an agent's OAuth tokens survive `destroy`/recreate). /// Public for the `InitConfig` approval path in `actions.rs` which seeds /// dirs without calling the full `spawn`. pub fn ensure_claude_dir(claude_dir: &Path) -> Result<()> { use std::io; if !claude_dir.exists() { std::fs::create_dir_all(claude_dir) .with_context(|| format!("create {}", claude_dir.display()))?; } // 0755: hive-core (different user from the agent) needs read+execute to // list the directory so `claude_has_session` can detect a valid session. // The credential files inside (`.credentials.json` etc.) are 0600 so the // secrets themselves stay private regardless of the directory mode. // // Best-effort: on the first container boot, `hive-agent-user-migrate` // chowns this dir to the agent user. After that, hive-core (a different // user) cannot chmod it (EPERM) — that's fine because the mode set during // initial creation (0755) is preserved through the chown. Any other error // (ENOENT, I/O error) is unexpected and propagated. #[cfg(unix)] { use std::os::unix::fs::PermissionsExt; match std::fs::set_permissions(claude_dir, std::fs::Permissions::from_mode(0o755)) { Ok(()) => {} Err(e) if e.kind() == io::ErrorKind::PermissionDenied => { tracing::debug!( path = %claude_dir.display(), "ensure_claude_dir: chmod 755 skipped (dir likely owned by agent user after migration)" ); } Err(e) => { return Err(e).with_context(|| format!("chmod 755 {}", claude_dir.display())); } } } Ok(()) } /// Public for the `InitConfig` approval path in `actions.rs` which seeds /// dirs without calling the full `spawn`. Also creates the sibling `harness/` /// dir so the first harness startup can write its sqlite files immediately. pub fn ensure_state_dir(notes_dir: &Path) -> Result<()> { if !notes_dir.exists() { std::fs::create_dir_all(notes_dir) .with_context(|| format!("create {}", notes_dir.display()))?; } // Harness dir is a sibling of the agent-visible state dir. if let Some(parent) = notes_dir.parent() { let harness_dir = parent.join("harness"); if !harness_dir.exists() { std::fs::create_dir_all(&harness_dir) .with_context(|| format!("create {}", harness_dir.display()))?; } } Ok(()) } fn initial_agent_nix(name: &str) -> String { format!( "{{ config, pkgs, lib, ... }}:\n{{\n # Per-agent overrides for {name}. This is a regular NixOS module\n # — add packages, services, modules, imports as needed.\n #\n # imports = [ ./extra-module.nix ];\n # environment.systemPackages = with pkgs; [ ];\n}}\n", ) } /// Module-only flake exposed by every agent's repo. Consumed by the /// hive-c0re-owned meta flake at `/var/lib/hyperhive/meta/` as a flake /// input. The wrapper is intentionally permissive: /// /// - Manager edits `inputs.* = …` to add other flakes (e.g. an MCP /// server's own flake) — the lock for those lands in the agent's /// own `flake.lock` and rolls up into meta's lock transitively. /// - The outputs block forwards every input (minus `self`) into /// `agent.nix` as the `flakeInputs` module argument, so the /// manager just references `flakeInputs..packages.${pkgs.system}.default` /// without further plumbing. /// /// Identity injection (`HIVE_PORT` / `HIVE_LABEL` / dashboard port / /// git committer) still lives in the meta flake's wrapper. pub fn initial_flake_nix() -> &'static str { "{\n description = \"hyperhive agent\";\n inputs = { };\n outputs =\n { self, ... }@inputs:\n {\n nixosModules.default = {\n imports = [ ./agent.nix ];\n _module.args.flakeInputs = builtins.removeAttrs inputs [ \"self\" ];\n };\n };\n}\n" } /// Return the SHA of the root (oldest, no-parent) commit in a repo. /// Used to seed the applied repo at the template baseline rather than at /// `main`, so the first `ApplyCommit` diff shows the manager's real changes. async fn git_root_commit(dir: &Path) -> Result { let out = git_command() .current_dir(dir) .args(["rev-list", "--max-parents=0", "HEAD"]) .output() .await .with_context(|| format!("git rev-list --max-parents=0 HEAD in {}", dir.display()))?; if !out.status.success() { anyhow::bail!( "git rev-list --max-parents=0 failed: {}", String::from_utf8_lossy(&out.stderr).trim() ); } Ok(String::from_utf8_lossy(&out.stdout).trim().to_owned()) } async fn git_commit(dir: &Path, message: &str) -> Result<()> { git( dir, &[ "-c", &format!("user.name={GIT_NAME}"), "-c", &format!("user.email={GIT_EMAIL}"), "commit", "-m", message, ], ) .await } /// Spawn `git` honoring the `HYPERHIVE_GIT` env var (absolute path baked in /// by the NixOS module), falling back to bare `git` (PATH lookup) otherwise. #[must_use] pub fn git_command() -> Command { let exe = std::env::var("HYPERHIVE_GIT").unwrap_or_else(|_| "git".into()); Command::new(exe) } pub async fn git(dir: &Path, args: &[&str]) -> Result<()> { let out = git_command() .current_dir(dir) .args(args) .output() .await .with_context(|| format!("git {} in {}", args.join(" "), dir.display()))?; if !out.status.success() { bail!( "git {} failed ({}): {}", args.join(" "), out.status, String::from_utf8_lossy(&out.stderr).trim() ); } Ok(()) } /// Fetch the commit `sha` from the `src` git repo into `dst` and pin /// it as `refs/tags/`. Used at `request_apply_commit` time so /// hive-c0re captures an immutable handle on the manager's commit; /// subsequent amendments / force-pushes in `src` no longer affect /// what gets built. Returns the resolved full sha. /// /// `sha` must be a commit sha (short or full) — the caller /// (`submit_apply_commit`) shape-checks it first. We resolve it /// LOCALLY against `src` rather than asking the remote to resolve /// it: `git fetch :` treats the left side as a /// remote *ref name*, and a bare sha is not one ("couldn't find /// remote ref ..."). Fetching by sha would need a full 40-hex sha /// plus `uploadpack.allow*SHA1InWant` on the remote, which the /// proposed repos don't set. hive-c0re has direct read access to /// `src`, so a local `rev-parse` + a branch-glob fetch sidesteps /// the whole sha-want negotiation. pub async fn git_fetch_to_tag(dst: &Path, src: &Path, sha: &str, tag: &str) -> Result { let src_str = src.display().to_string(); // Resolve the (short-or-full) sha to a full sha against the // source repo. The `^{commit}` peel + non-zero exit on a missing // object means a typo'd / stale sha fails loudly right here. let full = git_rev_parse(src, &format!("{sha}^{{commit}}")) .await .with_context(|| format!("commit '{sha}' not found in proposed repo {src_str}"))?; // Bring src's objects into dst. Fetching every head pulls the // wanted commit's history (always reachable from a branch in the // manager's flow) into dst's object db without sha-want. git( dst, &[ "fetch", "--no-tags", &src_str, "+refs/heads/*:refs/remotes/proposal-src/*", ], ) .await?; // Pin the exact commit as the proposal tag. The objects are now // local so this resolves without touching the remote. git(dst, &["tag", tag, &full]).await.with_context(|| { format!("tag {tag} at {full}: commit not reachable from any branch in proposed repo") })?; Ok(full) } /// Resolve `refname` (a tag, branch, or sha) in `dir` to its full sha. pub async fn git_rev_parse(dir: &Path, refname: &str) -> Result { let out = git_command() .current_dir(dir) .args(["rev-parse", refname]) .output() .await .with_context(|| format!("git rev-parse {refname} in {}", dir.display()))?; if !out.status.success() { bail!( "git rev-parse {refname} failed ({}): {}", out.status, String::from_utf8_lossy(&out.stderr).trim() ); } Ok(String::from_utf8_lossy(&out.stdout).trim().to_owned()) } /// Plant a lightweight tag at `target`. Errors if the tag already /// exists — we want loud failures on id reuse, not silent /// overwrites. pub async fn git_tag(dir: &Path, name: &str, target: &str) -> Result<()> { git(dir, &["tag", name, target]).await } /// Plant an annotated tag with `body` as the message. Used for /// `failed/` (body = build error) and `denied/` (body = /// operator note). Multi-line bodies handled via stdin so we don't /// have to escape anything. pub async fn git_tag_annotated(dir: &Path, name: &str, target: &str, body: &str) -> Result<()> { use tokio::io::AsyncWriteExt; // Annotated tags are git objects, so they need a tagger identity // (same constraint as a commit). Pass the hive-c0re identity // inline rather than relying on a global git config — applied // repos are hive-c0re-owned and the host's user might not have // user.email set. let mut child = git_command() .current_dir(dir) .args([ "-c", &format!("user.name={GIT_NAME}"), "-c", &format!("user.email={GIT_EMAIL}"), "tag", "-a", name, target, "-F", "-", ]) .stdin(std::process::Stdio::piped()) .stdout(std::process::Stdio::piped()) .stderr(std::process::Stdio::piped()) .spawn() .with_context(|| format!("spawn git tag -a {name} in {}", dir.display()))?; if let Some(mut stdin) = child.stdin.take() { stdin .write_all(body.as_bytes()) .await .context("write tag body to git stdin")?; // Drop closes stdin so git can finish reading. drop(stdin); } let out = child.wait_with_output().await.context("wait git tag -a")?; if !out.status.success() { bail!( "git tag -a {name} failed ({}): {}", out.status, String::from_utf8_lossy(&out.stderr).trim() ); } Ok(()) } /// Replace working tree + index with the tree at `target` without /// moving HEAD. `applied/main` stays pointing at the last known-good /// `deployed/*` while we let `nixos-container update` evaluate the /// candidate. On build failure callers reset back to HEAD; on /// success they fast-forward main to `target`. pub async fn git_read_tree_reset(dir: &Path, target: &str) -> Result<()> { git(dir, &["read-tree", "--reset", "-u", target]).await } /// Hard-set a ref to `target`. Used to fast-forward `refs/heads/main` /// to the just-deployed proposal commit. Uses `update-ref`, not /// `branch -f`, so it works regardless of where HEAD currently sits. pub async fn git_update_ref(dir: &Path, refname: &str, target: &str) -> Result<()> { git(dir, &["update-ref", refname, target]).await } /// Write a systemd drop-in for `container@.service` that applies /// our default resource caps. Goes under `/run/systemd/system/...` so it's /// ephemeral (regenerated on every spawn / rebuild). async fn set_resource_limits(container: &str, cpu_quota: &str, memory_max: &str) -> Result<()> { crate::priv_client::write_resource_limits(container, memory_max, cpu_quota).await } async fn systemd_daemon_reload() -> Result<()> { crate::priv_client::daemon_reload().await } /// Idempotently rewrite the lines in `/etc/nixos-containers/.conf` /// that hive-c0re owns: `PRIVATE_NETWORK` (forced 0 so the agent's web UI port /// is reachable on the host) and `EXTRA_NSPAWN_FLAGS` (the runtime-dir bind). /// The start script expands `$EXTRA_NSPAWN_FLAGS` unquoted into the /// `systemd-nspawn` command. /// Where in the container's filesystem the manager sees its agents tree. /// Matches the `/agents` path that pre-Phase-8 hosts declared via /// `containers.root.bindMounts."/agents"`. pub const CONTAINER_MANAGER_AGENTS_MOUNT: &str = "/agents"; /// Where the manager sees the applied trees of every agent, read-only. /// Manager runs `git fetch /applied//.git refs/tags/*:refs/tags/applied/*` /// to learn what hive-c0re deployed (or rejected, or failed to /// build); the RO bind makes accidental writes impossible from /// inside the container. pub const CONTAINER_MANAGER_APPLIED_MOUNT: &str = "/applied"; /// The on-host root that gets bind-mounted to `/agents` inside the manager. /// Hard-coded to match `AGENT_STATE_ROOT` in coordinator.rs (kept duplicated /// here so lifecycle stays usable as a leaf module). const HOST_AGENTS_ROOT: &str = "/var/lib/hyperhive/agents"; /// On-host applied repo root, mirrored RO into the manager. Matches /// `APPLIED_STATE_ROOT` in coordinator.rs. const HOST_APPLIED_ROOT: &str = "/var/lib/hyperhive/applied"; /// On-host meta repo root, mirrored RO into the manager. Matches /// `meta::meta_dir()` but duplicated here so lifecycle stays a leaf. const HOST_META_ROOT: &str = "/var/lib/hyperhive/meta"; /// Shared directory accessible to all agents. All agents bind-mount this RW. const HOST_SHARED_ROOT: &str = "/var/lib/hyperhive/shared"; /// Append bind flags for `child`'s state, harness, and config dirs into /// `binds`, all read-write. The RW on `state` is deliberate (recovery), /// not an oversight; see docs/persistence.md ("Parent access to child /// state") for the rationale. Creates missing host-side directories so /// nspawn doesn't refuse to start; missing dirs are non-fatal. fn bind_child_agent_dirs(child: &str, binds: &mut Vec) { let state_dir = format!("{HOST_AGENTS_ROOT}/{child}/state"); let harness_dir = format!("{HOST_AGENTS_ROOT}/{child}/harness"); let config_dir = format!("{HOST_AGENTS_ROOT}/{child}/config"); for dir in [&state_dir, &harness_dir, &config_dir] { let _ = std::fs::create_dir_all(dir); } binds.push(BindMount { host_path: state_dir, container_path: format!("/agents/{child}/state"), read_only: false, }); binds.push(BindMount { host_path: harness_dir, container_path: format!("/agents/{child}/harness"), read_only: false, }); binds.push(BindMount { host_path: config_dir, container_path: format!("/agents/{child}/config"), read_only: false, }); } #[allow(clippy::too_many_lines)] async fn set_nspawn_flags( container: &str, runtime_dir: &Path, claude_dir: &Path, notes_dir: &Path, ) -> Result<()> { // Ensure /shared directory exists before binding. systemd-nspawn requires the bind source to exist. std::fs::create_dir_all(HOST_SHARED_ROOT) .with_context(|| format!("create {HOST_SHARED_ROOT}"))?; // Make /shared writable by every agent. Containers share host uids (no // PrivateUsers), but each agent is a distinct unix user, so a root-owned // 0755 dir leaves them unable to write — the documented "read/write for // all agents" contract was broken (#1374). A setgid group would need a // pinned GID declared in every container plus all agent users joined to // it (cross-container coordination + a rebuild cascade); instead we use // the /tmp model — sticky world-writable (1777). The sticky bit lets any // agent create files while protecting each agent's entries from deletion // by the others, and matches /shared's documented "free-for-all, may be // deleted/lost" semantics without touching any per-agent config. { use std::os::unix::fs::PermissionsExt as _; let perms = std::fs::Permissions::from_mode(0o1777); std::fs::set_permissions(HOST_SHARED_ROOT, perms) .with_context(|| format!("chmod 1777 {HOST_SHARED_ROOT}"))?; } // Ensure /knowledge dir exists. It may be empty until forge seeds it; // nspawn refuses to start if the bind source is missing entirely. std::fs::create_dir_all(crate::knowledge::LOCAL_DIR) .with_context(|| format!("create {}", crate::knowledge::LOCAL_DIR))?; // Logical agent name — strip the `h-` prefix. // For the manager: `h-ruth` → `ruth`. For sub-agents: `h-iris` → `iris`. let agent_name = container.strip_prefix(AGENT_PREFIX).unwrap_or(container); // Claude credentials land at `/home//.claude` so the // `claude` CLI (which reads `$HOME/.claude`) finds them. The // harness service's environment sets `HOME` to the same path // (`agent-base.nix` / `manager.nix`), so no `--setenv` plumbing // is needed here — the bind alone is enough. let claude_mount = container_claude_mount(agent_name); let mut binds: Vec = vec![ BindMount { host_path: runtime_dir.to_string_lossy().into_owned(), container_path: CONTAINER_RUNTIME_MOUNT.to_owned(), read_only: false, }, BindMount { host_path: claude_dir.to_string_lossy().into_owned(), container_path: claude_mount, read_only: false, }, BindMount { host_path: HOST_SHARED_ROOT.to_owned(), container_path: CONTAINER_SHARED_MOUNT.to_owned(), read_only: false, }, BindMount { host_path: crate::knowledge::LOCAL_DIR.to_owned(), container_path: crate::knowledge::CONTAINER_MOUNT.to_owned(), read_only: true, }, ]; // Own state, harness, and config dirs — same for every agent including // the manager. Config is RO: an agent must not edit its own config; changes // only ever flow through the approval queue. binds.push(BindMount { host_path: notes_dir.to_string_lossy().into_owned(), container_path: format!("/agents/{agent_name}/state"), read_only: false, }); if let Some(state_parent) = notes_dir.parent() { let harness_dir = state_parent.join("harness"); if !harness_dir.exists() { let _ = std::fs::create_dir_all(&harness_dir); } binds.push(BindMount { host_path: harness_dir.to_string_lossy().into_owned(), container_path: format!("/agents/{agent_name}/harness"), read_only: false, }); } let own_config = format!("{HOST_AGENTS_ROOT}/{agent_name}/config"); std::fs::create_dir_all(&own_config).with_context(|| format!("create {own_config}"))?; binds.push(BindMount { host_path: own_config, container_path: format!("/agents/{agent_name}/config"), read_only: true, }); // Topology-driven child mounts: every direct child of this agent gets // its state, harness, and config dirs bind-mounted RW (parent reads + // writes child state for recovery, and manages config). See // `bind_child_agent_dirs`. let direct_children = crate::topology::children_of(agent_name); for child in &direct_children { bind_child_agent_dirs(child, &mut binds); } // `can_manage_top_level_agents` role: additionally mount every // parentless agent in the topology as a virtual child. Enables // recovery — a role holder can update those agents' configs even // when they are down. Also grants RO access to /applied and /meta. if crate::topology::has_role( agent_name, crate::topology::ROLE_CAN_MANAGE_TOP_LEVEL_AGENTS, ) { let top_level = crate::topology::top_level_agents(); for tl in &top_level { if !direct_children.contains(tl) { bind_child_agent_dirs(tl, &mut binds); } } // systemd-nspawn refuses to start a container whose bind // source doesn't exist. The meta repo is created by the // startup migration, but make sure the directory is there // before the role holder comes up in case set_nspawn_flags // fires first (e.g. cold start with no agents). std::fs::create_dir_all(HOST_META_ROOT) .with_context(|| format!("create {HOST_META_ROOT}"))?; binds.push(BindMount { host_path: HOST_APPLIED_ROOT.to_owned(), container_path: CONTAINER_MANAGER_APPLIED_MOUNT.to_owned(), read_only: true, }); binds.push(BindMount { host_path: HOST_META_ROOT.to_owned(), container_path: crate::meta::CONTAINER_MANAGER_META_MOUNT.to_owned(), read_only: true, }); } // Web-socket subdir: bind-mount `/run/hive-agent//` into the // container so the harness can bind `web.sock` there and the host-side // gateway sees it. Subdir bind (not socket file) keeps the inode // visible after the harness unlinks a stale socket on rebind. // Applies to manager and sub-agents alike. let socket_dir = crate::agent_sockets::agent_dir_for(agent_name); std::fs::create_dir_all(&socket_dir) .with_context(|| format!("create {}", socket_dir.display()))?; // Chown to the agent user so the non-root harness can bind(2) here. // Falls back to 0777 on first spawn when uid lookup returns None // (container /etc/passwd not yet rendered). if let Some((uid, gid)) = agent_uid_gid(agent_name) { if let Err(e) = crate::priv_client::chown_socket_dir(agent_name, uid, gid).await { tracing::warn!(%agent_name, error = ?e, "chown socket dir failed"); } } else if let Err(e) = crate::priv_client::chmod_socket_dir(agent_name, 0o777).await { tracing::warn!(%agent_name, error = ?e, "chmod socket dir failed"); } binds.push(BindMount { host_path: socket_dir.to_string_lossy().into_owned(), container_path: socket_dir.to_string_lossy().into_owned(), read_only: false, }); // Network isolation: when HIVE_NETWORK_ISOLATION=1 is set (by the // hive-network.nix module's `isolateContainers` option), flip the // container to a private network namespace with a veth pair attached // to the host bridge. Applies to all containers including the manager // (all hive-c0re<->agent comms go through bind-mounted UDS, not TCP). let isolation = { let isolate = std::env::var("HIVE_NETWORK_ISOLATION").ok().as_deref() == Some("1"); let bridge = std::env::var("HIVE_NETWORK_BRIDGE").unwrap_or_default(); let subnet = std::env::var("HIVE_NETWORK_SUBNET").unwrap_or_default(); if isolate && !bridge.is_empty() && !subnet.is_empty() { let Some(agent_ip) = agent_network_ip(agent_name, &subnet) else { tracing::warn!( %agent_name, %subnet, "HIVE_NETWORK_SUBNET is set but could not derive a valid IP for agent \ (bad CIDR? prefix too narrow?); skipping PRIVATE_NETWORK write to \ avoid misconfigured isolation" ); return crate::priv_client::write_nspawn_flags(container, &binds, None).await; }; tracing::info!(%agent_name, %agent_ip, %bridge, "network isolation: PRIVATE_NETWORK=1"); Some(hive_sh4re::priv_proto::NetworkIsolation { agent_ip, bridge }) } else { None } }; // Delegate the actual conf-file rewrite to hive-priv (runs as root). crate::priv_client::write_nspawn_flags(container, &binds, isolation).await } /// Build the per-line callback for `create_container_streaming` / /// `update_container_streaming`. Both ops share identical dispatch logic /// (stdout → info + `append_stdout`, stderr → warn + `append_stderr`); this /// helper avoids duplicating that match body across the two call sites. fn make_log_callback( logs: Option>, log_id: Option, cmdline: String, ) -> impl FnMut(hive_sh4re::priv_proto::PrivStream, &str) { use hive_sh4re::priv_proto::PrivStream; move |stream, line| match stream { PrivStream::Stdout => { tracing::info!(target: "nixos-container", cmdline = %cmdline, "{line}"); if let (Some(h), Some(id)) = (&logs, log_id) { h.append_stdout(id, line); } } PrivStream::Stderr => { tracing::warn!(target: "nixos-container", cmdline = %cmdline, "{line}"); if let (Some(h), Some(id)) = (&logs, log_id) { h.append_stderr(id, line); } } } } /// Execute a container operation via hive-priv and integrate with /// `build_logs.sqlite`. hive-priv runs as root and forwards output lines /// to hive-c0re in real time via the streaming priv protocol. Each line /// is appended to the build-log row as it arrives, so the dashboard /// shows live progress during long `nixos-container create` / `update` runs. async fn priv_run(kind: &str, name: &str) -> Result<()> { priv_run_inner(kind, name, None).await } /// Like `priv_run` but calls `on_log_id(log_id)` immediately after the /// build-log row is opened — before the actual container op starts. /// This lets callers surface the row id for live streaming (e.g. the /// rebuild-queue worker sets `build_log_id` on the queue entry so the /// dashboard can link to `/api/build-logs/id/{id}/stream`). /// /// The callback fires only when a build-log row is successfully opened /// (i.e. the global `BuildLogs` handle is installed AND `h.start()` /// succeeds). No-op when `on_log_id` is `None` — that's the path for /// all callers that don't need the id. async fn priv_run_inner( kind: &str, name: &str, on_log_id: Option<&(dyn Fn(i64) + Send + Sync)>, ) -> Result<()> { let container = container_name(name); let cmdline = format!("nixos-container {kind} {container}"); let logs = crate::build_logs::global(); let log_id = logs.as_ref().and_then(|h| { h.start(name, kind, &cmdline) .map_err(|e| { tracing::warn!(error = ?e, "build_logs: start failed (priv_run log dropped)"); }) .ok() }); // Notify the caller as soon as the log row exists so it can surface // the id for live streaming before the container op even starts. if let (Some(id), Some(cb)) = (log_id, on_log_id) { cb(id); } // For long-running ops use the streaming protocol so build_logs // receives lines in real time rather than as a batch at completion. let result: Result<()> = match kind { "create" => { crate::priv_client::create_container_streaming( name, make_log_callback(logs.clone(), log_id, cmdline.clone()), ) .await } "update" => { crate::priv_client::update_container_streaming( name, make_log_callback(logs.clone(), log_id, cmdline.clone()), ) .await } "start" => crate::priv_client::start_container(name).await, "stop" => crate::priv_client::stop_container(name).await, "kill" => crate::priv_client::kill_container(name).await, "destroy" => crate::priv_client::destroy_container(name).await, other => Err(anyhow::anyhow!("unknown container op: {other}")), }; let succeeded = result.is_ok(); if let (Some(h), Some(id)) = (&logs, log_id) { h.finish( id, if succeeded { crate::build_logs::BuildStatus::Ok } else { crate::build_logs::BuildStatus::Fail }, ); } match result { Ok(()) => Ok(()), Err(e) => { let journal = if kind == "update" { container_journal_tail(&container).await } else { String::new() }; match log_id { Some(id) => bail!("{e:#}; see build log #{id}{journal}"), None => bail!("{e:#}{journal}"), } } } } /// On a failed `nixos-container update`, the stderr nixos-container /// itself prints is often terse ("failed to reload container") — the /// real reason (which unit failed `switch-to-configuration` during /// the reload phase) lands in the *container's* own journal, not on /// the host. Fetch the tail of it so a failed rebuild self-documents /// the failing unit in the error string, no second round-trip. /// /// Scoped to `update`: that's the reload-phase case, and the /// container is still up (running the old generation) so /// `journalctl -M` works. Best-effort — returns "" for other verbs /// or when the journal can't be read (machine gone, journalctl /// missing); it never produces an error of its own. async fn container_journal_tail(container: &str) -> String { // `-M` enters the container namespace and needs root, so the read // is delegated to hive-priv (hive-c0re itself runs unprivileged). let res = crate::priv_client::read_container_journal( container, hive_sh4re::priv_proto::JournalQuery { lines: 40, ..Default::default() }, ) .await; match res { Ok((stdout, _)) if !stdout.is_empty() => format!( "\n--- last 40 journal lines from container '{container}' ---\n{}", stdout.trim_end() ), _ => String::new(), } } #[cfg(test)] mod tests { use super::*; /// Regression test: `setup_proposed` must seed both agent.nix and flake.nix /// in the initial commit. Before commit 5b5a93e flake.nix was missing from /// the scaffold, requiring manual creation (seen with the damocles agent). #[tokio::test] async fn setup_proposed_seeds_flake_nix() { let dir = tempfile::tempdir().expect("tempdir"); let proposed = dir.path().join("proposed"); setup_proposed(&proposed, "test-agent") .await .expect("setup_proposed"); // Both files must exist on disk. assert!(proposed.join("agent.nix").exists(), "agent.nix missing"); assert!(proposed.join("flake.nix").exists(), "flake.nix missing"); // flake.nix must export nixosModules.default (the meta-flake contract). let flake = std::fs::read_to_string(proposed.join("flake.nix")).unwrap(); assert!( flake.contains("nixosModules.default"), "flake.nix does not export nixosModules.default" ); // Both files must be tracked in the initial git commit. let out = git_command() .current_dir(&proposed) .args(["show", "--name-only", "--format=", "HEAD"]) .output() .await .expect("git show"); let tracked = String::from_utf8_lossy(&out.stdout); assert!(tracked.contains("agent.nix"), "agent.nix not committed"); assert!(tracked.contains("flake.nix"), "flake.nix not committed"); } #[test] fn agent_network_ip_is_in_subnet() { // Default subnet 10.42.0.0/24 — agents get .2 to .254. let ip = agent_network_ip("alice", "10.42.0.0/24").expect("should produce an IP"); let octets: Vec = ip.split('.').map(|o| o.parse().unwrap()).collect(); assert_eq!(&octets[..3], &[10, 42, 0], "wrong /24 prefix"); assert!( octets[3] >= 2 && octets[3] <= 254, "host byte {}", octets[3] ); } #[test] fn agent_network_ip_stable() { // Same name + subnet must always produce the same IP. let a = agent_network_ip("damocles", "10.42.0.0/24"); let b = agent_network_ip("damocles", "10.42.0.0/24"); assert_eq!(a, b); } #[test] fn agent_network_ip_different_agents() { // Different agent names very likely produce different IPs (not guaranteed, // but for these two names the hashes don't collide). let alice = agent_network_ip("alice", "10.42.0.0/24").unwrap(); let bob = agent_network_ip("bob", "10.42.0.0/24").unwrap(); assert_ne!(alice, bob, "alice and bob collide — rename one"); } #[test] fn agent_network_ip_different_subnet() { let ip = agent_network_ip("alice", "192.168.5.0/24").expect("should produce an IP"); let octets: Vec = ip.split('.').map(|o| o.parse().unwrap()).collect(); assert_eq!(&octets[..3], &[192, 168, 5]); } #[test] fn agent_network_ip_rejects_bad_input() { assert!(agent_network_ip("alice", "notanip/24").is_none()); assert!(agent_network_ip("alice", "10.0.0.0/33").is_none()); // prefix > 32 assert!(agent_network_ip("alice", "10.0.0.0/31").is_none()); // too small assert!(agent_network_ip("alice", "10.0.0.0").is_none()); // no prefix } #[test] fn agent_network_ip_normalizes_bridge_ip_subnet() { // HIVE_NETWORK_SUBNET carries the bridge IP (10.42.0.1/24), not // canonical network (10.42.0.0/24). Both must produce the same result // after host-bit masking. let from_bridge = agent_network_ip("alice", "10.42.0.1/24"); let from_canonical = agent_network_ip("alice", "10.42.0.0/24"); assert_eq!( from_bridge, from_canonical, "bridge-IP and canonical-network form should normalize to the same result" ); // Result must still be in .2-.254. let ip = from_bridge.unwrap(); let last: u8 = ip.rsplit('.').next().unwrap().parse().unwrap(); assert!((2..=254).contains(&last), "host byte {last}"); } /// `setup_proposed` is idempotent: calling it on an existing repo is a /// no-op (the fresh guard skips all writes). #[tokio::test] async fn setup_proposed_idempotent() { let dir = tempfile::tempdir().expect("tempdir"); let proposed = dir.path().join("proposed"); setup_proposed(&proposed, "test-agent") .await .expect("first call"); // Second call must not error even though .git already exists. setup_proposed(&proposed, "test-agent") .await .expect("second call"); // Still one commit. let out = git_command() .current_dir(&proposed) .args(["rev-list", "--count", "HEAD"]) .output() .await .expect("git rev-list"); let count = String::from_utf8_lossy(&out.stdout).trim().to_owned(); assert_eq!( count, "1", "expected exactly one commit after idempotent call" ); } }