jobs are now DAGs of primitive nodes (prebuild, stop-for-update, swap, reconcile, signal, drain, ...) driven by one scheduler with N build slots + per-agent lifecycle leases. per-agent power intent (wanted up/offline) is durable in agent_power.sqlite; Reconcile nodes converge observed state to it. kills the graceful-stop watcher thread, the deferred-start follow-up, and the cascade pre-enqueue (fan-out on MetaLock completion instead). tracker: #2166
1848 lines
76 KiB
Rust
1848 lines
76 KiB
Rust
//! `nixos-container` lifecycle + per-agent config flake generation.
|
|
|
|
use std::path::Path;
|
|
|
|
use anyhow::{Context, Result, bail};
|
|
use hive_sh4re::priv_proto::{BindMount, CredentialMount};
|
|
use tokio::process::Command;
|
|
|
|
use crate::coordinator::{AgentPaths, HiveEnv};
|
|
|
|
/// Sub-agent container prefix. `nixos-container` caps the total container name
|
|
/// at 11 chars (it gets encoded into network interface names), so the agent
|
|
/// name itself can be at most `MAX_AGENT_NAME` chars.
|
|
pub const AGENT_PREFIX: &str = "h-";
|
|
pub const MAX_AGENT_NAME: usize = 9;
|
|
/// Logical name of the manager agent (broker recipient, state-dir key,
|
|
/// meta flake attribute). All persistent state lives under `ruth/`.
|
|
pub const MANAGER_NAME: &str = "ruth";
|
|
/// Container name of the manager. Uses the same `h-` prefix as sub-agents
|
|
/// so `nixos-container list` output is uniform and the list filter is
|
|
/// a single `starts_with(AGENT_PREFIX)` check. Logical name → container
|
|
/// name: `ruth` → `h-ruth`.
|
|
pub const MANAGER_CONTAINER: &str = "h-ruth";
|
|
|
|
/// Mount point of the per-agent runtime directory inside the container.
|
|
pub const CONTAINER_RUNTIME_MOUNT: &str = "/run/hive";
|
|
|
|
/// Where the per-agent Claude credentials dir mounts inside the
|
|
/// container. The harness service runs as a non-root unix user
|
|
/// whose home is `/home/<agent>/`, so the mount path varies per
|
|
/// agent — `container_claude_mount(name)` returns
|
|
/// `/home/<name>/.claude` for every agent including the manager.
|
|
/// `claude` inside the container reads
|
|
/// `$HOME/.claude` and the service environment sets `HOME` to the
|
|
/// same path, so the OAuth session survives container restarts.
|
|
#[must_use]
|
|
pub fn container_claude_mount(name: &str) -> String {
|
|
format!("/home/{name}/.claude")
|
|
}
|
|
|
|
/// Mount point of the shared directory accessible to all agents.
|
|
/// All agents can read/write here; agents should only put things they're
|
|
/// willing to lose (other agents may delete them).
|
|
pub const CONTAINER_SHARED_MOUNT: &str = "/shared";
|
|
|
|
const GIT_NAME: &str = "c0re";
|
|
const GIT_EMAIL: &str = "c0re@hyperhive.local";
|
|
|
|
/// Sub-agent web UI port range. Deterministic from the agent's name (FNV-1a
|
|
/// hash mod range size), so the dashboard can compute the same port without
|
|
/// asking hive-c0re.
|
|
const WEB_PORT_BASE: u16 = 8100;
|
|
const WEB_PORT_RANGE: u16 = 900;
|
|
|
|
/// FNV-1a hash of a string — shared by `agent_web_port` and
|
|
/// `agent_network_ip` so the derivation rule is identical.
|
|
fn fnv1a(s: &str) -> u32 {
|
|
let mut hash: u32 = 2_166_136_261;
|
|
for b in s.bytes() {
|
|
hash ^= u32::from(b);
|
|
hash = hash.wrapping_mul(16_777_619);
|
|
}
|
|
hash
|
|
}
|
|
|
|
/// Per-agent web UI port — `WEB_PORT_BASE + FNV-1a(name) %
|
|
/// WEB_PORT_RANGE` for every agent including the manager. The port
|
|
/// allocation rule reads the same for every name; collisions are
|
|
/// possible (birthday paradox at ~30 agents) and the operator
|
|
/// resolves them by renaming an agent (different hash → different
|
|
/// port). Stable across hosts, restarts, and dashboard renders —
|
|
/// no state-file dance.
|
|
#[must_use]
|
|
pub fn agent_web_port(name: &str) -> u16 {
|
|
// Modulo of a u32 by a u16's value is guaranteed < u16::MAX, so try_from never fails.
|
|
WEB_PORT_BASE + u16::try_from(fnv1a(name) % u32::from(WEB_PORT_RANGE)).unwrap_or(0)
|
|
}
|
|
|
|
/// Deterministic IPv4 address for an agent inside an isolated subnet.
|
|
///
|
|
/// Parses `subnet_cidr` as `<network_ip>/<prefix_len>` (e.g.
|
|
/// `"10.42.0.0/24"`), then computes:
|
|
///
|
|
/// ```text
|
|
/// host_count = 2^(32 - prefix_len)
|
|
/// usable = host_count - 3 // skip .0 (network), .1 (gateway), .255 (broadcast)
|
|
/// offset = FNV-1a(name) % usable + 2 // .2 is the first agent slot
|
|
/// agent_ip = network_base_u32 + offset
|
|
/// ```
|
|
///
|
|
/// Returns `None` when `subnet_cidr` can't be parsed (invalid format,
|
|
/// prefix out of range, etc.) so callers can fall back gracefully.
|
|
/// Collisions are possible (birthday paradox) and the operator resolves
|
|
/// them by renaming an agent, same as for port collisions.
|
|
#[must_use]
|
|
pub fn agent_network_ip(name: &str, subnet_cidr: &str) -> Option<String> {
|
|
let (ip_str, prefix_str) = subnet_cidr.split_once('/')?;
|
|
let prefix_len: u32 = prefix_str.parse().ok()?;
|
|
if prefix_len > 30 {
|
|
// /31 and /32 have no room for agents; /30 has 1 usable slot.
|
|
// /0 (the other extreme) is handled further down: host_count
|
|
// overflows checked_shl(32) → 0 → usable = 0 → None.
|
|
return None;
|
|
}
|
|
// Parse dotted-decimal IPv4.
|
|
let octets: Vec<u8> = ip_str
|
|
.split('.')
|
|
.map(|o| o.parse::<u8>().ok())
|
|
.collect::<Option<Vec<_>>>()?;
|
|
if octets.len() != 4 {
|
|
return None;
|
|
}
|
|
let base_u32 = u32::from_be_bytes([octets[0], octets[1], octets[2], octets[3]]);
|
|
// Mask off host bits to get the true network address.
|
|
let mask = if prefix_len == 0 {
|
|
0u32
|
|
} else {
|
|
!0u32 << (32 - prefix_len)
|
|
};
|
|
let network_base = base_u32 & mask;
|
|
let host_count: u32 = 1u32.checked_shl(32 - prefix_len).unwrap_or(0);
|
|
// `.0` = network, `.1` = bridge gateway, last = broadcast → 3 reserved.
|
|
let usable = host_count.saturating_sub(3);
|
|
if usable == 0 {
|
|
return None;
|
|
}
|
|
let offset = fnv1a(name) % usable + 2; // +2: skip .0 and .1
|
|
let ip_u32 = network_base + offset;
|
|
let [a, b, c, d] = ip_u32.to_be_bytes();
|
|
Some(format!("{a}.{b}.{c}.{d}"))
|
|
}
|
|
|
|
/// Extract the bridge gateway IP from `HIVE_NETWORK_SUBNET`.
|
|
///
|
|
/// `HIVE_NETWORK_SUBNET` carries the host-side bridge address verbatim
|
|
/// (e.g. `10.42.0.1/24`), **not** the canonical network address — see
|
|
/// the note in `set_nspawn_flags` + `docs/network.md`. The IP part is
|
|
/// therefore the bridge IP itself: the host end of the bridge, the
|
|
/// default-route target for isolated containers, and the address the
|
|
/// hive dnsmasq resolver binds. Returns the dotted-decimal IP with the
|
|
/// `/<prefix>` stripped, or `None` if the input isn't a valid
|
|
/// `<ipv4>/<prefix>` pair.
|
|
///
|
|
/// Deliberately returns the operator-configured address verbatim rather
|
|
/// than deriving `network + 1`: an operator who sets `bridgeIp` to a
|
|
/// non-`.1` host address (e.g. `10.42.0.254`) runs the bridge + resolver
|
|
/// there, so that — not `.1` — is the real gateway.
|
|
#[must_use]
|
|
pub fn bridge_gateway_ip(subnet_cidr: &str) -> Option<String> {
|
|
let (ip_str, prefix_str) = subnet_cidr.split_once('/')?;
|
|
// Validate the prefix is a sane IPv4 CIDR length and the address is
|
|
// dotted-decimal IPv4 — same shape `agent_network_ip` accepts — so a
|
|
// malformed `HIVE_NETWORK_SUBNET` can't smuggle a bogus HOST_ADDRESS
|
|
// into the nspawn conf.
|
|
let prefix_len: u32 = prefix_str.parse().ok()?;
|
|
if prefix_len > 32 {
|
|
return None;
|
|
}
|
|
let octets: Vec<u8> = ip_str
|
|
.split('.')
|
|
.map(|o| o.parse::<u8>().ok())
|
|
.collect::<Option<Vec<_>>>()?;
|
|
if octets.len() != 4 {
|
|
return None;
|
|
}
|
|
Some(ip_str.to_owned())
|
|
}
|
|
|
|
#[must_use]
|
|
pub fn container_name(name: &str) -> String {
|
|
format!("{AGENT_PREFIX}{name}")
|
|
}
|
|
|
|
/// Read the agent user's `(uid, gid)` from the container's nixos-managed
|
|
/// `/etc/passwd`. Returns `None` when the container hasn't been built
|
|
/// yet, the passwd file is unparseable, or the agent user is missing
|
|
/// (e.g. legacy container that still runs as root).
|
|
///
|
|
/// Used by `forge` + `matrix` after writing per-agent state files so
|
|
/// the bind-mounted host file ends up readable by the agent user
|
|
/// without waiting for the next container activation to run the chown
|
|
/// fixup.
|
|
///
|
|
/// Notes:
|
|
/// - Reads the *container-local* passwd at
|
|
/// `/var/lib/nixos-containers/<container>/etc/passwd`, not the host's.
|
|
/// The container's user-namespace shares uids with the host (no
|
|
/// `PrivateUsers`), so the uid is directly usable in host-side
|
|
/// `chown(2)`.
|
|
/// - Best-effort: caller treats `None` as "skip the chown".
|
|
#[must_use]
|
|
pub fn agent_uid_gid(agent_name: &str) -> Option<(u32, u32)> {
|
|
let container = container_name(agent_name);
|
|
let passwd_path = format!("/var/lib/nixos-containers/{container}/etc/passwd");
|
|
let content = std::fs::read_to_string(&passwd_path).ok()?;
|
|
for line in content.lines() {
|
|
let mut parts = line.split(':');
|
|
let user = parts.next()?;
|
|
if user != agent_name {
|
|
continue;
|
|
}
|
|
let _ = parts.next()?; // x (password placeholder)
|
|
let uid: u32 = parts.next()?.parse().ok()?;
|
|
let gid: u32 = parts.next()?.parse().ok()?;
|
|
return Some((uid, gid));
|
|
}
|
|
None
|
|
}
|
|
|
|
/// Best-effort `chown(path, agent_uid, agent_gid)`. Resolves the agent's
|
|
/// uid/gid via [`agent_uid_gid`] and shells out to `std::os::unix::fs::chown`.
|
|
/// Silently no-ops when the container isn't built yet (`None` from
|
|
/// [`agent_uid_gid`]) and logs at debug on chown syscall failure — the
|
|
/// activation script in `harness-base.nix` is the steady-state safety
|
|
/// net. Used by per-agent state writers in `forge` + `matrix` so the
|
|
/// agent can read the file without waiting for the next container
|
|
/// rebuild.
|
|
pub fn chown_to_agent(name: &str, path: &Path, subsystem: &str) {
|
|
let Some((uid, gid)) = agent_uid_gid(name) else {
|
|
return;
|
|
};
|
|
if let Err(e) = std::os::unix::fs::chown(path, Some(uid), Some(gid)) {
|
|
tracing::debug!(%name, %subsystem, path = %path.display(), error = %e, "chown to agent failed");
|
|
}
|
|
}
|
|
|
|
fn validate(name: &str) -> Result<()> {
|
|
if name.is_empty() {
|
|
bail!("agent name must not be empty");
|
|
}
|
|
if name.len() > MAX_AGENT_NAME {
|
|
bail!(
|
|
"agent name '{name}' is too long ({} chars); max {MAX_AGENT_NAME}",
|
|
name.len()
|
|
);
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// First name (≠ `self_name`) currently running whose hashed port
|
|
/// matches this agent's. The harness inside the colliding container
|
|
/// would otherwise loop on `AddrInUse` forever; we surface the
|
|
/// conflict here so spawn / rebuild fails loudly with an actionable
|
|
/// message instead.
|
|
async fn port_collision(self_name: &str) -> Option<String> {
|
|
let port = agent_web_port(self_name);
|
|
let raw = list().await.unwrap_or_default();
|
|
for c in raw {
|
|
let Some(other) = c.strip_prefix(AGENT_PREFIX) else {
|
|
continue;
|
|
};
|
|
if other == self_name {
|
|
continue;
|
|
}
|
|
if agent_web_port(other) == port && is_running(other).await {
|
|
return Some(other.to_owned());
|
|
}
|
|
}
|
|
None
|
|
}
|
|
|
|
pub async fn spawn(name: &str, hive: &HiveEnv, paths: &AgentPaths) -> Result<()> {
|
|
create_container(name, hive, paths).await?;
|
|
write_dropins(name, hive, paths).await?;
|
|
priv_run("start", name).await
|
|
}
|
|
|
|
/// First-spawn provisioning + `nixos-container create`, without the
|
|
/// drop-in write or the start — the job queue's `Create` node.
|
|
/// `spawn` composes this with `write_dropins` + start for direct
|
|
/// callers (root-agent bootstrap).
|
|
pub async fn create_container(name: &str, hive: &HiveEnv, paths: &AgentPaths) -> Result<()> {
|
|
validate(name)?;
|
|
if let Some(other) = port_collision(name).await {
|
|
bail!(
|
|
"port {} is already taken by '{other}' — rename one of them and retry",
|
|
agent_web_port(name)
|
|
);
|
|
}
|
|
setup_proposed(&paths.proposed_dir, name).await?;
|
|
setup_applied(&paths.applied_dir, Some(&paths.proposed_dir), name).await?;
|
|
ensure_agent_state_subvolume(name).await?;
|
|
ensure_claude_dir(&paths.claude_dir)?;
|
|
ensure_state_dir(&paths.notes_dir)?;
|
|
// Meta flake gets the new agent's input + nixosConfiguration
|
|
// before `nixos-container create` so the `--flake meta#<name>`
|
|
// ref resolves.
|
|
let agents = agents_after_spawn(name).await?;
|
|
crate::meta::sync_agents(hive, &agents).await?;
|
|
priv_run("create", name).await
|
|
}
|
|
|
|
/// Re-apply the per-container host-side config: nspawn flags (bind
|
|
/// mounts etc.), the systemd resource-limits drop-in, and a daemon
|
|
/// reload so both take effect on the next unit (re)start. Idempotent —
|
|
/// the job queue's `WriteDropin` node, also folded into every `Swap`
|
|
/// (rebuild is the reconcile verb).
|
|
pub async fn write_dropins(name: &str, hive: &HiveEnv, paths: &AgentPaths) -> Result<()> {
|
|
validate(name)?;
|
|
let container = container_name(name);
|
|
set_nspawn_flags(
|
|
&container,
|
|
&paths.agent_dir,
|
|
&paths.claude_dir,
|
|
&paths.notes_dir,
|
|
)
|
|
.await?;
|
|
set_resource_limits(&container, &hive.agent_cpu_quota, &hive.agent_memory_max).await?;
|
|
systemd_daemon_reload().await
|
|
}
|
|
|
|
/// Rebuild-path preamble shared by the job queue's `Prebuild` node and
|
|
/// `rebuild_no_meta`: fail fast on a port collision, then make sure
|
|
/// the applied repo + state dirs exist. Container untouched.
|
|
pub async fn prepare_rebuild_dirs(name: &str, paths: &AgentPaths) -> Result<()> {
|
|
validate(name)?;
|
|
if let Some(other) = port_collision(name).await {
|
|
bail!(
|
|
"port {} is already taken by '{other}' — rename one of them and retry",
|
|
agent_web_port(name)
|
|
);
|
|
}
|
|
setup_applied(&paths.applied_dir, None, name).await?;
|
|
ensure_agent_state_subvolume(name).await?;
|
|
ensure_claude_dir(&paths.claude_dir)?;
|
|
ensure_state_dir(&paths.notes_dir)?;
|
|
Ok(())
|
|
}
|
|
|
|
/// Profile-swap for an existing, stopped container: re-apply the
|
|
/// drop-ins, then `nixos-container update`. The job queue's `Swap`
|
|
/// node. Requires the container stopped (the queue's `StopForUpdate`
|
|
/// upstream); does NOT start it — the DAG's tail `Reconcile` owns
|
|
/// bringing the agent back to its wanted power state.
|
|
pub async fn swap_update(
|
|
name: &str,
|
|
hive: &HiveEnv,
|
|
paths: &AgentPaths,
|
|
on_step: &(dyn Fn(&str) + Send + Sync),
|
|
on_build_log_id: &(dyn Fn(i64) + Send + Sync),
|
|
) -> Result<()> {
|
|
write_dropins(name, hive, paths).await?;
|
|
on_step("nixos-container update");
|
|
priv_run_inner("update", name, Some(on_build_log_id)).await
|
|
}
|
|
|
|
/// Build the `AgentSpec` list for the meta flake from `nixos-container
|
|
/// list` + a hypothetical extra name not yet in the list (for spawn
|
|
/// where the new agent's container doesn't exist yet). Pass empty
|
|
/// `name_to_add` from rebuild paths where the agent is already in the
|
|
/// container list.
|
|
///
|
|
/// Propagates errors from `list()` rather than swallowing them.
|
|
/// Using `.unwrap_or_default()` here would silently produce an empty
|
|
/// agent list when `nixos-container list` fails (priv helper down, race),
|
|
/// which `sync_agents` would then commit to meta — dropping every agent
|
|
/// from `flake.nix`. Callers that can tolerate failures (e.g. migration)
|
|
/// handle the `Err` themselves with `.unwrap_or_default()`.
|
|
async fn agents_for_meta(name_to_add: Option<&str>) -> Result<Vec<crate::meta::AgentSpec>> {
|
|
let containers = list().await?;
|
|
let mut out: Vec<crate::meta::AgentSpec> = containers
|
|
.into_iter()
|
|
.filter_map(|c| {
|
|
let name = c.strip_prefix(AGENT_PREFIX)?.to_owned();
|
|
Some(crate::meta::AgentSpec {
|
|
is_manager: name == MANAGER_NAME,
|
|
port: agent_web_port(&name),
|
|
name,
|
|
})
|
|
})
|
|
.collect();
|
|
if let Some(extra) = name_to_add
|
|
&& !out.iter().any(|a| a.name == extra)
|
|
{
|
|
out.push(crate::meta::AgentSpec {
|
|
is_manager: extra == MANAGER_NAME,
|
|
port: agent_web_port(extra),
|
|
name: extra.to_owned(),
|
|
});
|
|
}
|
|
out.sort_by(|a, b| a.name.cmp(&b.name));
|
|
Ok(out)
|
|
}
|
|
|
|
async fn agents_after_spawn(name: &str) -> Result<Vec<crate::meta::AgentSpec>> {
|
|
agents_for_meta(Some(name)).await
|
|
}
|
|
|
|
/// Like `agents_for_meta_listing` but with an extra agent added (for a
|
|
/// container that doesn't exist yet). Used by the first-spawn path in
|
|
/// `actions::run_apply_commit` to register the new agent in meta before
|
|
/// `prepare_deploy` tries to update its input lock.
|
|
pub async fn agents_for_meta_listing_with(extra: &str) -> Result<Vec<crate::meta::AgentSpec>> {
|
|
agents_for_meta(Some(extra)).await
|
|
}
|
|
|
|
/// Public enumeration of currently-existing agents (whatever
|
|
/// `nixos-container list` says), sorted, no extras. For callers
|
|
/// outside this module that need to reseed meta after lifecycle
|
|
/// changes — destroy, startup reconciliation, etc.
|
|
pub async fn agents_for_meta_listing() -> Result<Vec<crate::meta::AgentSpec>> {
|
|
agents_for_meta(None).await
|
|
}
|
|
|
|
/// True when the named container already exists (appears in
|
|
/// `nixos-container list`). Used by the apply-commit path to decide
|
|
/// between first-spawn (`nixos-container create`) and normal rebuild
|
|
/// (`nixos-container update`).
|
|
pub async fn container_exists(name: &str) -> bool {
|
|
let container = container_name(name);
|
|
list()
|
|
.await
|
|
.unwrap_or_default()
|
|
.iter()
|
|
.any(|c| c == &container)
|
|
}
|
|
|
|
pub async fn kill(name: &str) -> Result<()> {
|
|
validate(name)?;
|
|
priv_run("stop", name).await
|
|
}
|
|
|
|
pub async fn start(name: &str) -> Result<()> {
|
|
validate(name)?;
|
|
priv_run("start", name).await
|
|
}
|
|
|
|
/// Start with the cold-start fallback: when a plain start fails (the
|
|
/// activation-error shape), retry once via stop + kill + start before
|
|
/// giving up. Used by the queue's fast-lane `Start` handler and the
|
|
/// inline start-after-rebuild path.
|
|
/// See `docs/coordinator.md::Cold-start fallback`.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Propagates the retry's start error (annotated with the original
|
|
/// failure) when the fallback also fails.
|
|
pub async fn start_with_fallback(name: &str) -> Result<()> {
|
|
validate(name)?;
|
|
if let Err(start_err) = priv_run("start", name).await {
|
|
let container = container_name(name);
|
|
tracing::warn!(
|
|
container = %container,
|
|
error = %start_err,
|
|
"start failed (possible activation error); retrying via stop + kill + start"
|
|
);
|
|
priv_run("stop", name).await.unwrap_or_else(|e| {
|
|
tracing::warn!(
|
|
container = %container,
|
|
error = %e,
|
|
"stop before cold-start retry failed (ignored)"
|
|
);
|
|
});
|
|
priv_run("kill", name).await.unwrap_or_else(|e| {
|
|
tracing::warn!(
|
|
container = %container,
|
|
error = %e,
|
|
"kill before cold-start retry failed (ignored)"
|
|
);
|
|
});
|
|
priv_run("start", name).await.map_err(|e| {
|
|
anyhow::anyhow!(
|
|
"cold-start fallback also failed: {e:#} \
|
|
(original start error: {start_err:#})"
|
|
)
|
|
})
|
|
} else {
|
|
Ok(())
|
|
}
|
|
}
|
|
|
|
/// Stop + start without regenerating any config. For "kick the container"
|
|
/// without touching the flake or nspawn flags.
|
|
pub async fn restart(name: &str) -> Result<()> {
|
|
kill(name).await?;
|
|
start(name).await
|
|
}
|
|
|
|
/// True when the container's systemd unit is active. Used by the dashboard
|
|
/// to gate stop/restart buttons.
|
|
pub async fn is_running(name: &str) -> bool {
|
|
let container = container_name(name);
|
|
let unit = format!("container@{container}.service");
|
|
Command::new("systemctl")
|
|
.args(["is-active", "--quiet", &unit])
|
|
.status()
|
|
.await
|
|
.is_ok_and(|s| s.success())
|
|
}
|
|
|
|
/// Fully tear down a sub-agent's container: stop + remove via `nixos-container
|
|
/// destroy`, then clean our own systemd drop-in. Leaves it to the caller to
|
|
/// wipe `/var/lib/hyperhive/...` state and the per-agent runtime dir.
|
|
pub async fn destroy(name: &str) -> Result<()> {
|
|
validate(name)?;
|
|
let container = container_name(name);
|
|
// nixos-container destroy handles stop + removal of /var/lib/nixos-containers/<C>
|
|
// and /etc/nixos-containers/<C>.conf. Tolerate "no such container".
|
|
if let Err(e) = priv_run("destroy", name).await {
|
|
tracing::warn!(error = ?e, "nixos-container destroy returned an error; continuing cleanup");
|
|
}
|
|
// Remove the systemd resource-limits drop-in via hive-priv.
|
|
if let Err(e) = crate::priv_client::remove_service_dropin(&container).await {
|
|
tracing::warn!(error = ?e, "remove service drop-in failed (non-fatal)");
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Rebuild `name`'s container: sync the meta flake, optionally re-lock
|
|
/// the agent's input, then re-apply + restart via `nixos-container`.
|
|
///
|
|
/// When `relock` is `true` the agent's meta input is bumped to whatever
|
|
/// `applied/<n>/main` points at before the build. Pass `false` for
|
|
/// meta-update cascade rebuilds, where re-locking would revert the bump
|
|
/// the cascade just committed (see the inline note below).
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Propagates errors from meta-flake sync / lock-update and the
|
|
/// `nixos-container` apply + restart shellouts.
|
|
///
|
|
/// Returns `true` when `defer_start` suppressed the start-after-update —
|
|
/// the caller owns bringing the container back up (see
|
|
/// [`rebuild_no_meta`]).
|
|
pub async fn rebuild(
|
|
name: &str,
|
|
hive: &HiveEnv,
|
|
paths: &AgentPaths,
|
|
relock: bool,
|
|
defer_start: bool,
|
|
on_step: &(dyn Fn(&str) + Send + Sync),
|
|
on_build_log_id: &(dyn Fn(i64) + Send + Sync),
|
|
) -> Result<bool> {
|
|
// Sync the meta flake (idempotent — no-op when the rendered
|
|
// flake matches disk) so a manual rebuild from the dashboard
|
|
// can also recover from a divergent meta repo (e.g. an agent
|
|
// got added directly via `nixos-container create` outside
|
|
// hive-c0re).
|
|
let agents = agents_for_meta(None).await?;
|
|
crate::meta::sync_agents(hive, &agents).await?;
|
|
// Then bump just this agent's input — picks up whatever
|
|
// `applied/<n>/main` currently points at (deployed/<latest>).
|
|
// Commits the lock if it changed.
|
|
//
|
|
// `relock = false` skips this: a meta-update cascade has *just* set
|
|
// the meta lock deliberately, and `lock_update_for_rebuild` re-runs
|
|
// `nix flake update agent-<name>`, which re-resolves the agent's
|
|
// transitive inputs back to the agent's own flake.lock — reverting
|
|
// the input the meta-update just bumped. Cascade rebuilds therefore
|
|
// build against the freshly-set on-disk lock as-is.
|
|
if relock {
|
|
crate::meta::lock_update_for_rebuild(name).await?;
|
|
}
|
|
rebuild_no_meta(name, hive, paths, defer_start, on_step, on_build_log_id).await
|
|
}
|
|
|
|
/// Container-level rebuild without touching the meta repo. Callers
|
|
/// that own the meta side themselves (`actions::run_apply_commit`
|
|
/// drives meta through the two-phase prepare/finalize/abort flow)
|
|
/// use this directly. Public `rebuild` wraps it with idempotent meta
|
|
/// sync + lock-bump-and-commit.
|
|
///
|
|
/// `on_step` is called at each phase boundary with a short human-readable
|
|
/// label so callers can surface progress (e.g. update the rebuild-queue
|
|
/// step shown in the dashboard). Pass `&|_| ()` when progress reporting
|
|
/// is not needed.
|
|
///
|
|
/// `on_build_log_id` is called with the build-log row id immediately after
|
|
/// the `nixos-container update` log row opens, before the actual update
|
|
/// command starts. Callers can use this to link the queue entry to the log
|
|
/// for live streaming. Pass `&|_| ()` when not needed.
|
|
///
|
|
/// `defer_start` skips the start-after-update for a previously-running
|
|
/// container and returns `true` instead, so a queue-side caller can hand
|
|
/// the (potentially slow) container boot to the fast lane rather than
|
|
/// holding the serialized build lane through it. With `defer_start =
|
|
/// false` the start (with cold-start fallback) runs inline as before and
|
|
/// the return value is always `false`. The spawn path always starts
|
|
/// inline — a freshly-created container boots as part of provisioning.
|
|
pub async fn rebuild_no_meta(
|
|
name: &str,
|
|
hive: &HiveEnv,
|
|
paths: &AgentPaths,
|
|
defer_start: bool,
|
|
on_step: &(dyn Fn(&str) + Send + Sync),
|
|
on_build_log_id: &(dyn Fn(i64) + Send + Sync),
|
|
) -> Result<bool> {
|
|
prepare_rebuild_dirs(name, paths).await?;
|
|
let flake_ref = format!("{}#{name}", crate::meta::meta_dir().display());
|
|
if container_exists(name).await {
|
|
// Rebuild strategy: stop-before-update + pre-build.
|
|
// See `docs/coordinator.md::Container lifecycle`.
|
|
let was_running = is_running(name).await;
|
|
write_dropins(name, hive, paths).await?;
|
|
if was_running {
|
|
on_step("nix build");
|
|
prebuild_toplevel(name, &flake_ref, &|_| ()).await?;
|
|
on_step("nixos-container stop");
|
|
priv_run("stop", name).await?;
|
|
}
|
|
on_step("nixos-container update");
|
|
let update_result = priv_run_inner("update", name, Some(on_build_log_id)).await;
|
|
if let Err(ref update_err) = update_result {
|
|
// The update failed (e.g. nix build error). If the agent was
|
|
// running before we stopped it, try to bring it back up on the
|
|
// previous successful configuration so it doesn't stay dead.
|
|
// The start failure is logged but not promoted to an error —
|
|
// we always propagate the original update error (below).
|
|
if was_running {
|
|
tracing::warn!(
|
|
%name,
|
|
error = %update_err,
|
|
"nixos-container update failed; attempting restart on old config"
|
|
);
|
|
on_step("nixos-container start (recovery)");
|
|
if let Err(e) = priv_run("start", name).await {
|
|
tracing::warn!(%name, error = %e, "recovery start after failed update also failed");
|
|
}
|
|
}
|
|
}
|
|
update_result?;
|
|
if was_running {
|
|
if defer_start {
|
|
// The caller re-queues the start on the fast lane so the
|
|
// build lane is freed for the next entry instead of
|
|
// waiting out the container boot here.
|
|
return Ok(true);
|
|
}
|
|
on_step("nixos-container start");
|
|
start_with_fallback(name).await?;
|
|
}
|
|
Ok(false)
|
|
} else {
|
|
// Spawn path: create is atomic, no prebuild needed.
|
|
// See `docs/coordinator.md::Spawn path`.
|
|
on_step("nixos-container create");
|
|
priv_run("create", name).await?;
|
|
write_dropins(name, hive, paths).await?;
|
|
on_step("nixos-container start");
|
|
priv_run("start", name).await?;
|
|
Ok(false)
|
|
}
|
|
}
|
|
|
|
/// Pre-build `system.build.toplevel` against `meta#<name>` so the
|
|
/// subsequent `nixos-container update` finds the result cached and
|
|
/// skips straight to the profile-swap. Store-warming only — container
|
|
/// is untouched. See `docs/coordinator.md::Rebuild path` for why
|
|
/// the prebuild happens before stop, and `docs/coordinator.md::Prebuild
|
|
/// attr path` for why the explicit nixosConfigurations attr is required.
|
|
///
|
|
/// `on_build_log_id` fires with the `build_logs` row id as soon as the
|
|
/// row opens, so queue-side callers can link their node to the live
|
|
/// stream. Pass `&|_| ()` when not needed.
|
|
pub async fn prebuild_toplevel(
|
|
name: &str,
|
|
flake_ref: &str,
|
|
on_build_log_id: &(dyn Fn(i64) + Send + Sync),
|
|
) -> Result<()> {
|
|
use tokio::io::{AsyncBufReadExt, BufReader};
|
|
// Split `<root>#<name>` so we can re-emit with the explicit
|
|
// `nixosConfigurations.<name>` segment. The flake_ref shape is
|
|
// constructed by `rebuild_no_meta` and always contains exactly one
|
|
// `#`; `split_once` returning None here would be a programmer
|
|
// error we'd want to surface loudly rather than paper over.
|
|
let (flake_root, fragment) = flake_ref
|
|
.split_once('#')
|
|
.with_context(|| format!("flake_ref {flake_ref:?} missing '#<name>' fragment"))?;
|
|
// Sanity-check the fragment matches the agent name we were
|
|
// passed — guards against future calls that pass a divergent
|
|
// pair (no current callsite does, but the pair is redundant
|
|
// and worth checking once).
|
|
if fragment != name {
|
|
anyhow::bail!("prebuild_toplevel: flake_ref fragment '{fragment}' ≠ agent name '{name}'");
|
|
}
|
|
let attr = format!("{flake_root}#nixosConfigurations.{name}.config.system.build.toplevel");
|
|
let args = vec![
|
|
"--extra-experimental-features",
|
|
"nix-command flakes",
|
|
"build",
|
|
"--no-link",
|
|
"--print-out-paths",
|
|
&attr,
|
|
];
|
|
let cmdline = format!("nix {}", args.join(" "));
|
|
tracing::info!(%name, %cmdline, "prebuild: warming system toplevel");
|
|
|
|
// Open a build_logs row for this attempt (best-effort — None when
|
|
// the global handle hasn't been installed, e.g. early startup
|
|
// or standalone tests). Lines pumped from stdout/stderr append
|
|
// into the row; `finish` lands the terminal status before we bail.
|
|
let logs = crate::build_logs::global();
|
|
let log_id = logs.as_ref().and_then(|h| {
|
|
h.start(name, "prebuild", &cmdline)
|
|
.map_err(|e| {
|
|
tracing::warn!(error = ?e, "build_logs: start failed (prebuild log dropped)");
|
|
})
|
|
.ok()
|
|
});
|
|
if let Some(id) = log_id {
|
|
on_build_log_id(id);
|
|
}
|
|
|
|
let mut child = Command::new("nix")
|
|
.args(&args)
|
|
.stdout(std::process::Stdio::piped())
|
|
.stderr(std::process::Stdio::piped())
|
|
.spawn()
|
|
.with_context(|| format!("spawn {cmdline}"))?;
|
|
|
|
let stdout = child.stdout.take().expect("piped stdout");
|
|
let stderr = child.stderr.take().expect("piped stderr");
|
|
|
|
let stdout_cmdline = cmdline.clone();
|
|
let stdout_logs = logs.clone();
|
|
let pump_stdout = tokio::spawn(async move {
|
|
let mut lines = BufReader::new(stdout).lines();
|
|
while let Ok(Some(line)) = lines.next_line().await {
|
|
tracing::info!(target: "nix-prebuild", cmdline = %stdout_cmdline, "{line}");
|
|
if let (Some(h), Some(id)) = (&stdout_logs, log_id) {
|
|
h.append_stdout(id, &line);
|
|
}
|
|
}
|
|
});
|
|
|
|
let stderr_cmdline = cmdline.clone();
|
|
let stderr_logs = logs.clone();
|
|
let pump_stderr = tokio::spawn(async move {
|
|
let mut lines = BufReader::new(stderr).lines();
|
|
while let Ok(Some(line)) = lines.next_line().await {
|
|
tracing::warn!(target: "nix-prebuild", cmdline = %stderr_cmdline, "{line}");
|
|
if let (Some(h), Some(id)) = (&stderr_logs, log_id) {
|
|
h.append_stderr(id, &line);
|
|
}
|
|
}
|
|
});
|
|
|
|
let status = child
|
|
.wait()
|
|
.await
|
|
.with_context(|| format!("wait {cmdline}"))?;
|
|
let _ = pump_stdout.await;
|
|
let _ = pump_stderr.await;
|
|
|
|
let ok = status.success();
|
|
if let (Some(h), Some(id)) = (&logs, log_id) {
|
|
h.finish(
|
|
id,
|
|
if ok {
|
|
crate::build_logs::BuildStatus::Ok
|
|
} else {
|
|
crate::build_logs::BuildStatus::Fail
|
|
},
|
|
);
|
|
}
|
|
if !ok {
|
|
match log_id {
|
|
Some(id) => bail!("prebuild {cmdline} failed ({status}); see build log #{id}"),
|
|
None => bail!("prebuild {cmdline} failed ({status})"),
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
pub async fn list() -> Result<Vec<String>> {
|
|
let stdout = crate::priv_client::list_containers().await?;
|
|
Ok(stdout
|
|
.lines()
|
|
.map(str::trim)
|
|
.filter(|line| line.starts_with(AGENT_PREFIX))
|
|
.map(str::to_owned)
|
|
.collect())
|
|
}
|
|
|
|
/// Initialize the manager-editable proposed repo. Seeds two tracked
|
|
/// files: `agent.nix` (the module the manager edits) and `flake.nix`
|
|
/// (the boilerplate that lets the meta flake import this repo as an
|
|
/// input — meta locks at a specific sha and reads
|
|
/// `nixosModules.default`, so `flake.nix` must be in the commit). The
|
|
/// manager shouldn't edit `flake.nix` (the prompt says so) but it's
|
|
/// visible so they can introspect.
|
|
///
|
|
/// Touched by hive-c0re only on first spawn — never again — so the
|
|
/// manager can't be surprised by hive-c0re commits or working-tree
|
|
/// resets.
|
|
pub async fn setup_proposed(proposed_dir: &Path, name: &str) -> Result<()> {
|
|
let fresh = !proposed_dir.join(".git").exists();
|
|
if fresh {
|
|
std::fs::create_dir_all(proposed_dir)
|
|
.with_context(|| format!("create {}", proposed_dir.display()))?;
|
|
let agent_path = proposed_dir.join("agent.nix");
|
|
if !agent_path.exists() {
|
|
std::fs::write(&agent_path, initial_agent_nix(name))
|
|
.with_context(|| format!("write {}", agent_path.display()))?;
|
|
}
|
|
let flake_path = proposed_dir.join("flake.nix");
|
|
if !flake_path.exists() {
|
|
std::fs::write(&flake_path, initial_flake_nix())
|
|
.with_context(|| format!("write {}", flake_path.display()))?;
|
|
}
|
|
git(proposed_dir, &["init", "--initial-branch=main"]).await?;
|
|
git(proposed_dir, &["add", "agent.nix", "flake.nix"]).await?;
|
|
git_commit(proposed_dir, "hive-c0re init").await?;
|
|
}
|
|
// Idempotently wire the `applied` remote — purely for the
|
|
// manager's ergonomics. The URL is the path inside the manager
|
|
// container (`/applied/<n>/.git`), where the RO bind in
|
|
// `set_nspawn_flags` makes it real. hive-c0re itself never
|
|
// dereferences this remote; the host-side fetch in
|
|
// `request_apply_commit` uses absolute host paths.
|
|
ensure_applied_remote(proposed_dir, name).await
|
|
}
|
|
|
|
async fn ensure_applied_remote(proposed_dir: &Path, name: &str) -> Result<()> {
|
|
let want = format!("/applied/{name}/.git");
|
|
let existing = git_command()
|
|
.current_dir(proposed_dir)
|
|
.args(["remote", "get-url", "applied"])
|
|
.output()
|
|
.await
|
|
.with_context(|| format!("git remote get-url applied in {}", proposed_dir.display()))?;
|
|
if existing.status.success() {
|
|
let current = String::from_utf8_lossy(&existing.stdout).trim().to_owned();
|
|
if current == want {
|
|
return Ok(());
|
|
}
|
|
// URL drifted (path scheme changed, etc.) — re-point it.
|
|
return git(proposed_dir, &["remote", "set-url", "applied", &want]).await;
|
|
}
|
|
git(proposed_dir, &["remote", "add", "applied", &want]).await
|
|
}
|
|
|
|
/// Set up the applied repo. First-spawn only: init the repo, pull
|
|
/// proposed's initial commit in via `git fetch`, tag it `deployed/0`.
|
|
/// This is the *only* time hive-c0re reads from `proposed` for an
|
|
/// agent — subsequent proposals are fetched at `request_apply_commit`
|
|
/// time and tagged `proposal/<id>` (see `actions::approve` for the
|
|
/// tag state machine).
|
|
///
|
|
/// `proposed_dir` is `None` on rebuild paths where the repo already
|
|
/// exists — we just verify it's the right shape and bail otherwise.
|
|
/// Unlike the pre-overhaul code path, `flake.nix` is no longer
|
|
/// regenerated at the host level: it's tracked in proposed (seeded by
|
|
/// `setup_proposed`) and rides along on every fetch.
|
|
pub async fn setup_applied(
|
|
applied_dir: &Path,
|
|
proposed_dir: Option<&Path>,
|
|
name: &str,
|
|
) -> Result<()> {
|
|
std::fs::create_dir_all(applied_dir)
|
|
.with_context(|| format!("create {}", applied_dir.display()))?;
|
|
|
|
if !applied_dir.join(".git").exists() {
|
|
let Some(proposed) = proposed_dir else {
|
|
bail!(
|
|
"applied repo at {} is missing its .git directory; \
|
|
cannot rebuild without a proposed source to seed from. \
|
|
destroy --purge and re-spawn this agent.",
|
|
applied_dir.display()
|
|
);
|
|
};
|
|
git(applied_dir, &["init", "--initial-branch=main"]).await?;
|
|
let proposed_str = proposed.display().to_string();
|
|
// Seed the applied repo at the root (template) commit of proposed,
|
|
// not at `main`. This ensures `deployed/0` is the template baseline
|
|
// so the first ApplyCommit diff shows the manager's real changes
|
|
// rather than an empty diff (which happens when the manager has
|
|
// already committed their config and proposed/main == proposal/<id>).
|
|
let root_sha = git_root_commit(proposed).await?;
|
|
git(
|
|
applied_dir,
|
|
// --update-head-ok lets us fetch into refs/heads/main while
|
|
// HEAD still points there. git's default safeguard refuses
|
|
// to avoid index/working-tree desync, but the working tree
|
|
// is empty (we just `init`'d) and we read-tree-reset right
|
|
// after, so the safeguard is moot here.
|
|
&[
|
|
"fetch",
|
|
"--no-tags",
|
|
"--update-head-ok",
|
|
&proposed_str,
|
|
&format!("{root_sha}:refs/heads/main"),
|
|
],
|
|
)
|
|
.await?;
|
|
git_read_tree_reset(applied_dir, "refs/heads/main").await?;
|
|
git_tag(applied_dir, "deployed/0", "refs/heads/main").await?;
|
|
} else if git_rev_parse(applied_dir, "refs/tags/deployed/0")
|
|
.await
|
|
.is_err()
|
|
{
|
|
// Pre-overhaul applied repo — no deployed/* tag scheme,
|
|
// flake.nix may be untracked, agent.nix possibly authored by
|
|
// hive-c0re directly. The startup auto-migration fixes this
|
|
// in place; if it didn't run (or got skipped), surface a
|
|
// clear error.
|
|
bail!(
|
|
"applied repo at {} predates the meta-flake layout. \
|
|
Restart hive-c0re to let the auto-migration run, or \
|
|
destroy --purge {name} and re-spawn.",
|
|
applied_dir.display()
|
|
);
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Create the per-agent Claude credentials dir if missing. Mode 0755 — hive-core
|
|
/// needs read+execute to list the directory so `claude_has_session` can detect a
|
|
/// valid session; credential files inside (`.credentials.json` etc.) are 0600 so
|
|
/// secrets stay private regardless of the directory mode. Idempotent: existing
|
|
/// dirs are left untouched (an agent's OAuth tokens survive `destroy`/recreate).
|
|
/// Public for the `InitConfig` approval path in `actions.rs` which seeds
|
|
/// dirs without calling the full `spawn`.
|
|
pub fn ensure_claude_dir(claude_dir: &Path) -> Result<()> {
|
|
use std::io;
|
|
if !claude_dir.exists() {
|
|
std::fs::create_dir_all(claude_dir)
|
|
.with_context(|| format!("create {}", claude_dir.display()))?;
|
|
}
|
|
// 0755: hive-core (different user from the agent) needs read+execute to
|
|
// list the directory so `claude_has_session` can detect a valid session.
|
|
// The credential files inside (`.credentials.json` etc.) are 0600 so the
|
|
// secrets themselves stay private regardless of the directory mode.
|
|
//
|
|
// Best-effort: on the first container boot, `hive-agent-user-migrate`
|
|
// chowns this dir to the agent user. After that, hive-core (a different
|
|
// user) cannot chmod it (EPERM) — that's fine because the mode set during
|
|
// initial creation (0755) is preserved through the chown. Any other error
|
|
// (ENOENT, I/O error) is unexpected and propagated.
|
|
#[cfg(unix)]
|
|
{
|
|
use std::os::unix::fs::PermissionsExt;
|
|
match std::fs::set_permissions(claude_dir, std::fs::Permissions::from_mode(0o755)) {
|
|
Ok(()) => {}
|
|
Err(e) if e.kind() == io::ErrorKind::PermissionDenied => {
|
|
tracing::debug!(
|
|
path = %claude_dir.display(),
|
|
"ensure_claude_dir: chmod 755 skipped (dir likely owned by agent user after migration)"
|
|
);
|
|
}
|
|
Err(e) => {
|
|
return Err(e).with_context(|| format!("chmod 755 {}", claude_dir.display()));
|
|
}
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Public for the `InitConfig` approval path in `actions.rs` which seeds
|
|
/// dirs without calling the full `spawn`. Also creates the sibling `harness/`
|
|
/// dir so the first harness startup can write its sqlite files immediately.
|
|
pub fn ensure_state_dir(notes_dir: &Path) -> Result<()> {
|
|
if !notes_dir.exists() {
|
|
std::fs::create_dir_all(notes_dir)
|
|
.with_context(|| format!("create {}", notes_dir.display()))?;
|
|
}
|
|
// Harness dir is a sibling of the agent-visible state dir.
|
|
if let Some(parent) = notes_dir.parent() {
|
|
let harness_dir = parent.join("harness");
|
|
if !harness_dir.exists() {
|
|
std::fs::create_dir_all(&harness_dir)
|
|
.with_context(|| format!("create {}", harness_dir.display()))?;
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Ensure agent `name`'s persistent state root
|
|
/// (`/var/lib/hyperhive/agents/<name>`) is a btrfs subvolume — when the host
|
|
/// filesystem supports it — BEFORE the per-agent subdirs (`state/`, `claude/`,
|
|
/// `harness/`) are created by `ensure_state_dir` / `ensure_claude_dir`.
|
|
///
|
|
/// Progressive enhancement: if the root already exists
|
|
/// (any agent provisioned before this landed, plain dir or subvol) it's left
|
|
/// exactly as-is — no auto-migration — and the priv round-trip is skipped. On
|
|
/// a non-btrfs host the priv op no-ops and the root is later created as a
|
|
/// plain dir by `ensure_*_dir`, identical to the old behaviour. Only a
|
|
/// brand-new agent on a btrfs host gets a real subvolume. Subvolume creation
|
|
/// is privileged, so it's delegated to hive-priv.
|
|
pub async fn ensure_agent_state_subvolume(name: &str) -> Result<()> {
|
|
let root = Path::new(HOST_AGENTS_ROOT).join(name);
|
|
if root.exists() {
|
|
return Ok(());
|
|
}
|
|
crate::priv_client::ensure_agent_subvolume(name)
|
|
.await
|
|
.with_context(|| format!("ensure btrfs subvolume for agent {name}"))
|
|
}
|
|
|
|
fn initial_agent_nix(name: &str) -> String {
|
|
format!(
|
|
"{{ config, pkgs, lib, ... }}:\n{{\n # Per-agent overrides for {name}. This is a regular NixOS module\n # — add packages, services, modules, imports as needed.\n #\n # imports = [ ./extra-module.nix ];\n # environment.systemPackages = with pkgs; [ ];\n}}\n",
|
|
)
|
|
}
|
|
|
|
/// Module-only flake exposed by every agent's repo. Consumed by the
|
|
/// hive-c0re-owned meta flake at `/var/lib/hyperhive/meta/` as a flake
|
|
/// input. The wrapper is intentionally permissive:
|
|
///
|
|
/// - Manager edits `inputs.* = …` to add other flakes (e.g. an MCP
|
|
/// server's own flake) — the lock for those lands in the agent's
|
|
/// own `flake.lock` and rolls up into meta's lock transitively.
|
|
/// - The outputs block forwards every input (minus `self`) into
|
|
/// `agent.nix` as the `flakeInputs` module argument, so the
|
|
/// manager just references `flakeInputs.<name>.packages.${pkgs.system}.default`
|
|
/// without further plumbing.
|
|
///
|
|
/// Identity injection (`HIVE_PORT` / `HIVE_LABEL` / dashboard port /
|
|
/// git committer) still lives in the meta flake's wrapper.
|
|
pub fn initial_flake_nix() -> &'static str {
|
|
"{\n description = \"hyperhive agent\";\n inputs = { };\n outputs =\n { self, ... }@inputs:\n {\n nixosModules.default = {\n imports = [ ./agent.nix ];\n _module.args.flakeInputs = builtins.removeAttrs inputs [ \"self\" ];\n };\n };\n}\n"
|
|
}
|
|
|
|
/// Return the SHA of the root (oldest, no-parent) commit in a repo.
|
|
/// Used to seed the applied repo at the template baseline rather than at
|
|
/// `main`, so the first `ApplyCommit` diff shows the manager's real changes.
|
|
async fn git_root_commit(dir: &Path) -> Result<String> {
|
|
let out = git_command()
|
|
.current_dir(dir)
|
|
.args(["rev-list", "--max-parents=0", "HEAD"])
|
|
.output()
|
|
.await
|
|
.with_context(|| format!("git rev-list --max-parents=0 HEAD in {}", dir.display()))?;
|
|
if !out.status.success() {
|
|
anyhow::bail!(
|
|
"git rev-list --max-parents=0 failed: {}",
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
Ok(String::from_utf8_lossy(&out.stdout).trim().to_owned())
|
|
}
|
|
|
|
async fn git_commit(dir: &Path, message: &str) -> Result<()> {
|
|
git(
|
|
dir,
|
|
&[
|
|
"-c",
|
|
&format!("user.name={GIT_NAME}"),
|
|
"-c",
|
|
&format!("user.email={GIT_EMAIL}"),
|
|
"commit",
|
|
"-m",
|
|
message,
|
|
],
|
|
)
|
|
.await
|
|
}
|
|
|
|
/// Spawn `git` honoring the `HYPERHIVE_GIT` env var (absolute path baked in
|
|
/// by the NixOS module), falling back to bare `git` (PATH lookup) otherwise.
|
|
#[must_use]
|
|
pub fn git_command() -> Command {
|
|
let exe = std::env::var("HYPERHIVE_GIT").unwrap_or_else(|_| "git".into());
|
|
Command::new(exe)
|
|
}
|
|
|
|
pub async fn git(dir: &Path, args: &[&str]) -> Result<()> {
|
|
let out = git_command()
|
|
.current_dir(dir)
|
|
.args(args)
|
|
.output()
|
|
.await
|
|
.with_context(|| format!("git {} in {}", args.join(" "), dir.display()))?;
|
|
if !out.status.success() {
|
|
bail!(
|
|
"git {} failed ({}): {}",
|
|
args.join(" "),
|
|
out.status,
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Fetch the commit `sha` from the `src` git repo into `dst` and pin
|
|
/// it as `refs/tags/<tag>`. Used at `request_apply_commit` time so
|
|
/// hive-c0re captures an immutable handle on the manager's commit;
|
|
/// subsequent amendments / force-pushes in `src` no longer affect
|
|
/// what gets built. Returns the resolved full sha.
|
|
///
|
|
/// `sha` must be a commit sha (short or full) — the caller
|
|
/// (`submit_apply_commit`) shape-checks it first. We resolve it
|
|
/// LOCALLY against `src` rather than asking the remote to resolve
|
|
/// it: `git fetch <remote> <sha>:<dst>` treats the left side as a
|
|
/// remote *ref name*, and a bare sha is not one ("couldn't find
|
|
/// remote ref ..."). Fetching by sha would need a full 40-hex sha
|
|
/// plus `uploadpack.allow*SHA1InWant` on the remote, which the
|
|
/// proposed repos don't set. hive-c0re has direct read access to
|
|
/// `src`, so a local `rev-parse` + a branch-glob fetch sidesteps
|
|
/// the whole sha-want negotiation.
|
|
pub async fn git_fetch_to_tag(dst: &Path, src: &Path, sha: &str, tag: &str) -> Result<String> {
|
|
let src_str = src.display().to_string();
|
|
// Resolve the (short-or-full) sha to a full sha against the
|
|
// source repo. The `^{commit}` peel + non-zero exit on a missing
|
|
// object means a typo'd / stale sha fails loudly right here.
|
|
let full = git_rev_parse(src, &format!("{sha}^{{commit}}"))
|
|
.await
|
|
.with_context(|| format!("commit '{sha}' not found in proposed repo {src_str}"))?;
|
|
// Bring src's objects into dst. Fetching every head pulls the
|
|
// wanted commit's history (always reachable from a branch in the
|
|
// manager's flow) into dst's object db without sha-want.
|
|
git(
|
|
dst,
|
|
&[
|
|
"fetch",
|
|
"--no-tags",
|
|
&src_str,
|
|
"+refs/heads/*:refs/remotes/proposal-src/*",
|
|
],
|
|
)
|
|
.await?;
|
|
// Pin the exact commit as the proposal tag. The objects are now
|
|
// local so this resolves without touching the remote.
|
|
git(dst, &["tag", tag, &full]).await.with_context(|| {
|
|
format!("tag {tag} at {full}: commit not reachable from any branch in proposed repo")
|
|
})?;
|
|
Ok(full)
|
|
}
|
|
|
|
/// Resolve `refname` (a tag, branch, or sha) in `dir` to its full sha.
|
|
pub async fn git_rev_parse(dir: &Path, refname: &str) -> Result<String> {
|
|
let out = git_command()
|
|
.current_dir(dir)
|
|
.args(["rev-parse", refname])
|
|
.output()
|
|
.await
|
|
.with_context(|| format!("git rev-parse {refname} in {}", dir.display()))?;
|
|
if !out.status.success() {
|
|
bail!(
|
|
"git rev-parse {refname} failed ({}): {}",
|
|
out.status,
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
Ok(String::from_utf8_lossy(&out.stdout).trim().to_owned())
|
|
}
|
|
|
|
/// Plant a lightweight tag at `target`. Errors if the tag already
|
|
/// exists — we want loud failures on id reuse, not silent
|
|
/// overwrites.
|
|
pub async fn git_tag(dir: &Path, name: &str, target: &str) -> Result<()> {
|
|
git(dir, &["tag", name, target]).await
|
|
}
|
|
|
|
/// Plant an annotated tag with `body` as the message. Used for
|
|
/// `failed/<id>` (body = build error) and `denied/<id>` (body =
|
|
/// operator note). Multi-line bodies handled via stdin so we don't
|
|
/// have to escape anything.
|
|
pub async fn git_tag_annotated(dir: &Path, name: &str, target: &str, body: &str) -> Result<()> {
|
|
use tokio::io::AsyncWriteExt;
|
|
// Annotated tags are git objects, so they need a tagger identity
|
|
// (same constraint as a commit). Pass the hive-c0re identity
|
|
// inline rather than relying on a global git config — applied
|
|
// repos are hive-c0re-owned and the host's user might not have
|
|
// user.email set.
|
|
let mut child = git_command()
|
|
.current_dir(dir)
|
|
.args([
|
|
"-c",
|
|
&format!("user.name={GIT_NAME}"),
|
|
"-c",
|
|
&format!("user.email={GIT_EMAIL}"),
|
|
"tag",
|
|
"-a",
|
|
name,
|
|
target,
|
|
"-F",
|
|
"-",
|
|
])
|
|
.stdin(std::process::Stdio::piped())
|
|
.stdout(std::process::Stdio::piped())
|
|
.stderr(std::process::Stdio::piped())
|
|
.spawn()
|
|
.with_context(|| format!("spawn git tag -a {name} in {}", dir.display()))?;
|
|
if let Some(mut stdin) = child.stdin.take() {
|
|
stdin
|
|
.write_all(body.as_bytes())
|
|
.await
|
|
.context("write tag body to git stdin")?;
|
|
// Drop closes stdin so git can finish reading.
|
|
drop(stdin);
|
|
}
|
|
let out = child.wait_with_output().await.context("wait git tag -a")?;
|
|
if !out.status.success() {
|
|
bail!(
|
|
"git tag -a {name} failed ({}): {}",
|
|
out.status,
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Replace working tree + index with the tree at `target` without
|
|
/// moving HEAD. `applied/main` stays pointing at the last known-good
|
|
/// `deployed/*` while we let `nixos-container update` evaluate the
|
|
/// candidate. On build failure callers reset back to HEAD; on
|
|
/// success they fast-forward main to `target`.
|
|
pub async fn git_read_tree_reset(dir: &Path, target: &str) -> Result<()> {
|
|
git(dir, &["read-tree", "--reset", "-u", target]).await
|
|
}
|
|
|
|
/// Hard-set a ref to `target`. Used to fast-forward `refs/heads/main`
|
|
/// to the just-deployed proposal commit. Uses `update-ref`, not
|
|
/// `branch -f`, so it works regardless of where HEAD currently sits.
|
|
pub async fn git_update_ref(dir: &Path, refname: &str, target: &str) -> Result<()> {
|
|
git(dir, &["update-ref", refname, target]).await
|
|
}
|
|
|
|
/// Write a systemd drop-in for `container@<container>.service` that applies
|
|
/// our default resource caps. Goes under `/run/systemd/system/...` so it's
|
|
/// ephemeral (regenerated on every spawn / rebuild).
|
|
async fn set_resource_limits(container: &str, cpu_quota: &str, memory_max: &str) -> Result<()> {
|
|
crate::priv_client::write_resource_limits(container, memory_max, cpu_quota).await
|
|
}
|
|
|
|
async fn systemd_daemon_reload() -> Result<()> {
|
|
crate::priv_client::daemon_reload().await
|
|
}
|
|
|
|
/// Idempotently rewrite the lines in `/etc/nixos-containers/<container>.conf`
|
|
/// that hive-c0re owns: `PRIVATE_NETWORK` (forced 0 so the agent's web UI port
|
|
/// is reachable on the host) and `EXTRA_NSPAWN_FLAGS` (the runtime-dir bind).
|
|
/// The start script expands `$EXTRA_NSPAWN_FLAGS` unquoted into the
|
|
/// `systemd-nspawn` command.
|
|
/// Where in the container's filesystem the manager sees its agents tree.
|
|
/// Matches the `/agents` path that pre-Phase-8 hosts declared via
|
|
/// `containers.root.bindMounts."/agents"`.
|
|
pub const CONTAINER_MANAGER_AGENTS_MOUNT: &str = "/agents";
|
|
|
|
/// Where the manager sees the applied trees of every agent, read-only.
|
|
/// Manager runs `git fetch /applied/<n>/.git refs/tags/*:refs/tags/applied/*`
|
|
/// to learn what hive-c0re deployed (or rejected, or failed to
|
|
/// build); the RO bind makes accidental writes impossible from
|
|
/// inside the container.
|
|
pub const CONTAINER_MANAGER_APPLIED_MOUNT: &str = "/applied";
|
|
|
|
/// The on-host root that gets bind-mounted to `/agents` inside the manager.
|
|
/// Hard-coded to match `AGENT_STATE_ROOT` in coordinator.rs (kept duplicated
|
|
/// here so lifecycle stays usable as a leaf module).
|
|
const HOST_AGENTS_ROOT: &str = "/var/lib/hyperhive/agents";
|
|
|
|
/// On-host applied repo root, mirrored RO into the manager. Matches
|
|
/// `APPLIED_STATE_ROOT` in coordinator.rs.
|
|
const HOST_APPLIED_ROOT: &str = "/var/lib/hyperhive/applied";
|
|
|
|
/// On-host meta repo root, mirrored RO into the manager. Matches
|
|
/// `meta::meta_dir()` but duplicated here so lifecycle stays a leaf.
|
|
const HOST_META_ROOT: &str = "/var/lib/hyperhive/meta";
|
|
|
|
/// Shared directory accessible to all agents. All agents bind-mount this RW.
|
|
const HOST_SHARED_ROOT: &str = "/var/lib/hyperhive/shared";
|
|
|
|
/// Append bind flags for `child`'s state, harness, and config dirs into
|
|
/// `binds`, all read-write. The RW on `state` is deliberate (recovery),
|
|
/// not an oversight; see docs/persistence.md ("Parent access to child
|
|
/// state") for the rationale. Creates missing host-side directories so
|
|
/// nspawn doesn't refuse to start; missing dirs are non-fatal.
|
|
fn bind_child_agent_dirs(child: &str, binds: &mut Vec<BindMount>) {
|
|
let state_dir = format!("{HOST_AGENTS_ROOT}/{child}/state");
|
|
let harness_dir = format!("{HOST_AGENTS_ROOT}/{child}/harness");
|
|
let config_dir = format!("{HOST_AGENTS_ROOT}/{child}/config");
|
|
for dir in [&state_dir, &harness_dir, &config_dir] {
|
|
let _ = std::fs::create_dir_all(dir);
|
|
}
|
|
binds.push(BindMount {
|
|
host_path: state_dir,
|
|
container_path: format!("/agents/{child}/state"),
|
|
read_only: false,
|
|
});
|
|
binds.push(BindMount {
|
|
host_path: harness_dir,
|
|
container_path: format!("/agents/{child}/harness"),
|
|
read_only: false,
|
|
});
|
|
binds.push(BindMount {
|
|
host_path: config_dir,
|
|
container_path: format!("/agents/{child}/config"),
|
|
read_only: false,
|
|
});
|
|
}
|
|
|
|
/// Hive-wide secrets forwarded into every agent container via nspawn
|
|
/// `--load-credential=<name>:<host_path>`. Currently just the OTEL
|
|
/// auth-header secret, when `services.hyperhive.otel.headersCredential`
|
|
/// is set (surfaced as `HYPERHIVE_OTEL_HEADERS_CREDENTIAL` on hive-c0re's
|
|
/// unit env — the same host option meta.rs reads to inject
|
|
/// `hyperhive.otel.headersCredential`). The inner harness unit reads it
|
|
/// via `LoadCredential=otel-headers` (inherit). The secret never lands in
|
|
/// a bind mount, the nix store, or the generated config.
|
|
///
|
|
/// A configured-but-missing file is skipped with a warning rather than
|
|
/// forwarded (nspawn would refuse to start the container otherwise): a
|
|
/// host-level secret typo shouldn't take down every agent's start; OTEL
|
|
/// just exports without the auth header until the file appears.
|
|
fn hive_load_credentials() -> Vec<CredentialMount> {
|
|
let mut out = Vec::new();
|
|
let Ok(path) = std::env::var("HYPERHIVE_OTEL_HEADERS_CREDENTIAL") else {
|
|
return out;
|
|
};
|
|
if path.is_empty() {
|
|
return out;
|
|
}
|
|
if std::path::Path::new(&path).is_file() {
|
|
out.push(CredentialMount {
|
|
name: "otel-headers".to_owned(),
|
|
host_path: path,
|
|
});
|
|
} else {
|
|
tracing::warn!(
|
|
%path,
|
|
"HYPERHIVE_OTEL_HEADERS_CREDENTIAL is set but the file is missing; \
|
|
skipping --load-credential (OTEL will export without the auth header)"
|
|
);
|
|
}
|
|
out
|
|
}
|
|
|
|
#[allow(
|
|
clippy::too_many_lines,
|
|
reason = "one contiguous nspawn-flag assembly block; the length is the flag \
|
|
surface itself, splitting it would just hide the shape"
|
|
)]
|
|
async fn set_nspawn_flags(
|
|
container: &str,
|
|
runtime_dir: &Path,
|
|
claude_dir: &Path,
|
|
notes_dir: &Path,
|
|
) -> Result<()> {
|
|
// Ensure /shared directory exists before binding. systemd-nspawn requires the bind source to exist.
|
|
std::fs::create_dir_all(HOST_SHARED_ROOT)
|
|
.with_context(|| format!("create {HOST_SHARED_ROOT}"))?;
|
|
// Make /shared writable by every agent. Containers share host uids (no
|
|
// PrivateUsers), but each agent is a distinct unix user, so a root-owned
|
|
// 0755 dir leaves them unable to write — the documented "read/write for
|
|
// all agents" contract was broken. A setgid group would need a
|
|
// pinned GID declared in every container plus all agent users joined to
|
|
// it (cross-container coordination + a rebuild cascade); instead we use
|
|
// the /tmp model — sticky world-writable (1777). The sticky bit lets any
|
|
// agent create files while protecting each agent's entries from deletion
|
|
// by the others, and matches /shared's documented "free-for-all, may be
|
|
// deleted/lost" semantics without touching any per-agent config.
|
|
{
|
|
use std::os::unix::fs::PermissionsExt as _;
|
|
let perms = std::fs::Permissions::from_mode(0o1777);
|
|
std::fs::set_permissions(HOST_SHARED_ROOT, perms)
|
|
.with_context(|| format!("chmod 1777 {HOST_SHARED_ROOT}"))?;
|
|
}
|
|
// Ensure /knowledge dir exists. It may be empty until forge seeds it;
|
|
// nspawn refuses to start if the bind source is missing entirely.
|
|
std::fs::create_dir_all(crate::knowledge::LOCAL_DIR)
|
|
.with_context(|| format!("create {}", crate::knowledge::LOCAL_DIR))?;
|
|
|
|
// Logical agent name — strip the `h-` prefix.
|
|
// For the manager: `h-ruth` → `ruth`. For sub-agents: `h-iris` → `iris`.
|
|
let agent_name = container.strip_prefix(AGENT_PREFIX).unwrap_or(container);
|
|
|
|
// Claude credentials land at `/home/<agent>/.claude` so the
|
|
// `claude` CLI (which reads `$HOME/.claude`) finds them. The
|
|
// harness service's environment sets `HOME` to the same path
|
|
// (`agent-base.nix` / `manager.nix`), so no `--setenv` plumbing
|
|
// is needed here — the bind alone is enough.
|
|
let claude_mount = container_claude_mount(agent_name);
|
|
|
|
// Hive-wide secrets forwarded into the container's credential store
|
|
// (currently just the OTEL auth-header). Same for every agent.
|
|
let load_creds = hive_load_credentials();
|
|
|
|
let mut binds: Vec<BindMount> = vec![
|
|
BindMount {
|
|
host_path: runtime_dir.to_string_lossy().into_owned(),
|
|
container_path: CONTAINER_RUNTIME_MOUNT.to_owned(),
|
|
read_only: false,
|
|
},
|
|
BindMount {
|
|
host_path: claude_dir.to_string_lossy().into_owned(),
|
|
container_path: claude_mount,
|
|
read_only: false,
|
|
},
|
|
BindMount {
|
|
host_path: HOST_SHARED_ROOT.to_owned(),
|
|
container_path: CONTAINER_SHARED_MOUNT.to_owned(),
|
|
read_only: false,
|
|
},
|
|
BindMount {
|
|
host_path: crate::knowledge::LOCAL_DIR.to_owned(),
|
|
container_path: crate::knowledge::CONTAINER_MOUNT.to_owned(),
|
|
read_only: true,
|
|
},
|
|
];
|
|
|
|
// Own state, harness, and config dirs — same for every agent including
|
|
// the manager. Config is RO: an agent must not edit its own config; changes
|
|
// only ever flow through the approval queue.
|
|
binds.push(BindMount {
|
|
host_path: notes_dir.to_string_lossy().into_owned(),
|
|
container_path: format!("/agents/{agent_name}/state"),
|
|
read_only: false,
|
|
});
|
|
if let Some(state_parent) = notes_dir.parent() {
|
|
let harness_dir = state_parent.join("harness");
|
|
if !harness_dir.exists() {
|
|
let _ = std::fs::create_dir_all(&harness_dir);
|
|
}
|
|
binds.push(BindMount {
|
|
host_path: harness_dir.to_string_lossy().into_owned(),
|
|
container_path: format!("/agents/{agent_name}/harness"),
|
|
read_only: false,
|
|
});
|
|
}
|
|
let own_config = format!("{HOST_AGENTS_ROOT}/{agent_name}/config");
|
|
std::fs::create_dir_all(&own_config).with_context(|| format!("create {own_config}"))?;
|
|
binds.push(BindMount {
|
|
host_path: own_config,
|
|
container_path: format!("/agents/{agent_name}/config"),
|
|
read_only: true,
|
|
});
|
|
|
|
// Topology-driven child mounts: every direct child of this agent gets
|
|
// its state, harness, and config dirs bind-mounted RW (parent reads +
|
|
// writes child state for recovery, and manages config). See
|
|
// `bind_child_agent_dirs`.
|
|
let direct_children = crate::topology::children_of(agent_name);
|
|
for child in &direct_children {
|
|
bind_child_agent_dirs(child, &mut binds);
|
|
}
|
|
|
|
// `can_manage_top_level_agents` role: additionally mount every
|
|
// parentless agent in the topology as a virtual child. Enables
|
|
// recovery — a role holder can update those agents' configs even
|
|
// when they are down. Also grants RO access to /applied and /meta.
|
|
if crate::topology::has_role(
|
|
agent_name,
|
|
crate::topology::ROLE_CAN_MANAGE_TOP_LEVEL_AGENTS,
|
|
) {
|
|
let top_level = crate::topology::top_level_agents();
|
|
for tl in &top_level {
|
|
if !direct_children.contains(tl) {
|
|
bind_child_agent_dirs(tl, &mut binds);
|
|
}
|
|
}
|
|
// systemd-nspawn refuses to start a container whose bind
|
|
// source doesn't exist. The meta repo is created by the
|
|
// startup migration, but make sure the directory is there
|
|
// before the role holder comes up in case set_nspawn_flags
|
|
// fires first (e.g. cold start with no agents).
|
|
std::fs::create_dir_all(HOST_META_ROOT)
|
|
.with_context(|| format!("create {HOST_META_ROOT}"))?;
|
|
binds.push(BindMount {
|
|
host_path: HOST_APPLIED_ROOT.to_owned(),
|
|
container_path: CONTAINER_MANAGER_APPLIED_MOUNT.to_owned(),
|
|
read_only: true,
|
|
});
|
|
binds.push(BindMount {
|
|
host_path: HOST_META_ROOT.to_owned(),
|
|
container_path: crate::meta::CONTAINER_MANAGER_META_MOUNT.to_owned(),
|
|
read_only: true,
|
|
});
|
|
}
|
|
|
|
// Web-socket subdir: bind-mount `/run/hive-agent/<name>/` into the
|
|
// container so the harness can bind `web.sock` there and the host-side
|
|
// gateway sees it. Subdir bind (not socket file) keeps the inode
|
|
// visible after the harness unlinks a stale socket on rebind.
|
|
// Applies to manager and sub-agents alike.
|
|
let socket_dir = crate::agent_sockets::agent_dir_for(agent_name);
|
|
std::fs::create_dir_all(&socket_dir)
|
|
.with_context(|| format!("create {}", socket_dir.display()))?;
|
|
// Chown to the agent user so the non-root harness can bind(2) here.
|
|
// Falls back to 0777 on first spawn when uid lookup returns None
|
|
// (container /etc/passwd not yet rendered).
|
|
if let Some((uid, gid)) = agent_uid_gid(agent_name) {
|
|
if let Err(e) = crate::priv_client::chown_socket_dir(agent_name, uid, gid).await {
|
|
tracing::warn!(%agent_name, error = ?e, "chown socket dir failed");
|
|
}
|
|
} else if let Err(e) = crate::priv_client::chmod_socket_dir(agent_name, 0o777).await {
|
|
tracing::warn!(%agent_name, error = ?e, "chmod socket dir failed");
|
|
}
|
|
binds.push(BindMount {
|
|
host_path: socket_dir.to_string_lossy().into_owned(),
|
|
container_path: socket_dir.to_string_lossy().into_owned(),
|
|
read_only: false,
|
|
});
|
|
|
|
// Network isolation: when HIVE_NETWORK_ISOLATION=1 is set (by the
|
|
// hive-network.nix module's `isolateContainers` option), flip the
|
|
// container to a private network namespace with a veth pair attached
|
|
// to the host bridge. Applies to all containers including the manager
|
|
// (all hive-c0re<->agent comms go through bind-mounted UDS, not TCP).
|
|
let isolation = {
|
|
let isolate = std::env::var("HIVE_NETWORK_ISOLATION").ok().as_deref() == Some("1");
|
|
let bridge = std::env::var("HIVE_NETWORK_BRIDGE").unwrap_or_default();
|
|
let subnet = std::env::var("HIVE_NETWORK_SUBNET").unwrap_or_default();
|
|
if isolate && !bridge.is_empty() && !subnet.is_empty() {
|
|
let Some(agent_ip) = agent_network_ip(agent_name, &subnet) else {
|
|
tracing::warn!(
|
|
%agent_name, %subnet,
|
|
"HIVE_NETWORK_SUBNET is set but could not derive a valid IP for agent \
|
|
(bad CIDR? prefix too narrow?); skipping PRIVATE_NETWORK write to \
|
|
avoid misconfigured isolation"
|
|
);
|
|
return crate::priv_client::write_nspawn_flags(
|
|
container,
|
|
&binds,
|
|
None,
|
|
&load_creds,
|
|
)
|
|
.await;
|
|
};
|
|
let Some(gateway_ip) = bridge_gateway_ip(&subnet) else {
|
|
tracing::warn!(
|
|
%agent_name, %subnet,
|
|
"HIVE_NETWORK_SUBNET is set but the bridge gateway IP is unparseable; \
|
|
skipping PRIVATE_NETWORK write to avoid an isolated container with no \
|
|
default route or resolver"
|
|
);
|
|
return crate::priv_client::write_nspawn_flags(
|
|
container,
|
|
&binds,
|
|
None,
|
|
&load_creds,
|
|
)
|
|
.await;
|
|
};
|
|
tracing::info!(
|
|
%agent_name, %agent_ip, %gateway_ip, %bridge,
|
|
"network isolation: PRIVATE_NETWORK=1"
|
|
);
|
|
Some(hive_sh4re::priv_proto::NetworkIsolation {
|
|
agent_ip,
|
|
bridge,
|
|
gateway_ip,
|
|
})
|
|
} else {
|
|
None
|
|
}
|
|
};
|
|
|
|
// Delegate the actual conf-file rewrite to hive-priv (runs as root).
|
|
crate::priv_client::write_nspawn_flags(container, &binds, isolation, &load_creds).await
|
|
}
|
|
|
|
/// Build the per-line callback for `create_container_streaming` /
|
|
/// `update_container_streaming`. Both ops share identical dispatch logic
|
|
/// (stdout → info + `append_stdout`, stderr → warn + `append_stderr`); this
|
|
/// helper avoids duplicating that match body across the two call sites.
|
|
fn make_log_callback(
|
|
logs: Option<std::sync::Arc<crate::build_logs::BuildLogs>>,
|
|
log_id: Option<i64>,
|
|
cmdline: String,
|
|
) -> impl FnMut(hive_sh4re::priv_proto::PrivStream, &str) {
|
|
use hive_sh4re::priv_proto::PrivStream;
|
|
move |stream, line| match stream {
|
|
PrivStream::Stdout => {
|
|
tracing::info!(target: "nixos-container", cmdline = %cmdline, "{line}");
|
|
if let (Some(h), Some(id)) = (&logs, log_id) {
|
|
h.append_stdout(id, line);
|
|
}
|
|
}
|
|
PrivStream::Stderr => {
|
|
tracing::warn!(target: "nixos-container", cmdline = %cmdline, "{line}");
|
|
if let (Some(h), Some(id)) = (&logs, log_id) {
|
|
h.append_stderr(id, line);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Execute a container operation via hive-priv and integrate with
|
|
/// `build_logs.sqlite`. hive-priv runs as root and forwards output lines
|
|
/// to hive-c0re in real time via the streaming priv protocol. Each line
|
|
/// is appended to the build-log row as it arrives, so the dashboard
|
|
/// shows live progress during long `nixos-container create` / `update` runs.
|
|
async fn priv_run(kind: &str, name: &str) -> Result<()> {
|
|
priv_run_inner(kind, name, None).await
|
|
}
|
|
|
|
/// Like `priv_run` but calls `on_log_id(log_id)` immediately after the
|
|
/// build-log row is opened — before the actual container op starts.
|
|
/// This lets callers surface the row id for live streaming (e.g. the
|
|
/// rebuild-queue worker sets `build_log_id` on the queue entry so the
|
|
/// dashboard can link to `/api/build-logs/id/{id}/stream`).
|
|
///
|
|
/// The callback fires only when a build-log row is successfully opened
|
|
/// (i.e. the global `BuildLogs` handle is installed AND `h.start()`
|
|
/// succeeds). No-op when `on_log_id` is `None` — that's the path for
|
|
/// all callers that don't need the id.
|
|
async fn priv_run_inner(
|
|
kind: &str,
|
|
name: &str,
|
|
on_log_id: Option<&(dyn Fn(i64) + Send + Sync)>,
|
|
) -> Result<()> {
|
|
let container = container_name(name);
|
|
let cmdline = format!("nixos-container {kind} {container}");
|
|
|
|
let logs = crate::build_logs::global();
|
|
let log_id = logs.as_ref().and_then(|h| {
|
|
h.start(name, kind, &cmdline)
|
|
.map_err(|e| {
|
|
tracing::warn!(error = ?e, "build_logs: start failed (priv_run log dropped)");
|
|
})
|
|
.ok()
|
|
});
|
|
// Notify the caller as soon as the log row exists so it can surface
|
|
// the id for live streaming before the container op even starts.
|
|
if let (Some(id), Some(cb)) = (log_id, on_log_id) {
|
|
cb(id);
|
|
}
|
|
|
|
// For long-running ops use the streaming protocol so build_logs
|
|
// receives lines in real time rather than as a batch at completion.
|
|
let result: Result<()> = match kind {
|
|
"create" => {
|
|
crate::priv_client::create_container_streaming(
|
|
name,
|
|
make_log_callback(logs.clone(), log_id, cmdline.clone()),
|
|
)
|
|
.await
|
|
}
|
|
"update" => {
|
|
crate::priv_client::update_container_streaming(
|
|
name,
|
|
make_log_callback(logs.clone(), log_id, cmdline.clone()),
|
|
)
|
|
.await
|
|
}
|
|
"start" => crate::priv_client::start_container(name).await,
|
|
"stop" => crate::priv_client::stop_container(name).await,
|
|
"kill" => crate::priv_client::kill_container(name).await,
|
|
"destroy" => crate::priv_client::destroy_container(name).await,
|
|
other => Err(anyhow::anyhow!("unknown container op: {other}")),
|
|
};
|
|
|
|
let succeeded = result.is_ok();
|
|
if let (Some(h), Some(id)) = (&logs, log_id) {
|
|
h.finish(
|
|
id,
|
|
if succeeded {
|
|
crate::build_logs::BuildStatus::Ok
|
|
} else {
|
|
crate::build_logs::BuildStatus::Fail
|
|
},
|
|
);
|
|
}
|
|
|
|
match result {
|
|
Ok(()) => Ok(()),
|
|
Err(e) => {
|
|
let journal = if kind == "update" {
|
|
container_journal_tail(&container).await
|
|
} else {
|
|
String::new()
|
|
};
|
|
match log_id {
|
|
Some(id) => bail!("{e:#}; see build log #{id}{journal}"),
|
|
None => bail!("{e:#}{journal}"),
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// On a failed `nixos-container update`, the stderr nixos-container
|
|
/// itself prints is often terse ("failed to reload container") — the
|
|
/// real reason (which unit failed `switch-to-configuration` during
|
|
/// the reload phase) lands in the *container's* own journal, not on
|
|
/// the host. Fetch the tail of it so a failed rebuild self-documents
|
|
/// the failing unit in the error string, no second round-trip.
|
|
///
|
|
/// Scoped to `update`: that's the reload-phase case, and the
|
|
/// container is still up (running the old generation) so
|
|
/// `journalctl -M` works. Best-effort — returns "" for other verbs
|
|
/// or when the journal can't be read (machine gone, journalctl
|
|
/// missing); it never produces an error of its own.
|
|
async fn container_journal_tail(container: &str) -> String {
|
|
// `-M` enters the container namespace and needs root, so the read
|
|
// is delegated to hive-priv (hive-c0re itself runs unprivileged).
|
|
let res = crate::priv_client::read_container_journal(
|
|
container,
|
|
hive_sh4re::priv_proto::JournalQuery {
|
|
lines: 40,
|
|
..Default::default()
|
|
},
|
|
)
|
|
.await;
|
|
match res {
|
|
Ok((stdout, _)) if !stdout.is_empty() => format!(
|
|
"\n--- last 40 journal lines from container '{container}' ---\n{}",
|
|
stdout.trim_end()
|
|
),
|
|
_ => String::new(),
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
/// Regression test: `setup_proposed` must seed both agent.nix and flake.nix
|
|
/// in the initial commit. Before commit 5b5a93e flake.nix was missing from
|
|
/// the scaffold, requiring manual creation (seen with the damocles agent).
|
|
#[tokio::test]
|
|
async fn setup_proposed_seeds_flake_nix() {
|
|
let dir = tempfile::tempdir().expect("tempdir");
|
|
let proposed = dir.path().join("proposed");
|
|
setup_proposed(&proposed, "test-agent")
|
|
.await
|
|
.expect("setup_proposed");
|
|
|
|
// Both files must exist on disk.
|
|
assert!(proposed.join("agent.nix").exists(), "agent.nix missing");
|
|
assert!(proposed.join("flake.nix").exists(), "flake.nix missing");
|
|
|
|
// flake.nix must export nixosModules.default (the meta-flake contract).
|
|
let flake = std::fs::read_to_string(proposed.join("flake.nix")).unwrap();
|
|
assert!(
|
|
flake.contains("nixosModules.default"),
|
|
"flake.nix does not export nixosModules.default"
|
|
);
|
|
|
|
// Both files must be tracked in the initial git commit.
|
|
let out = git_command()
|
|
.current_dir(&proposed)
|
|
.args(["show", "--name-only", "--format=", "HEAD"])
|
|
.output()
|
|
.await
|
|
.expect("git show");
|
|
let tracked = String::from_utf8_lossy(&out.stdout);
|
|
assert!(tracked.contains("agent.nix"), "agent.nix not committed");
|
|
assert!(tracked.contains("flake.nix"), "flake.nix not committed");
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_is_in_subnet() {
|
|
// Default subnet 10.42.0.0/24 — agents get .2 to .254.
|
|
let ip = agent_network_ip("alice", "10.42.0.0/24").expect("should produce an IP");
|
|
let octets: Vec<u8> = ip.split('.').map(|o| o.parse().unwrap()).collect();
|
|
assert_eq!(&octets[..3], &[10, 42, 0], "wrong /24 prefix");
|
|
assert!(
|
|
octets[3] >= 2 && octets[3] <= 254,
|
|
"host byte {}",
|
|
octets[3]
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_stable() {
|
|
// Same name + subnet must always produce the same IP.
|
|
let a = agent_network_ip("damocles", "10.42.0.0/24");
|
|
let b = agent_network_ip("damocles", "10.42.0.0/24");
|
|
assert_eq!(a, b);
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_different_agents() {
|
|
// Different agent names very likely produce different IPs (not guaranteed,
|
|
// but for these two names the hashes don't collide).
|
|
let alice = agent_network_ip("alice", "10.42.0.0/24").unwrap();
|
|
let bob = agent_network_ip("bob", "10.42.0.0/24").unwrap();
|
|
assert_ne!(alice, bob, "alice and bob collide — rename one");
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_different_subnet() {
|
|
let ip = agent_network_ip("alice", "192.168.5.0/24").expect("should produce an IP");
|
|
let octets: Vec<u8> = ip.split('.').map(|o| o.parse().unwrap()).collect();
|
|
assert_eq!(&octets[..3], &[192, 168, 5]);
|
|
}
|
|
|
|
#[test]
|
|
fn bridge_gateway_ip_extracts_verbatim_address() {
|
|
// HIVE_NETWORK_SUBNET carries the bridge IP verbatim, not the
|
|
// canonical network — the gateway is the address before the `/`.
|
|
assert_eq!(
|
|
bridge_gateway_ip("10.42.0.1/24").as_deref(),
|
|
Some("10.42.0.1")
|
|
);
|
|
// Non-`.1` operator override: the gateway is wherever the bridge is.
|
|
assert_eq!(
|
|
bridge_gateway_ip("10.42.0.254/24").as_deref(),
|
|
Some("10.42.0.254")
|
|
);
|
|
assert_eq!(
|
|
bridge_gateway_ip("172.30.0.1/16").as_deref(),
|
|
Some("172.30.0.1")
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn bridge_gateway_ip_rejects_bad_input() {
|
|
assert!(bridge_gateway_ip("notanip/24").is_none());
|
|
assert!(bridge_gateway_ip("10.42.0.1").is_none()); // no prefix
|
|
assert!(bridge_gateway_ip("10.42.0.1/33").is_none()); // prefix > 32
|
|
assert!(bridge_gateway_ip("10.42.0.999/24").is_none()); // octet > 255
|
|
assert!(bridge_gateway_ip("10.42.0/24").is_none()); // 3 octets
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_rejects_bad_input() {
|
|
assert!(agent_network_ip("alice", "notanip/24").is_none());
|
|
assert!(agent_network_ip("alice", "10.0.0.0/33").is_none()); // prefix > 32
|
|
assert!(agent_network_ip("alice", "10.0.0.0/31").is_none()); // too small
|
|
assert!(agent_network_ip("alice", "10.0.0.0").is_none()); // no prefix
|
|
}
|
|
|
|
#[test]
|
|
fn agent_network_ip_normalizes_bridge_ip_subnet() {
|
|
// HIVE_NETWORK_SUBNET carries the bridge IP (10.42.0.1/24), not
|
|
// canonical network (10.42.0.0/24). Both must produce the same result
|
|
// after host-bit masking.
|
|
let from_bridge = agent_network_ip("alice", "10.42.0.1/24");
|
|
let from_canonical = agent_network_ip("alice", "10.42.0.0/24");
|
|
assert_eq!(
|
|
from_bridge, from_canonical,
|
|
"bridge-IP and canonical-network form should normalize to the same result"
|
|
);
|
|
// Result must still be in .2-.254.
|
|
let ip = from_bridge.unwrap();
|
|
let last: u8 = ip.rsplit('.').next().unwrap().parse().unwrap();
|
|
assert!((2..=254).contains(&last), "host byte {last}");
|
|
}
|
|
|
|
/// `setup_proposed` is idempotent: calling it on an existing repo is a
|
|
/// no-op (the fresh guard skips all writes).
|
|
#[tokio::test]
|
|
async fn setup_proposed_idempotent() {
|
|
let dir = tempfile::tempdir().expect("tempdir");
|
|
let proposed = dir.path().join("proposed");
|
|
setup_proposed(&proposed, "test-agent")
|
|
.await
|
|
.expect("first call");
|
|
// Second call must not error even though .git already exists.
|
|
setup_proposed(&proposed, "test-agent")
|
|
.await
|
|
.expect("second call");
|
|
// Still one commit.
|
|
let out = git_command()
|
|
.current_dir(&proposed)
|
|
.args(["rev-list", "--count", "HEAD"])
|
|
.output()
|
|
.await
|
|
.expect("git rev-list");
|
|
let count = String::from_utf8_lossy(&out.stdout).trim().to_owned();
|
|
assert_eq!(
|
|
count, "1",
|
|
"expected exactly one commit after idempotent call"
|
|
);
|
|
}
|
|
}
|