diff --git a/docs/gateway.md b/docs/gateway.md index 54f56860..a0b16d63 100644 --- a/docs/gateway.md +++ b/docs/gateway.md @@ -97,11 +97,15 @@ now set unconditionally for every agent. The mechanism: that haven't yet been rebuilt under the new config. The gateway container bind-mounts `/var/lib/hyperhive/gateway/` at `/run/hive-state/`; nginx includes `/run/hive-state/agents.conf`. - After each write, c0re triggers `nginx -s reload` inside the - gateway container from the HOST via - `systemd-run --machine=hive-gateway --wait nginx -s reload`. This is - intentionally host-side: `IN_MOVED_TO` from an atomic rename does - not propagate across the nspawn mount-namespace boundary, so a + After each write, c0re triggers the appropriate nginx action inside + the gateway container via `hive-priv` (which runs as root and has + `--machine=hive-gateway` transport rights that hive-c0re lacks). + `hive-priv` queries `ActiveState` and dispatches: + - active → `systemctl reload nginx` (SIGHUP, zero-downtime) + - failed → `systemctl reset-failed nginx` + `systemctl start nginx` + - otherwise → `systemctl start nginx` + This is intentionally host-side: `IN_MOVED_TO` from an atomic rename + does not propagate across the nspawn mount-namespace boundary, so a path unit inside the container would never fire. c0re regenerates `agents.conf` (and triggers a reload) on two diff --git a/hive-c0re/src/agent_sockets.rs b/hive-c0re/src/agent_sockets.rs index a18a1bbd..19b7644e 100644 --- a/hive-c0re/src/agent_sockets.rs +++ b/hive-c0re/src/agent_sockets.rs @@ -194,12 +194,12 @@ pub fn spawn_poll() { // selection (UDS vs TCP) depends on .bound markers // which change independently of topology. Write is // idempotent; skips rename when nothing changed. - if let Err(e) = crate::gateway_nginx::write(&names) { + if let Err(e) = crate::gateway_nginx::write(&names).await { tracing::debug!(error = ?e, "gateway_nginx poll write failed"); } // Retry a pending nginx reload that failed on a // previous tick (no-op if no reload is pending). - crate::gateway_nginx::reload_if_pending(); + crate::gateway_nginx::reload_if_pending().await; } Err(e) => { tracing::debug!(error = ?e, "agent_sockets poll: failed to list agents"); diff --git a/hive-c0re/src/gateway_nginx.rs b/hive-c0re/src/gateway_nginx.rs index 56d95189..70468784 100644 --- a/hive-c0re/src/gateway_nginx.rs +++ b/hive-c0re/src/gateway_nginx.rs @@ -11,6 +11,8 @@ use std::path::PathBuf; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::time::{SystemTime, UNIX_EPOCH}; +use crate::priv_client; + use crate::agent_sockets; use crate::lifecycle; @@ -176,20 +178,20 @@ fn render(names: &[String], frontend_dir: Option<&str>) -> String { /// body matches what's already on disk (idempotent; avoids spurious /// gateway reloads on a quiet tick). /// -/// After a successful write, triggers an nginx reload inside the -/// gateway container from the HOST side via -/// `systemd-run --machine=hive-gateway nginx -s reload`. This is -/// intentionally host-side rather than relying on a systemd path unit -/// inside the container watching the bind-mounted file: `IN_MOVED_TO` -/// (fired by the atomic rename) does not reliably propagate across the -/// nspawn mount-namespace boundary, so the path-unit approach was -/// silently broken (see `docs/gateway.md` for the failure analysis). +/// After a successful write, triggers the appropriate nginx action inside +/// the gateway container via `hive-priv` (which has the +/// `--machine=hive-gateway` transport rights hive-c0re lacks): +/// reload when nginx is active, reset-failed+start when in a failed +/// state, plain start otherwise. This is intentionally host-side rather +/// than relying on a systemd path unit inside the container watching the +/// bind-mounted file: `IN_MOVED_TO` (fired by the atomic rename) does +/// not reliably propagate across the nspawn mount-namespace boundary, so +/// the path-unit approach was silently broken (see `docs/gateway.md`). /// -/// The `systemd-run` call is best-effort — a failed reload is logged -/// but not fatal. nginx will pick up the new include on its next -/// housekeeping restart or the next manual reload; the host's agent -/// topology has already been written correctly. -pub fn write(names: &[String]) -> Result<()> { +/// The priv call is best-effort — a failed sync is logged but not fatal. +/// `reload_if_pending` retries on the next `spawn_poll` tick so a +/// transient gateway-down situation converges without manual intervention. +pub async fn write(names: &[String]) -> Result<()> { let frontend_dir = std::env::var("HIVE_AGENT_FRONTEND_DIR") .ok() .filter(|s| !s.is_empty()); @@ -217,7 +219,7 @@ pub fn write(names: &[String]) -> Result<()> { // Mark reload pending before attempting so a failed attempt is // retried by the next spawn_poll tick (see `reload_if_pending`). RELOAD_PENDING.store(true, Ordering::Relaxed); - reload_gateway_nginx(); + reload_gateway_nginx().await; Ok(()) } @@ -231,7 +233,7 @@ pub fn write(names: &[String]) -> Result<()> { /// A fresh `write()` call always resets the backoff (new RELOAD_PENDING /// set to `true` + immediate attempt) so topology changes are still /// applied promptly. -pub fn reload_if_pending() { +pub async fn reload_if_pending() { if !RELOAD_PENDING.load(Ordering::Relaxed) { return; } @@ -245,116 +247,32 @@ pub fn reload_if_pending() { return; } } - reload_gateway_nginx(); + reload_gateway_nginx().await; } -/// Query the nginx unit's `ActiveState` inside the gateway container. -/// Returns the raw state string from `systemctl show --property=ActiveState -/// --value` (e.g. `"active"`, `"failed"`, `"inactive"`, `"activating"`). -/// Returns `"unknown"` on any error so callers can branch safely. -fn nginx_active_state() -> String { - let out = std::process::Command::new("systemctl") - .args([ - "--machine=hive-gateway", - "show", - "--property=ActiveState", - "--value", - "nginx", - ]) - .output(); - match out { - Ok(o) if o.status.success() => String::from_utf8_lossy(&o.stdout).trim().to_owned(), - Ok(o) => { - tracing::warn!( - exit_code = ?o.status.code(), - "systemctl show ActiveState exited non-zero" - ); - "unknown".to_owned() - } - Err(e) => { - tracing::warn!(error = %e, "systemctl show ActiveState failed"); - "unknown".to_owned() - } - } -} - -/// Send a `systemctl --machine=hive-gateway ` command and -/// return whether it succeeded. Best-effort: errors are logged. -fn gateway_systemctl(args: &[&str]) -> bool { - let mut cmd = std::process::Command::new("systemctl"); - cmd.arg("--machine=hive-gateway"); - cmd.args(args); - match cmd.status() { - Ok(s) if s.success() => true, - Ok(s) => { - tracing::warn!( - args = ?args, - exit_code = ?s.code(), - "gateway systemctl exited non-zero" - ); - false - } - Err(e) => { - tracing::warn!(args = ?args, error = %e, "gateway systemctl invocation failed"); - false - } - } -} - -/// Synchronise the gateway nginx unit with the current agents.conf: -/// -/// - **active**: send `nginx -s reload` (SIGHUP to master, zero-downtime -/// worker replacement). Keeps `RELOAD_PENDING` set on failure so the -/// next poll tick retries. -/// - **failed / start-limit-hit**: run `systemctl reset-failed nginx` -/// then `systemctl start nginx`. This is the self-healing path: a -/// transient bad agents.conf that causes five instant `nginx -t` -/// failures trips systemd's start-limit. Once c0re publishes a correct -/// config, the next reload attempt clears the failure and restarts. -/// - **inactive / other**: run `systemctl start nginx` directly (no -/// reset-failed needed when the unit isn't in a failed state). +/// Synchronise the gateway nginx unit with the current agents.conf via +/// `hive-priv` (privileged helper). The state-aware logic (active → +/// reload; failed → reset-failed + start; inactive/unknown → start) +/// runs inside hive-priv where it has the `--machine=hive-gateway` +/// transport rights that hive-c0re (unprivileged) lacks. /// /// `RELOAD_PENDING` is cleared only after a successful operation so /// `reload_if_pending` keeps retrying on failure. -fn reload_gateway_nginx() { - let state = nginx_active_state(); - let success = match state.as_str() { - "active" => { - // nginx master is running — ask systemd to reload the unit - // (SIGHUP to master, zero-downtime worker replacement). - // `systemctl -M hive-gateway reload nginx` lets systemd - // resolve the binary path; avoids the exit-203 (EXEC) - // failure that `systemd-run -- nginx` hit on NixOS where - // the limited transient-unit PATH misses /run/current-system/sw/bin/. - let ok = gateway_systemctl(&["reload", "nginx"]); - if ok { - tracing::debug!("gateway nginx reload signal sent"); - } - ok +async fn reload_gateway_nginx() { + match priv_client::reload_gateway_nginx().await { + Ok(()) => { + tracing::debug!("gateway nginx sync succeeded"); + RELOAD_PENDING.store(false, Ordering::Relaxed); + LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); } - "failed" => { - // Unit hit start-limit (e.g. repeated nginx -t failures from - // a bad agents.conf). reset-failed clears the rate-limit so - // start can proceed. - tracing::info!("gateway nginx unit in failed state — resetting and starting"); - gateway_systemctl(&["reset-failed", "nginx"]) && gateway_systemctl(&["start", "nginx"]) + Err(e) => { + tracing::warn!(error = %e, "gateway nginx sync failed — will retry"); + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); } - other => { - // inactive, deactivating, activating, unknown — just try start. - tracing::info!(state = other, "gateway nginx unit not active — starting"); - gateway_systemctl(&["start", "nginx"]) - } - }; - - if success { - RELOAD_PENDING.store(false, Ordering::Relaxed); - LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); - } else { - let now = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_secs(); - LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); } } diff --git a/hive-c0re/src/meta.rs b/hive-c0re/src/meta.rs index f4653462..e8613ca4 100644 --- a/hive-c0re/src/meta.rs +++ b/hive-c0re/src/meta.rs @@ -126,13 +126,11 @@ pub async fn sync_agents(hive: &HiveEnv, agents: &[AgentSpec]) -> Result<()> { tracing::warn!(error = ?e, "agent_sockets::write failed (non-fatal)"); } - // Refresh /var/lib/hyperhive/agents.conf — the nginx include file - // the gateway picks up at runtime without needing a - // nixos-rebuild. The gateway container bind-mounts - // /var/lib/hyperhive/ and a systemd path unit fires - // `nginx -s reload` when this file changes. Same - // best-effort + non-fatal shape. - if let Err(e) = crate::gateway_nginx::write(&agent_names) { + // Refresh /var/lib/hyperhive/gateway/agents.conf — the nginx include + // file the gateway container bind-mounts and nginx reads at runtime. + // c0re triggers a reload (or start) inside hive-gateway via hive-priv + // after writing the file. Same best-effort + non-fatal shape. + if let Err(e) = crate::gateway_nginx::write(&agent_names).await { tracing::warn!(error = ?e, "gateway_nginx::write failed (non-fatal)"); } diff --git a/hive-priv/src/main.rs b/hive-priv/src/main.rs index 978626bb..3c435c27 100644 --- a/hive-priv/src/main.rs +++ b/hive-priv/src/main.rs @@ -288,28 +288,7 @@ async fn exec(req: PrivRequest, writer: &mut OwnedWriteHalf) -> Result<(String, Ok((String::new(), String::new())) } - PrivRequest::ReloadGatewayNginx => { - let out = Command::new("systemd-run") - .args([ - "--machine=hive-gateway", - "--quiet", - "--", - "nginx", - "-s", - "reload", - ]) - .output() - .await - .context("invoke systemd-run for gateway nginx reload")?; - if !out.status.success() { - bail!( - "gateway nginx reload failed ({}): {}", - out.status, - String::from_utf8_lossy(&out.stderr).trim() - ); - } - Ok((String::new(), String::new())) - } + PrivRequest::ReloadGatewayNginx => sync_gateway_nginx().await, PrivRequest::ChownSocketDir { ref agent_name, @@ -581,6 +560,92 @@ async fn read_container_journal( Ok((stdout, stderr)) } +/// Synchronise the nginx unit inside the `hive-gateway` container. +/// +/// Queries `ActiveState` via `systemctl --machine=hive-gateway` (requires +/// root — machine-bus transport enters the container namespace), then +/// dispatches: +/// - `active` → `systemctl reload nginx` (SIGHUP, zero-downtime) +/// - `failed` → `systemctl reset-failed nginx` + `systemctl start nginx` +/// - otherwise → `systemctl start nginx` +/// +/// Returns `(String::new(), String::new())` on success so it fits the +/// `exec` return type directly. +async fn sync_gateway_nginx() -> Result<(String, String)> { + let state_out = Command::new("systemctl") + .args([ + "--machine=hive-gateway", + "show", + "--property=ActiveState", + "--value", + "nginx", + ]) + .output() + .await + .context("query nginx ActiveState in hive-gateway")?; + if !state_out.status.success() { + tracing::warn!( + exit_code = ?state_out.status.code(), + stderr = %String::from_utf8_lossy(&state_out.stderr).trim(), + "systemctl show ActiveState exited non-zero — gateway container may be down" + ); + } + let state = String::from_utf8_lossy(&state_out.stdout).trim().to_owned(); + // State-aware dispatch: reload when running; reset+start after + // start-limit failure; plain start when inactive or unknown. + match state.as_str() { + "active" => { + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "reload", "nginx"]) + .output() + .await + .context("reload nginx in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx reload failed ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } + "failed" => { + // Clear start-limit so the next start can proceed. + let _ = Command::new("systemctl") + .args(["--machine=hive-gateway", "reset-failed", "nginx"]) + .status() + .await; + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "start", "nginx"]) + .output() + .await + .context("start nginx after reset-failed in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx start (after reset-failed) failed ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } + _ => { + // inactive, activating, deactivating, unknown — just start. + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "start", "nginx"]) + .output() + .await + .context("start nginx in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx start failed (state={state:?}) ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } + } + Ok((String::new(), String::new())) +} + /// Return the system container name for a logical agent name. /// All agents (including the manager) use the `h-` prefix. fn container_system_name(name: &str) -> String { diff --git a/hive-sh4re/src/priv_proto.rs b/hive-sh4re/src/priv_proto.rs index 3638084c..288d53c1 100644 --- a/hive-sh4re/src/priv_proto.rs +++ b/hive-sh4re/src/priv_proto.rs @@ -182,8 +182,13 @@ pub enum PrivRequest { /// Run `systemctl daemon-reload`. DaemonReload, - /// Reload nginx inside the `hive-gateway` container via - /// `systemd-run --machine=hive-gateway nginx -s reload`. + /// Synchronise the nginx unit inside the `hive-gateway` container. + /// hive-priv queries `ActiveState` and dispatches: + /// - `active` → `systemctl reload nginx` (SIGHUP, zero-downtime) + /// - `failed` → `systemctl reset-failed nginx` + `systemctl start nginx` + /// - otherwise → `systemctl start nginx` + /// Requires root: `--machine=hive-gateway` enters the container + /// namespace via the machine bus (forbidden for unprivileged users). ReloadGatewayNginx, // --- Socket dir ownership ---