From 1d062d1e3e829be8eaf684c39a97e73819903c59 Mon Sep 17 00:00:00 2001 From: atlas Date: Thu, 4 Jun 2026 09:48:56 +0200 Subject: [PATCH] fix: route gateway nginx control through hive-priv MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit systemctl --machine=hive-gateway requires root (machine-bus transport enters the container namespace). hive-c0re is unprivileged, so every call to nginx_active_state() and gateway_systemctl() silently failed with exit 1, causing a continuous 30s retry loop without ever syncing nginx. Fix: - Move state-aware nginx logic into hive-priv ReloadGatewayNginx: check ActiveState, then reload/reset-start/start accordingly. hive-priv already runs as root and has machine-bus rights. - Remove nginx_active_state() and gateway_systemctl() from gateway_nginx.rs (they were always running unprivileged, always failing silently). - Make write(), reload_if_pending(), reload_gateway_nginx() async so they can call the async priv_client without a blocking bridge. - Update callers in agent_sockets::spawn_poll and meta::sync_agents to await the now-async functions. The priv_client::reload_gateway_nginx() call and PrivRequest::ReloadGatewayNginx wire type already existed — the gateway_nginx module was just not using them. --- hive-c0re/src/agent_sockets.rs | 4 +- hive-c0re/src/gateway_nginx.rs | 154 ++++++++------------------------- hive-c0re/src/meta.rs | 12 ++- hive-priv/src/main.rs | 72 ++++++++++++--- 4 files changed, 103 insertions(+), 139 deletions(-) diff --git a/hive-c0re/src/agent_sockets.rs b/hive-c0re/src/agent_sockets.rs index a18a1bbd..19b7644e 100644 --- a/hive-c0re/src/agent_sockets.rs +++ b/hive-c0re/src/agent_sockets.rs @@ -194,12 +194,12 @@ pub fn spawn_poll() { // selection (UDS vs TCP) depends on .bound markers // which change independently of topology. Write is // idempotent; skips rename when nothing changed. - if let Err(e) = crate::gateway_nginx::write(&names) { + if let Err(e) = crate::gateway_nginx::write(&names).await { tracing::debug!(error = ?e, "gateway_nginx poll write failed"); } // Retry a pending nginx reload that failed on a // previous tick (no-op if no reload is pending). - crate::gateway_nginx::reload_if_pending(); + crate::gateway_nginx::reload_if_pending().await; } Err(e) => { tracing::debug!(error = ?e, "agent_sockets poll: failed to list agents"); diff --git a/hive-c0re/src/gateway_nginx.rs b/hive-c0re/src/gateway_nginx.rs index 56d95189..70468784 100644 --- a/hive-c0re/src/gateway_nginx.rs +++ b/hive-c0re/src/gateway_nginx.rs @@ -11,6 +11,8 @@ use std::path::PathBuf; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::time::{SystemTime, UNIX_EPOCH}; +use crate::priv_client; + use crate::agent_sockets; use crate::lifecycle; @@ -176,20 +178,20 @@ fn render(names: &[String], frontend_dir: Option<&str>) -> String { /// body matches what's already on disk (idempotent; avoids spurious /// gateway reloads on a quiet tick). /// -/// After a successful write, triggers an nginx reload inside the -/// gateway container from the HOST side via -/// `systemd-run --machine=hive-gateway nginx -s reload`. This is -/// intentionally host-side rather than relying on a systemd path unit -/// inside the container watching the bind-mounted file: `IN_MOVED_TO` -/// (fired by the atomic rename) does not reliably propagate across the -/// nspawn mount-namespace boundary, so the path-unit approach was -/// silently broken (see `docs/gateway.md` for the failure analysis). +/// After a successful write, triggers the appropriate nginx action inside +/// the gateway container via `hive-priv` (which has the +/// `--machine=hive-gateway` transport rights hive-c0re lacks): +/// reload when nginx is active, reset-failed+start when in a failed +/// state, plain start otherwise. This is intentionally host-side rather +/// than relying on a systemd path unit inside the container watching the +/// bind-mounted file: `IN_MOVED_TO` (fired by the atomic rename) does +/// not reliably propagate across the nspawn mount-namespace boundary, so +/// the path-unit approach was silently broken (see `docs/gateway.md`). /// -/// The `systemd-run` call is best-effort — a failed reload is logged -/// but not fatal. nginx will pick up the new include on its next -/// housekeeping restart or the next manual reload; the host's agent -/// topology has already been written correctly. -pub fn write(names: &[String]) -> Result<()> { +/// The priv call is best-effort — a failed sync is logged but not fatal. +/// `reload_if_pending` retries on the next `spawn_poll` tick so a +/// transient gateway-down situation converges without manual intervention. +pub async fn write(names: &[String]) -> Result<()> { let frontend_dir = std::env::var("HIVE_AGENT_FRONTEND_DIR") .ok() .filter(|s| !s.is_empty()); @@ -217,7 +219,7 @@ pub fn write(names: &[String]) -> Result<()> { // Mark reload pending before attempting so a failed attempt is // retried by the next spawn_poll tick (see `reload_if_pending`). RELOAD_PENDING.store(true, Ordering::Relaxed); - reload_gateway_nginx(); + reload_gateway_nginx().await; Ok(()) } @@ -231,7 +233,7 @@ pub fn write(names: &[String]) -> Result<()> { /// A fresh `write()` call always resets the backoff (new RELOAD_PENDING /// set to `true` + immediate attempt) so topology changes are still /// applied promptly. -pub fn reload_if_pending() { +pub async fn reload_if_pending() { if !RELOAD_PENDING.load(Ordering::Relaxed) { return; } @@ -245,116 +247,32 @@ pub fn reload_if_pending() { return; } } - reload_gateway_nginx(); + reload_gateway_nginx().await; } -/// Query the nginx unit's `ActiveState` inside the gateway container. -/// Returns the raw state string from `systemctl show --property=ActiveState -/// --value` (e.g. `"active"`, `"failed"`, `"inactive"`, `"activating"`). -/// Returns `"unknown"` on any error so callers can branch safely. -fn nginx_active_state() -> String { - let out = std::process::Command::new("systemctl") - .args([ - "--machine=hive-gateway", - "show", - "--property=ActiveState", - "--value", - "nginx", - ]) - .output(); - match out { - Ok(o) if o.status.success() => String::from_utf8_lossy(&o.stdout).trim().to_owned(), - Ok(o) => { - tracing::warn!( - exit_code = ?o.status.code(), - "systemctl show ActiveState exited non-zero" - ); - "unknown".to_owned() - } - Err(e) => { - tracing::warn!(error = %e, "systemctl show ActiveState failed"); - "unknown".to_owned() - } - } -} - -/// Send a `systemctl --machine=hive-gateway ` command and -/// return whether it succeeded. Best-effort: errors are logged. -fn gateway_systemctl(args: &[&str]) -> bool { - let mut cmd = std::process::Command::new("systemctl"); - cmd.arg("--machine=hive-gateway"); - cmd.args(args); - match cmd.status() { - Ok(s) if s.success() => true, - Ok(s) => { - tracing::warn!( - args = ?args, - exit_code = ?s.code(), - "gateway systemctl exited non-zero" - ); - false - } - Err(e) => { - tracing::warn!(args = ?args, error = %e, "gateway systemctl invocation failed"); - false - } - } -} - -/// Synchronise the gateway nginx unit with the current agents.conf: -/// -/// - **active**: send `nginx -s reload` (SIGHUP to master, zero-downtime -/// worker replacement). Keeps `RELOAD_PENDING` set on failure so the -/// next poll tick retries. -/// - **failed / start-limit-hit**: run `systemctl reset-failed nginx` -/// then `systemctl start nginx`. This is the self-healing path: a -/// transient bad agents.conf that causes five instant `nginx -t` -/// failures trips systemd's start-limit. Once c0re publishes a correct -/// config, the next reload attempt clears the failure and restarts. -/// - **inactive / other**: run `systemctl start nginx` directly (no -/// reset-failed needed when the unit isn't in a failed state). +/// Synchronise the gateway nginx unit with the current agents.conf via +/// `hive-priv` (privileged helper). The state-aware logic (active → +/// reload; failed → reset-failed + start; inactive/unknown → start) +/// runs inside hive-priv where it has the `--machine=hive-gateway` +/// transport rights that hive-c0re (unprivileged) lacks. /// /// `RELOAD_PENDING` is cleared only after a successful operation so /// `reload_if_pending` keeps retrying on failure. -fn reload_gateway_nginx() { - let state = nginx_active_state(); - let success = match state.as_str() { - "active" => { - // nginx master is running — ask systemd to reload the unit - // (SIGHUP to master, zero-downtime worker replacement). - // `systemctl -M hive-gateway reload nginx` lets systemd - // resolve the binary path; avoids the exit-203 (EXEC) - // failure that `systemd-run -- nginx` hit on NixOS where - // the limited transient-unit PATH misses /run/current-system/sw/bin/. - let ok = gateway_systemctl(&["reload", "nginx"]); - if ok { - tracing::debug!("gateway nginx reload signal sent"); - } - ok +async fn reload_gateway_nginx() { + match priv_client::reload_gateway_nginx().await { + Ok(()) => { + tracing::debug!("gateway nginx sync succeeded"); + RELOAD_PENDING.store(false, Ordering::Relaxed); + LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); } - "failed" => { - // Unit hit start-limit (e.g. repeated nginx -t failures from - // a bad agents.conf). reset-failed clears the rate-limit so - // start can proceed. - tracing::info!("gateway nginx unit in failed state — resetting and starting"); - gateway_systemctl(&["reset-failed", "nginx"]) && gateway_systemctl(&["start", "nginx"]) + Err(e) => { + tracing::warn!(error = %e, "gateway nginx sync failed — will retry"); + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); } - other => { - // inactive, deactivating, activating, unknown — just try start. - tracing::info!(state = other, "gateway nginx unit not active — starting"); - gateway_systemctl(&["start", "nginx"]) - } - }; - - if success { - RELOAD_PENDING.store(false, Ordering::Relaxed); - LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); - } else { - let now = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_secs(); - LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); } } diff --git a/hive-c0re/src/meta.rs b/hive-c0re/src/meta.rs index f4653462..e8613ca4 100644 --- a/hive-c0re/src/meta.rs +++ b/hive-c0re/src/meta.rs @@ -126,13 +126,11 @@ pub async fn sync_agents(hive: &HiveEnv, agents: &[AgentSpec]) -> Result<()> { tracing::warn!(error = ?e, "agent_sockets::write failed (non-fatal)"); } - // Refresh /var/lib/hyperhive/agents.conf — the nginx include file - // the gateway picks up at runtime without needing a - // nixos-rebuild. The gateway container bind-mounts - // /var/lib/hyperhive/ and a systemd path unit fires - // `nginx -s reload` when this file changes. Same - // best-effort + non-fatal shape. - if let Err(e) = crate::gateway_nginx::write(&agent_names) { + // Refresh /var/lib/hyperhive/gateway/agents.conf — the nginx include + // file the gateway container bind-mounts and nginx reads at runtime. + // c0re triggers a reload (or start) inside hive-gateway via hive-priv + // after writing the file. Same best-effort + non-fatal shape. + if let Err(e) = crate::gateway_nginx::write(&agent_names).await { tracing::warn!(error = ?e, "gateway_nginx::write failed (non-fatal)"); } diff --git a/hive-priv/src/main.rs b/hive-priv/src/main.rs index 978626bb..ba96273d 100644 --- a/hive-priv/src/main.rs +++ b/hive-priv/src/main.rs @@ -289,24 +289,72 @@ async fn exec(req: PrivRequest, writer: &mut OwnedWriteHalf) -> Result<(String, } PrivRequest::ReloadGatewayNginx => { - let out = Command::new("systemd-run") + // Query the nginx unit's ActiveState inside the gateway container. + // Requires root: --machine= transport enters the container namespace + // via the machine bus, which is forbidden for unprivileged users. + let state_out = Command::new("systemctl") .args([ "--machine=hive-gateway", - "--quiet", - "--", + "show", + "--property=ActiveState", + "--value", "nginx", - "-s", - "reload", ]) .output() .await - .context("invoke systemd-run for gateway nginx reload")?; - if !out.status.success() { - bail!( - "gateway nginx reload failed ({}): {}", - out.status, - String::from_utf8_lossy(&out.stderr).trim() - ); + .context("query nginx ActiveState in hive-gateway")?; + let state = String::from_utf8_lossy(&state_out.stdout).trim().to_owned(); + // State-aware action: reload when running; reset+start after + // start-limit failure; plain start when inactive or unknown. + match state.as_str() { + "active" => { + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "reload", "nginx"]) + .output() + .await + .context("reload nginx in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx reload failed ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } + "failed" => { + // Clear start-limit hit so the next start can proceed. + let _ = Command::new("systemctl") + .args(["--machine=hive-gateway", "reset-failed", "nginx"]) + .status() + .await; + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "start", "nginx"]) + .output() + .await + .context("start nginx after reset-failed in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx start (after reset-failed) failed ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } + _ => { + // inactive, activating, deactivating, unknown — just start. + let out = Command::new("systemctl") + .args(["--machine=hive-gateway", "start", "nginx"]) + .output() + .await + .context("start nginx in hive-gateway")?; + if !out.status.success() { + bail!( + "gateway nginx start failed (state={state}) ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + } } Ok((String::new(), String::new())) }