diff --git a/docs/gateway.md b/docs/gateway.md index a0b16d63..54f56860 100644 --- a/docs/gateway.md +++ b/docs/gateway.md @@ -97,15 +97,11 @@ now set unconditionally for every agent. The mechanism: that haven't yet been rebuilt under the new config. The gateway container bind-mounts `/var/lib/hyperhive/gateway/` at `/run/hive-state/`; nginx includes `/run/hive-state/agents.conf`. - After each write, c0re triggers the appropriate nginx action inside - the gateway container via `hive-priv` (which runs as root and has - `--machine=hive-gateway` transport rights that hive-c0re lacks). - `hive-priv` queries `ActiveState` and dispatches: - - active → `systemctl reload nginx` (SIGHUP, zero-downtime) - - failed → `systemctl reset-failed nginx` + `systemctl start nginx` - - otherwise → `systemctl start nginx` - This is intentionally host-side: `IN_MOVED_TO` from an atomic rename - does not propagate across the nspawn mount-namespace boundary, so a + After each write, c0re triggers `nginx -s reload` inside the + gateway container from the HOST via + `systemd-run --machine=hive-gateway --wait nginx -s reload`. This is + intentionally host-side: `IN_MOVED_TO` from an atomic rename does + not propagate across the nspawn mount-namespace boundary, so a path unit inside the container would never fire. c0re regenerates `agents.conf` (and triggers a reload) on two diff --git a/hive-c0re/src/agent_sockets.rs b/hive-c0re/src/agent_sockets.rs index 19b7644e..a18a1bbd 100644 --- a/hive-c0re/src/agent_sockets.rs +++ b/hive-c0re/src/agent_sockets.rs @@ -194,12 +194,12 @@ pub fn spawn_poll() { // selection (UDS vs TCP) depends on .bound markers // which change independently of topology. Write is // idempotent; skips rename when nothing changed. - if let Err(e) = crate::gateway_nginx::write(&names).await { + if let Err(e) = crate::gateway_nginx::write(&names) { tracing::debug!(error = ?e, "gateway_nginx poll write failed"); } // Retry a pending nginx reload that failed on a // previous tick (no-op if no reload is pending). - crate::gateway_nginx::reload_if_pending().await; + crate::gateway_nginx::reload_if_pending(); } Err(e) => { tracing::debug!(error = ?e, "agent_sockets poll: failed to list agents"); diff --git a/hive-c0re/src/gateway_nginx.rs b/hive-c0re/src/gateway_nginx.rs index 70468784..56d95189 100644 --- a/hive-c0re/src/gateway_nginx.rs +++ b/hive-c0re/src/gateway_nginx.rs @@ -11,8 +11,6 @@ use std::path::PathBuf; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::time::{SystemTime, UNIX_EPOCH}; -use crate::priv_client; - use crate::agent_sockets; use crate::lifecycle; @@ -178,20 +176,20 @@ fn render(names: &[String], frontend_dir: Option<&str>) -> String { /// body matches what's already on disk (idempotent; avoids spurious /// gateway reloads on a quiet tick). /// -/// After a successful write, triggers the appropriate nginx action inside -/// the gateway container via `hive-priv` (which has the -/// `--machine=hive-gateway` transport rights hive-c0re lacks): -/// reload when nginx is active, reset-failed+start when in a failed -/// state, plain start otherwise. This is intentionally host-side rather -/// than relying on a systemd path unit inside the container watching the -/// bind-mounted file: `IN_MOVED_TO` (fired by the atomic rename) does -/// not reliably propagate across the nspawn mount-namespace boundary, so -/// the path-unit approach was silently broken (see `docs/gateway.md`). +/// After a successful write, triggers an nginx reload inside the +/// gateway container from the HOST side via +/// `systemd-run --machine=hive-gateway nginx -s reload`. This is +/// intentionally host-side rather than relying on a systemd path unit +/// inside the container watching the bind-mounted file: `IN_MOVED_TO` +/// (fired by the atomic rename) does not reliably propagate across the +/// nspawn mount-namespace boundary, so the path-unit approach was +/// silently broken (see `docs/gateway.md` for the failure analysis). /// -/// The priv call is best-effort — a failed sync is logged but not fatal. -/// `reload_if_pending` retries on the next `spawn_poll` tick so a -/// transient gateway-down situation converges without manual intervention. -pub async fn write(names: &[String]) -> Result<()> { +/// The `systemd-run` call is best-effort — a failed reload is logged +/// but not fatal. nginx will pick up the new include on its next +/// housekeeping restart or the next manual reload; the host's agent +/// topology has already been written correctly. +pub fn write(names: &[String]) -> Result<()> { let frontend_dir = std::env::var("HIVE_AGENT_FRONTEND_DIR") .ok() .filter(|s| !s.is_empty()); @@ -219,7 +217,7 @@ pub async fn write(names: &[String]) -> Result<()> { // Mark reload pending before attempting so a failed attempt is // retried by the next spawn_poll tick (see `reload_if_pending`). RELOAD_PENDING.store(true, Ordering::Relaxed); - reload_gateway_nginx().await; + reload_gateway_nginx(); Ok(()) } @@ -233,7 +231,7 @@ pub async fn write(names: &[String]) -> Result<()> { /// A fresh `write()` call always resets the backoff (new RELOAD_PENDING /// set to `true` + immediate attempt) so topology changes are still /// applied promptly. -pub async fn reload_if_pending() { +pub fn reload_if_pending() { if !RELOAD_PENDING.load(Ordering::Relaxed) { return; } @@ -247,32 +245,116 @@ pub async fn reload_if_pending() { return; } } - reload_gateway_nginx().await; + reload_gateway_nginx(); } -/// Synchronise the gateway nginx unit with the current agents.conf via -/// `hive-priv` (privileged helper). The state-aware logic (active → -/// reload; failed → reset-failed + start; inactive/unknown → start) -/// runs inside hive-priv where it has the `--machine=hive-gateway` -/// transport rights that hive-c0re (unprivileged) lacks. +/// Query the nginx unit's `ActiveState` inside the gateway container. +/// Returns the raw state string from `systemctl show --property=ActiveState +/// --value` (e.g. `"active"`, `"failed"`, `"inactive"`, `"activating"`). +/// Returns `"unknown"` on any error so callers can branch safely. +fn nginx_active_state() -> String { + let out = std::process::Command::new("systemctl") + .args([ + "--machine=hive-gateway", + "show", + "--property=ActiveState", + "--value", + "nginx", + ]) + .output(); + match out { + Ok(o) if o.status.success() => String::from_utf8_lossy(&o.stdout).trim().to_owned(), + Ok(o) => { + tracing::warn!( + exit_code = ?o.status.code(), + "systemctl show ActiveState exited non-zero" + ); + "unknown".to_owned() + } + Err(e) => { + tracing::warn!(error = %e, "systemctl show ActiveState failed"); + "unknown".to_owned() + } + } +} + +/// Send a `systemctl --machine=hive-gateway ` command and +/// return whether it succeeded. Best-effort: errors are logged. +fn gateway_systemctl(args: &[&str]) -> bool { + let mut cmd = std::process::Command::new("systemctl"); + cmd.arg("--machine=hive-gateway"); + cmd.args(args); + match cmd.status() { + Ok(s) if s.success() => true, + Ok(s) => { + tracing::warn!( + args = ?args, + exit_code = ?s.code(), + "gateway systemctl exited non-zero" + ); + false + } + Err(e) => { + tracing::warn!(args = ?args, error = %e, "gateway systemctl invocation failed"); + false + } + } +} + +/// Synchronise the gateway nginx unit with the current agents.conf: +/// +/// - **active**: send `nginx -s reload` (SIGHUP to master, zero-downtime +/// worker replacement). Keeps `RELOAD_PENDING` set on failure so the +/// next poll tick retries. +/// - **failed / start-limit-hit**: run `systemctl reset-failed nginx` +/// then `systemctl start nginx`. This is the self-healing path: a +/// transient bad agents.conf that causes five instant `nginx -t` +/// failures trips systemd's start-limit. Once c0re publishes a correct +/// config, the next reload attempt clears the failure and restarts. +/// - **inactive / other**: run `systemctl start nginx` directly (no +/// reset-failed needed when the unit isn't in a failed state). /// /// `RELOAD_PENDING` is cleared only after a successful operation so /// `reload_if_pending` keeps retrying on failure. -async fn reload_gateway_nginx() { - match priv_client::reload_gateway_nginx().await { - Ok(()) => { - tracing::debug!("gateway nginx sync succeeded"); - RELOAD_PENDING.store(false, Ordering::Relaxed); - LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); +fn reload_gateway_nginx() { + let state = nginx_active_state(); + let success = match state.as_str() { + "active" => { + // nginx master is running — ask systemd to reload the unit + // (SIGHUP to master, zero-downtime worker replacement). + // `systemctl -M hive-gateway reload nginx` lets systemd + // resolve the binary path; avoids the exit-203 (EXEC) + // failure that `systemd-run -- nginx` hit on NixOS where + // the limited transient-unit PATH misses /run/current-system/sw/bin/. + let ok = gateway_systemctl(&["reload", "nginx"]); + if ok { + tracing::debug!("gateway nginx reload signal sent"); + } + ok } - Err(e) => { - tracing::warn!(error = %e, "gateway nginx sync failed — will retry"); - let now = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_secs(); - LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); + "failed" => { + // Unit hit start-limit (e.g. repeated nginx -t failures from + // a bad agents.conf). reset-failed clears the rate-limit so + // start can proceed. + tracing::info!("gateway nginx unit in failed state — resetting and starting"); + gateway_systemctl(&["reset-failed", "nginx"]) && gateway_systemctl(&["start", "nginx"]) } + other => { + // inactive, deactivating, activating, unknown — just try start. + tracing::info!(state = other, "gateway nginx unit not active — starting"); + gateway_systemctl(&["start", "nginx"]) + } + }; + + if success { + RELOAD_PENDING.store(false, Ordering::Relaxed); + LAST_FAILED_RELOAD.store(0, Ordering::Relaxed); + } else { + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + LAST_FAILED_RELOAD.store(now, Ordering::Relaxed); } } diff --git a/hive-c0re/src/meta.rs b/hive-c0re/src/meta.rs index e8613ca4..f4653462 100644 --- a/hive-c0re/src/meta.rs +++ b/hive-c0re/src/meta.rs @@ -126,11 +126,13 @@ pub async fn sync_agents(hive: &HiveEnv, agents: &[AgentSpec]) -> Result<()> { tracing::warn!(error = ?e, "agent_sockets::write failed (non-fatal)"); } - // Refresh /var/lib/hyperhive/gateway/agents.conf — the nginx include - // file the gateway container bind-mounts and nginx reads at runtime. - // c0re triggers a reload (or start) inside hive-gateway via hive-priv - // after writing the file. Same best-effort + non-fatal shape. - if let Err(e) = crate::gateway_nginx::write(&agent_names).await { + // Refresh /var/lib/hyperhive/agents.conf — the nginx include file + // the gateway picks up at runtime without needing a + // nixos-rebuild. The gateway container bind-mounts + // /var/lib/hyperhive/ and a systemd path unit fires + // `nginx -s reload` when this file changes. Same + // best-effort + non-fatal shape. + if let Err(e) = crate::gateway_nginx::write(&agent_names) { tracing::warn!(error = ?e, "gateway_nginx::write failed (non-fatal)"); } diff --git a/hive-priv/src/main.rs b/hive-priv/src/main.rs index 3c435c27..978626bb 100644 --- a/hive-priv/src/main.rs +++ b/hive-priv/src/main.rs @@ -288,7 +288,28 @@ async fn exec(req: PrivRequest, writer: &mut OwnedWriteHalf) -> Result<(String, Ok((String::new(), String::new())) } - PrivRequest::ReloadGatewayNginx => sync_gateway_nginx().await, + PrivRequest::ReloadGatewayNginx => { + let out = Command::new("systemd-run") + .args([ + "--machine=hive-gateway", + "--quiet", + "--", + "nginx", + "-s", + "reload", + ]) + .output() + .await + .context("invoke systemd-run for gateway nginx reload")?; + if !out.status.success() { + bail!( + "gateway nginx reload failed ({}): {}", + out.status, + String::from_utf8_lossy(&out.stderr).trim() + ); + } + Ok((String::new(), String::new())) + } PrivRequest::ChownSocketDir { ref agent_name, @@ -560,92 +581,6 @@ async fn read_container_journal( Ok((stdout, stderr)) } -/// Synchronise the nginx unit inside the `hive-gateway` container. -/// -/// Queries `ActiveState` via `systemctl --machine=hive-gateway` (requires -/// root — machine-bus transport enters the container namespace), then -/// dispatches: -/// - `active` → `systemctl reload nginx` (SIGHUP, zero-downtime) -/// - `failed` → `systemctl reset-failed nginx` + `systemctl start nginx` -/// - otherwise → `systemctl start nginx` -/// -/// Returns `(String::new(), String::new())` on success so it fits the -/// `exec` return type directly. -async fn sync_gateway_nginx() -> Result<(String, String)> { - let state_out = Command::new("systemctl") - .args([ - "--machine=hive-gateway", - "show", - "--property=ActiveState", - "--value", - "nginx", - ]) - .output() - .await - .context("query nginx ActiveState in hive-gateway")?; - if !state_out.status.success() { - tracing::warn!( - exit_code = ?state_out.status.code(), - stderr = %String::from_utf8_lossy(&state_out.stderr).trim(), - "systemctl show ActiveState exited non-zero — gateway container may be down" - ); - } - let state = String::from_utf8_lossy(&state_out.stdout).trim().to_owned(); - // State-aware dispatch: reload when running; reset+start after - // start-limit failure; plain start when inactive or unknown. - match state.as_str() { - "active" => { - let out = Command::new("systemctl") - .args(["--machine=hive-gateway", "reload", "nginx"]) - .output() - .await - .context("reload nginx in hive-gateway")?; - if !out.status.success() { - bail!( - "gateway nginx reload failed ({}): {}", - out.status, - String::from_utf8_lossy(&out.stderr).trim() - ); - } - } - "failed" => { - // Clear start-limit so the next start can proceed. - let _ = Command::new("systemctl") - .args(["--machine=hive-gateway", "reset-failed", "nginx"]) - .status() - .await; - let out = Command::new("systemctl") - .args(["--machine=hive-gateway", "start", "nginx"]) - .output() - .await - .context("start nginx after reset-failed in hive-gateway")?; - if !out.status.success() { - bail!( - "gateway nginx start (after reset-failed) failed ({}): {}", - out.status, - String::from_utf8_lossy(&out.stderr).trim() - ); - } - } - _ => { - // inactive, activating, deactivating, unknown — just start. - let out = Command::new("systemctl") - .args(["--machine=hive-gateway", "start", "nginx"]) - .output() - .await - .context("start nginx in hive-gateway")?; - if !out.status.success() { - bail!( - "gateway nginx start failed (state={state:?}) ({}): {}", - out.status, - String::from_utf8_lossy(&out.stderr).trim() - ); - } - } - } - Ok((String::new(), String::new())) -} - /// Return the system container name for a logical agent name. /// All agents (including the manager) use the `h-` prefix. fn container_system_name(name: &str) -> String { diff --git a/hive-sh4re/src/priv_proto.rs b/hive-sh4re/src/priv_proto.rs index 288d53c1..3638084c 100644 --- a/hive-sh4re/src/priv_proto.rs +++ b/hive-sh4re/src/priv_proto.rs @@ -182,13 +182,8 @@ pub enum PrivRequest { /// Run `systemctl daemon-reload`. DaemonReload, - /// Synchronise the nginx unit inside the `hive-gateway` container. - /// hive-priv queries `ActiveState` and dispatches: - /// - `active` → `systemctl reload nginx` (SIGHUP, zero-downtime) - /// - `failed` → `systemctl reset-failed nginx` + `systemctl start nginx` - /// - otherwise → `systemctl start nginx` - /// Requires root: `--machine=hive-gateway` enters the container - /// namespace via the machine bus (forbidden for unprivileged users). + /// Reload nginx inside the `hive-gateway` container via + /// `systemd-run --machine=hive-gateway nginx -s reload`. ReloadGatewayNginx, // --- Socket dir ownership ---