fix: route gateway nginx control through hive-priv

systemctl --machine=hive-gateway requires root (machine-bus transport
enters the container namespace). hive-c0re is unprivileged, so every
call to nginx_active_state() and gateway_systemctl() silently failed
with exit 1, causing a continuous 30s retry loop without ever
syncing nginx.

Fix:
- Move state-aware nginx logic into hive-priv ReloadGatewayNginx:
  check ActiveState, then reload/reset-start/start accordingly.
  hive-priv already runs as root and has machine-bus rights.
- Remove nginx_active_state() and gateway_systemctl() from
  gateway_nginx.rs (they were always running unprivileged, always
  failing silently).
- Make write(), reload_if_pending(), reload_gateway_nginx() async so
  they can call the async priv_client without a blocking bridge.
- Update callers in agent_sockets::spawn_poll and meta::sync_agents
  to await the now-async functions.

The priv_client::reload_gateway_nginx() call and PrivRequest::ReloadGatewayNginx
wire type already existed — the gateway_nginx module was just not using them.
This commit is contained in:
atlas 2026-06-04 09:48:56 +02:00
commit 1d062d1e3e
4 changed files with 103 additions and 139 deletions

View file

@ -194,12 +194,12 @@ pub fn spawn_poll() {
// selection (UDS vs TCP) depends on .bound markers
// which change independently of topology. Write is
// idempotent; skips rename when nothing changed.
if let Err(e) = crate::gateway_nginx::write(&names) {
if let Err(e) = crate::gateway_nginx::write(&names).await {
tracing::debug!(error = ?e, "gateway_nginx poll write failed");
}
// Retry a pending nginx reload that failed on a
// previous tick (no-op if no reload is pending).
crate::gateway_nginx::reload_if_pending();
crate::gateway_nginx::reload_if_pending().await;
}
Err(e) => {
tracing::debug!(error = ?e, "agent_sockets poll: failed to list agents");

View file

@ -11,6 +11,8 @@ use std::path::PathBuf;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
use std::time::{SystemTime, UNIX_EPOCH};
use crate::priv_client;
use crate::agent_sockets;
use crate::lifecycle;
@ -176,20 +178,20 @@ fn render(names: &[String], frontend_dir: Option<&str>) -> String {
/// body matches what's already on disk (idempotent; avoids spurious
/// gateway reloads on a quiet tick).
///
/// After a successful write, triggers an nginx reload inside the
/// gateway container from the HOST side via
/// `systemd-run --machine=hive-gateway nginx -s reload`. This is
/// intentionally host-side rather than relying on a systemd path unit
/// inside the container watching the bind-mounted file: `IN_MOVED_TO`
/// (fired by the atomic rename) does not reliably propagate across the
/// nspawn mount-namespace boundary, so the path-unit approach was
/// silently broken (see `docs/gateway.md` for the failure analysis).
/// After a successful write, triggers the appropriate nginx action inside
/// the gateway container via `hive-priv` (which has the
/// `--machine=hive-gateway` transport rights hive-c0re lacks):
/// reload when nginx is active, reset-failed+start when in a failed
/// state, plain start otherwise. This is intentionally host-side rather
/// than relying on a systemd path unit inside the container watching the
/// bind-mounted file: `IN_MOVED_TO` (fired by the atomic rename) does
/// not reliably propagate across the nspawn mount-namespace boundary, so
/// the path-unit approach was silently broken (see `docs/gateway.md`).
///
/// The `systemd-run` call is best-effort — a failed reload is logged
/// but not fatal. nginx will pick up the new include on its next
/// housekeeping restart or the next manual reload; the host's agent
/// topology has already been written correctly.
pub fn write(names: &[String]) -> Result<()> {
/// The priv call is best-effort — a failed sync is logged but not fatal.
/// `reload_if_pending` retries on the next `spawn_poll` tick so a
/// transient gateway-down situation converges without manual intervention.
pub async fn write(names: &[String]) -> Result<()> {
let frontend_dir = std::env::var("HIVE_AGENT_FRONTEND_DIR")
.ok()
.filter(|s| !s.is_empty());
@ -217,7 +219,7 @@ pub fn write(names: &[String]) -> Result<()> {
// Mark reload pending before attempting so a failed attempt is
// retried by the next spawn_poll tick (see `reload_if_pending`).
RELOAD_PENDING.store(true, Ordering::Relaxed);
reload_gateway_nginx();
reload_gateway_nginx().await;
Ok(())
}
@ -231,7 +233,7 @@ pub fn write(names: &[String]) -> Result<()> {
/// A fresh `write()` call always resets the backoff (new RELOAD_PENDING
/// set to `true` + immediate attempt) so topology changes are still
/// applied promptly.
pub fn reload_if_pending() {
pub async fn reload_if_pending() {
if !RELOAD_PENDING.load(Ordering::Relaxed) {
return;
}
@ -245,116 +247,32 @@ pub fn reload_if_pending() {
return;
}
}
reload_gateway_nginx();
reload_gateway_nginx().await;
}
/// Query the nginx unit's `ActiveState` inside the gateway container.
/// Returns the raw state string from `systemctl show --property=ActiveState
/// --value` (e.g. `"active"`, `"failed"`, `"inactive"`, `"activating"`).
/// Returns `"unknown"` on any error so callers can branch safely.
fn nginx_active_state() -> String {
let out = std::process::Command::new("systemctl")
.args([
"--machine=hive-gateway",
"show",
"--property=ActiveState",
"--value",
"nginx",
])
.output();
match out {
Ok(o) if o.status.success() => String::from_utf8_lossy(&o.stdout).trim().to_owned(),
Ok(o) => {
tracing::warn!(
exit_code = ?o.status.code(),
"systemctl show ActiveState exited non-zero"
);
"unknown".to_owned()
}
Err(e) => {
tracing::warn!(error = %e, "systemctl show ActiveState failed");
"unknown".to_owned()
}
}
}
/// Send a `systemctl --machine=hive-gateway <args...>` command and
/// return whether it succeeded. Best-effort: errors are logged.
fn gateway_systemctl(args: &[&str]) -> bool {
let mut cmd = std::process::Command::new("systemctl");
cmd.arg("--machine=hive-gateway");
cmd.args(args);
match cmd.status() {
Ok(s) if s.success() => true,
Ok(s) => {
tracing::warn!(
args = ?args,
exit_code = ?s.code(),
"gateway systemctl exited non-zero"
);
false
}
Err(e) => {
tracing::warn!(args = ?args, error = %e, "gateway systemctl invocation failed");
false
}
}
}
/// Synchronise the gateway nginx unit with the current agents.conf:
///
/// - **active**: send `nginx -s reload` (SIGHUP to master, zero-downtime
/// worker replacement). Keeps `RELOAD_PENDING` set on failure so the
/// next poll tick retries.
/// - **failed / start-limit-hit**: run `systemctl reset-failed nginx`
/// then `systemctl start nginx`. This is the self-healing path: a
/// transient bad agents.conf that causes five instant `nginx -t`
/// failures trips systemd's start-limit. Once c0re publishes a correct
/// config, the next reload attempt clears the failure and restarts.
/// - **inactive / other**: run `systemctl start nginx` directly (no
/// reset-failed needed when the unit isn't in a failed state).
/// Synchronise the gateway nginx unit with the current agents.conf via
/// `hive-priv` (privileged helper). The state-aware logic (active →
/// reload; failed → reset-failed + start; inactive/unknown → start)
/// runs inside hive-priv where it has the `--machine=hive-gateway`
/// transport rights that hive-c0re (unprivileged) lacks.
///
/// `RELOAD_PENDING` is cleared only after a successful operation so
/// `reload_if_pending` keeps retrying on failure.
fn reload_gateway_nginx() {
let state = nginx_active_state();
let success = match state.as_str() {
"active" => {
// nginx master is running — ask systemd to reload the unit
// (SIGHUP to master, zero-downtime worker replacement).
// `systemctl -M hive-gateway reload nginx` lets systemd
// resolve the binary path; avoids the exit-203 (EXEC)
// failure that `systemd-run -- nginx` hit on NixOS where
// the limited transient-unit PATH misses /run/current-system/sw/bin/.
let ok = gateway_systemctl(&["reload", "nginx"]);
if ok {
tracing::debug!("gateway nginx reload signal sent");
}
ok
async fn reload_gateway_nginx() {
match priv_client::reload_gateway_nginx().await {
Ok(()) => {
tracing::debug!("gateway nginx sync succeeded");
RELOAD_PENDING.store(false, Ordering::Relaxed);
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
}
"failed" => {
// Unit hit start-limit (e.g. repeated nginx -t failures from
// a bad agents.conf). reset-failed clears the rate-limit so
// start can proceed.
tracing::info!("gateway nginx unit in failed state — resetting and starting");
gateway_systemctl(&["reset-failed", "nginx"]) && gateway_systemctl(&["start", "nginx"])
Err(e) => {
tracing::warn!(error = %e, "gateway nginx sync failed — will retry");
let now = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
LAST_FAILED_RELOAD.store(now, Ordering::Relaxed);
}
other => {
// inactive, deactivating, activating, unknown — just try start.
tracing::info!(state = other, "gateway nginx unit not active — starting");
gateway_systemctl(&["start", "nginx"])
}
};
if success {
RELOAD_PENDING.store(false, Ordering::Relaxed);
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
} else {
let now = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
LAST_FAILED_RELOAD.store(now, Ordering::Relaxed);
}
}

View file

@ -126,13 +126,11 @@ pub async fn sync_agents(hive: &HiveEnv, agents: &[AgentSpec]) -> Result<()> {
tracing::warn!(error = ?e, "agent_sockets::write failed (non-fatal)");
}
// Refresh /var/lib/hyperhive/agents.conf — the nginx include file
// the gateway picks up at runtime without needing a
// nixos-rebuild. The gateway container bind-mounts
// /var/lib/hyperhive/ and a systemd path unit fires
// `nginx -s reload` when this file changes. Same
// best-effort + non-fatal shape.
if let Err(e) = crate::gateway_nginx::write(&agent_names) {
// Refresh /var/lib/hyperhive/gateway/agents.conf — the nginx include
// file the gateway container bind-mounts and nginx reads at runtime.
// c0re triggers a reload (or start) inside hive-gateway via hive-priv
// after writing the file. Same best-effort + non-fatal shape.
if let Err(e) = crate::gateway_nginx::write(&agent_names).await {
tracing::warn!(error = ?e, "gateway_nginx::write failed (non-fatal)");
}

View file

@ -289,24 +289,72 @@ async fn exec(req: PrivRequest, writer: &mut OwnedWriteHalf) -> Result<(String,
}
PrivRequest::ReloadGatewayNginx => {
let out = Command::new("systemd-run")
// Query the nginx unit's ActiveState inside the gateway container.
// Requires root: --machine= transport enters the container namespace
// via the machine bus, which is forbidden for unprivileged users.
let state_out = Command::new("systemctl")
.args([
"--machine=hive-gateway",
"--quiet",
"--",
"show",
"--property=ActiveState",
"--value",
"nginx",
"-s",
"reload",
])
.output()
.await
.context("invoke systemd-run for gateway nginx reload")?;
if !out.status.success() {
bail!(
"gateway nginx reload failed ({}): {}",
out.status,
String::from_utf8_lossy(&out.stderr).trim()
);
.context("query nginx ActiveState in hive-gateway")?;
let state = String::from_utf8_lossy(&state_out.stdout).trim().to_owned();
// State-aware action: reload when running; reset+start after
// start-limit failure; plain start when inactive or unknown.
match state.as_str() {
"active" => {
let out = Command::new("systemctl")
.args(["--machine=hive-gateway", "reload", "nginx"])
.output()
.await
.context("reload nginx in hive-gateway")?;
if !out.status.success() {
bail!(
"gateway nginx reload failed ({}): {}",
out.status,
String::from_utf8_lossy(&out.stderr).trim()
);
}
}
"failed" => {
// Clear start-limit hit so the next start can proceed.
let _ = Command::new("systemctl")
.args(["--machine=hive-gateway", "reset-failed", "nginx"])
.status()
.await;
let out = Command::new("systemctl")
.args(["--machine=hive-gateway", "start", "nginx"])
.output()
.await
.context("start nginx after reset-failed in hive-gateway")?;
if !out.status.success() {
bail!(
"gateway nginx start (after reset-failed) failed ({}): {}",
out.status,
String::from_utf8_lossy(&out.stderr).trim()
);
}
}
_ => {
// inactive, activating, deactivating, unknown — just start.
let out = Command::new("systemctl")
.args(["--machine=hive-gateway", "start", "nginx"])
.output()
.await
.context("start nginx in hive-gateway")?;
if !out.status.success() {
bail!(
"gateway nginx start failed (state={state}) ({}): {}",
out.status,
String::from_utf8_lossy(&out.stderr).trim()
);
}
}
}
Ok((String::new(), String::new()))
}