fix: route gateway nginx control through hive-priv
systemctl --machine=hive-gateway requires root (machine-bus transport enters the container namespace). hive-c0re is unprivileged, so every call to nginx_active_state() and gateway_systemctl() silently failed with exit 1, causing a continuous 30s retry loop without ever syncing nginx. Fix: - Move state-aware nginx logic into hive-priv ReloadGatewayNginx: check ActiveState, then reload/reset-start/start accordingly. hive-priv already runs as root and has machine-bus rights. - Remove nginx_active_state() and gateway_systemctl() from gateway_nginx.rs (they were always running unprivileged, always failing silently). - Make write(), reload_if_pending(), reload_gateway_nginx() async so they can call the async priv_client without a blocking bridge. - Update callers in agent_sockets::spawn_poll and meta::sync_agents to await the now-async functions. The priv_client::reload_gateway_nginx() call and PrivRequest::ReloadGatewayNginx wire type already existed — the gateway_nginx module was just not using them.
This commit is contained in:
parent
7cf7f043ad
commit
1d062d1e3e
4 changed files with 103 additions and 139 deletions
|
|
@ -194,12 +194,12 @@ pub fn spawn_poll() {
|
|||
// selection (UDS vs TCP) depends on .bound markers
|
||||
// which change independently of topology. Write is
|
||||
// idempotent; skips rename when nothing changed.
|
||||
if let Err(e) = crate::gateway_nginx::write(&names) {
|
||||
if let Err(e) = crate::gateway_nginx::write(&names).await {
|
||||
tracing::debug!(error = ?e, "gateway_nginx poll write failed");
|
||||
}
|
||||
// Retry a pending nginx reload that failed on a
|
||||
// previous tick (no-op if no reload is pending).
|
||||
crate::gateway_nginx::reload_if_pending();
|
||||
crate::gateway_nginx::reload_if_pending().await;
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::debug!(error = ?e, "agent_sockets poll: failed to list agents");
|
||||
|
|
|
|||
|
|
@ -11,6 +11,8 @@ use std::path::PathBuf;
|
|||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::priv_client;
|
||||
|
||||
use crate::agent_sockets;
|
||||
use crate::lifecycle;
|
||||
|
||||
|
|
@ -176,20 +178,20 @@ fn render(names: &[String], frontend_dir: Option<&str>) -> String {
|
|||
/// body matches what's already on disk (idempotent; avoids spurious
|
||||
/// gateway reloads on a quiet tick).
|
||||
///
|
||||
/// After a successful write, triggers an nginx reload inside the
|
||||
/// gateway container from the HOST side via
|
||||
/// `systemd-run --machine=hive-gateway nginx -s reload`. This is
|
||||
/// intentionally host-side rather than relying on a systemd path unit
|
||||
/// inside the container watching the bind-mounted file: `IN_MOVED_TO`
|
||||
/// (fired by the atomic rename) does not reliably propagate across the
|
||||
/// nspawn mount-namespace boundary, so the path-unit approach was
|
||||
/// silently broken (see `docs/gateway.md` for the failure analysis).
|
||||
/// After a successful write, triggers the appropriate nginx action inside
|
||||
/// the gateway container via `hive-priv` (which has the
|
||||
/// `--machine=hive-gateway` transport rights hive-c0re lacks):
|
||||
/// reload when nginx is active, reset-failed+start when in a failed
|
||||
/// state, plain start otherwise. This is intentionally host-side rather
|
||||
/// than relying on a systemd path unit inside the container watching the
|
||||
/// bind-mounted file: `IN_MOVED_TO` (fired by the atomic rename) does
|
||||
/// not reliably propagate across the nspawn mount-namespace boundary, so
|
||||
/// the path-unit approach was silently broken (see `docs/gateway.md`).
|
||||
///
|
||||
/// The `systemd-run` call is best-effort — a failed reload is logged
|
||||
/// but not fatal. nginx will pick up the new include on its next
|
||||
/// housekeeping restart or the next manual reload; the host's agent
|
||||
/// topology has already been written correctly.
|
||||
pub fn write(names: &[String]) -> Result<()> {
|
||||
/// The priv call is best-effort — a failed sync is logged but not fatal.
|
||||
/// `reload_if_pending` retries on the next `spawn_poll` tick so a
|
||||
/// transient gateway-down situation converges without manual intervention.
|
||||
pub async fn write(names: &[String]) -> Result<()> {
|
||||
let frontend_dir = std::env::var("HIVE_AGENT_FRONTEND_DIR")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty());
|
||||
|
|
@ -217,7 +219,7 @@ pub fn write(names: &[String]) -> Result<()> {
|
|||
// Mark reload pending before attempting so a failed attempt is
|
||||
// retried by the next spawn_poll tick (see `reload_if_pending`).
|
||||
RELOAD_PENDING.store(true, Ordering::Relaxed);
|
||||
reload_gateway_nginx();
|
||||
reload_gateway_nginx().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
|
@ -231,7 +233,7 @@ pub fn write(names: &[String]) -> Result<()> {
|
|||
/// A fresh `write()` call always resets the backoff (new RELOAD_PENDING
|
||||
/// set to `true` + immediate attempt) so topology changes are still
|
||||
/// applied promptly.
|
||||
pub fn reload_if_pending() {
|
||||
pub async fn reload_if_pending() {
|
||||
if !RELOAD_PENDING.load(Ordering::Relaxed) {
|
||||
return;
|
||||
}
|
||||
|
|
@ -245,116 +247,32 @@ pub fn reload_if_pending() {
|
|||
return;
|
||||
}
|
||||
}
|
||||
reload_gateway_nginx();
|
||||
reload_gateway_nginx().await;
|
||||
}
|
||||
|
||||
/// Query the nginx unit's `ActiveState` inside the gateway container.
|
||||
/// Returns the raw state string from `systemctl show --property=ActiveState
|
||||
/// --value` (e.g. `"active"`, `"failed"`, `"inactive"`, `"activating"`).
|
||||
/// Returns `"unknown"` on any error so callers can branch safely.
|
||||
fn nginx_active_state() -> String {
|
||||
let out = std::process::Command::new("systemctl")
|
||||
.args([
|
||||
"--machine=hive-gateway",
|
||||
"show",
|
||||
"--property=ActiveState",
|
||||
"--value",
|
||||
"nginx",
|
||||
])
|
||||
.output();
|
||||
match out {
|
||||
Ok(o) if o.status.success() => String::from_utf8_lossy(&o.stdout).trim().to_owned(),
|
||||
Ok(o) => {
|
||||
tracing::warn!(
|
||||
exit_code = ?o.status.code(),
|
||||
"systemctl show ActiveState exited non-zero"
|
||||
);
|
||||
"unknown".to_owned()
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(error = %e, "systemctl show ActiveState failed");
|
||||
"unknown".to_owned()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Send a `systemctl --machine=hive-gateway <args...>` command and
|
||||
/// return whether it succeeded. Best-effort: errors are logged.
|
||||
fn gateway_systemctl(args: &[&str]) -> bool {
|
||||
let mut cmd = std::process::Command::new("systemctl");
|
||||
cmd.arg("--machine=hive-gateway");
|
||||
cmd.args(args);
|
||||
match cmd.status() {
|
||||
Ok(s) if s.success() => true,
|
||||
Ok(s) => {
|
||||
tracing::warn!(
|
||||
args = ?args,
|
||||
exit_code = ?s.code(),
|
||||
"gateway systemctl exited non-zero"
|
||||
);
|
||||
false
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(args = ?args, error = %e, "gateway systemctl invocation failed");
|
||||
false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Synchronise the gateway nginx unit with the current agents.conf:
|
||||
///
|
||||
/// - **active**: send `nginx -s reload` (SIGHUP to master, zero-downtime
|
||||
/// worker replacement). Keeps `RELOAD_PENDING` set on failure so the
|
||||
/// next poll tick retries.
|
||||
/// - **failed / start-limit-hit**: run `systemctl reset-failed nginx`
|
||||
/// then `systemctl start nginx`. This is the self-healing path: a
|
||||
/// transient bad agents.conf that causes five instant `nginx -t`
|
||||
/// failures trips systemd's start-limit. Once c0re publishes a correct
|
||||
/// config, the next reload attempt clears the failure and restarts.
|
||||
/// - **inactive / other**: run `systemctl start nginx` directly (no
|
||||
/// reset-failed needed when the unit isn't in a failed state).
|
||||
/// Synchronise the gateway nginx unit with the current agents.conf via
|
||||
/// `hive-priv` (privileged helper). The state-aware logic (active →
|
||||
/// reload; failed → reset-failed + start; inactive/unknown → start)
|
||||
/// runs inside hive-priv where it has the `--machine=hive-gateway`
|
||||
/// transport rights that hive-c0re (unprivileged) lacks.
|
||||
///
|
||||
/// `RELOAD_PENDING` is cleared only after a successful operation so
|
||||
/// `reload_if_pending` keeps retrying on failure.
|
||||
fn reload_gateway_nginx() {
|
||||
let state = nginx_active_state();
|
||||
let success = match state.as_str() {
|
||||
"active" => {
|
||||
// nginx master is running — ask systemd to reload the unit
|
||||
// (SIGHUP to master, zero-downtime worker replacement).
|
||||
// `systemctl -M hive-gateway reload nginx` lets systemd
|
||||
// resolve the binary path; avoids the exit-203 (EXEC)
|
||||
// failure that `systemd-run -- nginx` hit on NixOS where
|
||||
// the limited transient-unit PATH misses /run/current-system/sw/bin/.
|
||||
let ok = gateway_systemctl(&["reload", "nginx"]);
|
||||
if ok {
|
||||
tracing::debug!("gateway nginx reload signal sent");
|
||||
}
|
||||
ok
|
||||
async fn reload_gateway_nginx() {
|
||||
match priv_client::reload_gateway_nginx().await {
|
||||
Ok(()) => {
|
||||
tracing::debug!("gateway nginx sync succeeded");
|
||||
RELOAD_PENDING.store(false, Ordering::Relaxed);
|
||||
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
|
||||
}
|
||||
"failed" => {
|
||||
// Unit hit start-limit (e.g. repeated nginx -t failures from
|
||||
// a bad agents.conf). reset-failed clears the rate-limit so
|
||||
// start can proceed.
|
||||
tracing::info!("gateway nginx unit in failed state — resetting and starting");
|
||||
gateway_systemctl(&["reset-failed", "nginx"]) && gateway_systemctl(&["start", "nginx"])
|
||||
Err(e) => {
|
||||
tracing::warn!(error = %e, "gateway nginx sync failed — will retry");
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_secs();
|
||||
LAST_FAILED_RELOAD.store(now, Ordering::Relaxed);
|
||||
}
|
||||
other => {
|
||||
// inactive, deactivating, activating, unknown — just try start.
|
||||
tracing::info!(state = other, "gateway nginx unit not active — starting");
|
||||
gateway_systemctl(&["start", "nginx"])
|
||||
}
|
||||
};
|
||||
|
||||
if success {
|
||||
RELOAD_PENDING.store(false, Ordering::Relaxed);
|
||||
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
|
||||
} else {
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_secs();
|
||||
LAST_FAILED_RELOAD.store(now, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -126,13 +126,11 @@ pub async fn sync_agents(hive: &HiveEnv, agents: &[AgentSpec]) -> Result<()> {
|
|||
tracing::warn!(error = ?e, "agent_sockets::write failed (non-fatal)");
|
||||
}
|
||||
|
||||
// Refresh /var/lib/hyperhive/agents.conf — the nginx include file
|
||||
// the gateway picks up at runtime without needing a
|
||||
// nixos-rebuild. The gateway container bind-mounts
|
||||
// /var/lib/hyperhive/ and a systemd path unit fires
|
||||
// `nginx -s reload` when this file changes. Same
|
||||
// best-effort + non-fatal shape.
|
||||
if let Err(e) = crate::gateway_nginx::write(&agent_names) {
|
||||
// Refresh /var/lib/hyperhive/gateway/agents.conf — the nginx include
|
||||
// file the gateway container bind-mounts and nginx reads at runtime.
|
||||
// c0re triggers a reload (or start) inside hive-gateway via hive-priv
|
||||
// after writing the file. Same best-effort + non-fatal shape.
|
||||
if let Err(e) = crate::gateway_nginx::write(&agent_names).await {
|
||||
tracing::warn!(error = ?e, "gateway_nginx::write failed (non-fatal)");
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -289,24 +289,72 @@ async fn exec(req: PrivRequest, writer: &mut OwnedWriteHalf) -> Result<(String,
|
|||
}
|
||||
|
||||
PrivRequest::ReloadGatewayNginx => {
|
||||
let out = Command::new("systemd-run")
|
||||
// Query the nginx unit's ActiveState inside the gateway container.
|
||||
// Requires root: --machine= transport enters the container namespace
|
||||
// via the machine bus, which is forbidden for unprivileged users.
|
||||
let state_out = Command::new("systemctl")
|
||||
.args([
|
||||
"--machine=hive-gateway",
|
||||
"--quiet",
|
||||
"--",
|
||||
"show",
|
||||
"--property=ActiveState",
|
||||
"--value",
|
||||
"nginx",
|
||||
"-s",
|
||||
"reload",
|
||||
])
|
||||
.output()
|
||||
.await
|
||||
.context("invoke systemd-run for gateway nginx reload")?;
|
||||
if !out.status.success() {
|
||||
bail!(
|
||||
"gateway nginx reload failed ({}): {}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
.context("query nginx ActiveState in hive-gateway")?;
|
||||
let state = String::from_utf8_lossy(&state_out.stdout).trim().to_owned();
|
||||
// State-aware action: reload when running; reset+start after
|
||||
// start-limit failure; plain start when inactive or unknown.
|
||||
match state.as_str() {
|
||||
"active" => {
|
||||
let out = Command::new("systemctl")
|
||||
.args(["--machine=hive-gateway", "reload", "nginx"])
|
||||
.output()
|
||||
.await
|
||||
.context("reload nginx in hive-gateway")?;
|
||||
if !out.status.success() {
|
||||
bail!(
|
||||
"gateway nginx reload failed ({}): {}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
}
|
||||
}
|
||||
"failed" => {
|
||||
// Clear start-limit hit so the next start can proceed.
|
||||
let _ = Command::new("systemctl")
|
||||
.args(["--machine=hive-gateway", "reset-failed", "nginx"])
|
||||
.status()
|
||||
.await;
|
||||
let out = Command::new("systemctl")
|
||||
.args(["--machine=hive-gateway", "start", "nginx"])
|
||||
.output()
|
||||
.await
|
||||
.context("start nginx after reset-failed in hive-gateway")?;
|
||||
if !out.status.success() {
|
||||
bail!(
|
||||
"gateway nginx start (after reset-failed) failed ({}): {}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
// inactive, activating, deactivating, unknown — just start.
|
||||
let out = Command::new("systemctl")
|
||||
.args(["--machine=hive-gateway", "start", "nginx"])
|
||||
.output()
|
||||
.await
|
||||
.context("start nginx in hive-gateway")?;
|
||||
if !out.status.success() {
|
||||
bail!(
|
||||
"gateway nginx start failed (state={state}) ({}): {}",
|
||||
out.status,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok((String::new(), String::new()))
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in a new issue