gateway: rate-limit nginx reload retries after failure
RELOAD_PENDING stays true on any failed reload, and reload_if_pending fires on every 10-second spawn_poll tick. When the gateway is down or nginx config is bad this hammers systemctl indefinitely. Fix: track LAST_FAILED_RELOAD (unix timestamp). reload_if_pending backs off to one attempt per RELOAD_RETRY_SECS (30s) after a failure. Fresh write() calls reset the backoff so new topology changes still attempt reload immediately. Fixes #1113.
This commit is contained in:
parent
e54c1b84d7
commit
ee0ffa64f8
1 changed files with 40 additions and 3 deletions
|
|
@ -8,7 +8,8 @@
|
|||
use anyhow::{Context, Result};
|
||||
use std::fmt::Write as _;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::agent_sockets;
|
||||
use crate::lifecycle;
|
||||
|
|
@ -20,6 +21,15 @@ use crate::lifecycle;
|
|||
/// systemd-run not found) without re-writing the already-correct file.
|
||||
static RELOAD_PENDING: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
/// Unix timestamp (seconds) of the last failed reload attempt.
|
||||
/// `reload_if_pending` backs off to once per `RELOAD_RETRY_SECS` after a
|
||||
/// failure so a permanently broken gateway (bad config, container down)
|
||||
/// doesn't hammer `systemctl` on every 10-second `spawn_poll` tick.
|
||||
static LAST_FAILED_RELOAD: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
/// Minimum gap between retry attempts after a reload failure (30 s).
|
||||
const RELOAD_RETRY_SECS: u64 = 30;
|
||||
|
||||
const HOST_CONF_PATH: &str = "/var/lib/hyperhive/gateway/agents.conf";
|
||||
|
||||
/// Host-side path where c0re writes the generated nginx include file.
|
||||
|
|
@ -201,6 +211,9 @@ pub fn write(names: &[String]) -> Result<()> {
|
|||
path.display()
|
||||
)
|
||||
})?;
|
||||
// Reset backoff so a fresh topology change gets an immediate attempt,
|
||||
// not a stale cooldown from a previous failure.
|
||||
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
|
||||
// Mark reload pending before attempting so a failed attempt is
|
||||
// retried by the next spawn_poll tick (see `reload_if_pending`).
|
||||
RELOAD_PENDING.store(true, Ordering::Relaxed);
|
||||
|
|
@ -212,10 +225,27 @@ pub fn write(names: &[String]) -> Result<()> {
|
|||
/// Called by `spawn_poll` on each tick so a transient failure
|
||||
/// (gateway container temporarily down, systemd-run error) is
|
||||
/// recovered automatically without requiring a new file write.
|
||||
///
|
||||
/// Backs off to one retry per [`RELOAD_RETRY_SECS`] after a failure so a
|
||||
/// permanently broken gateway doesn't hammer `systemctl` on every tick.
|
||||
/// A fresh `write()` call always resets the backoff (new RELOAD_PENDING
|
||||
/// set to `true` + immediate attempt) so topology changes are still
|
||||
/// applied promptly.
|
||||
pub fn reload_if_pending() {
|
||||
if RELOAD_PENDING.load(Ordering::Relaxed) {
|
||||
reload_gateway_nginx();
|
||||
if !RELOAD_PENDING.load(Ordering::Relaxed) {
|
||||
return;
|
||||
}
|
||||
let last_failed = LAST_FAILED_RELOAD.load(Ordering::Relaxed);
|
||||
if last_failed > 0 {
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_secs();
|
||||
if now.saturating_sub(last_failed) < RELOAD_RETRY_SECS {
|
||||
return;
|
||||
}
|
||||
}
|
||||
reload_gateway_nginx();
|
||||
}
|
||||
|
||||
/// Query the nginx unit's `ActiveState` inside the gateway container.
|
||||
|
|
@ -318,6 +348,13 @@ fn reload_gateway_nginx() {
|
|||
|
||||
if success {
|
||||
RELOAD_PENDING.store(false, Ordering::Relaxed);
|
||||
LAST_FAILED_RELOAD.store(0, Ordering::Relaxed);
|
||||
} else {
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_secs();
|
||||
LAST_FAILED_RELOAD.store(now, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue