From b3b42d327979c18240734f52786ecf7e44c6bc29 Mon Sep 17 00:00:00 2001 From: atlas Date: Tue, 29 Sep 2026 23:15:29 +0200 Subject: [PATCH] credential units: 24h retry shape; start a failed nginx when the cert lands Six credential-fetch units retried 4 times at 15s, so an apply during which the store or gateway was down for more than about a minute left them in start-limit-hit, and nothing started them again once the store came back. The swarm-services leaf could also land after nginx had already given up on it, and the hook that propagates a new leaf only reloaded a running nginx, so a stopped one stayed down until a second apply. - nix/host-modules/lib/store-retry.nix: the 2880 x 30s / 25h window shape swarm-services-cert already had, as one attrset. - swarm-services-cert, swarm-bao-otel-oidc, swarm-bao-forwarder-oidc, swarm-bao-matrix-token, swarm-bao-queue-agent, swarm-bao-grafana-oidc, hive-agent-bao-identity and hive-agent-forge-token use it. queue-identity.nix no longer has a fetch unit (ccb5bd3b), and forge-token.nix is a fetch unit with the same short budget that was added after the census in #4662. - The swarm-services-cert propagation hook now reset-fails and starts (--no-block) a loaded nginx that is not active; an active nginx keeps the re-import + reload. - module-eval-bao-grants: one case pinning the shape on every host-side fetch unit, swarm-services-cert included. Refs #4662 --- nix/agent-modules/bao.nix | 19 ++++----- nix/agent-modules/forge-token.nix | 10 ++--- nix/host-modules/glue-matrix-bao-token.nix | 35 ++++++---------- .../glue-queue-agent-credential.nix | 32 +++++---------- nix/host-modules/hive-tls.nix | 40 +++++++++---------- nix/host-modules/lib/store-retry.nix | 32 +++++++++++++++ nix/host-modules/swarm-bao.nix | 16 +++----- nix/host-modules/swarm-grafana.nix | 25 ++++-------- nix/host-modules/swarm-otel.nix | 25 ++++-------- nix/module-eval/bao-grants.nix | 28 +++++++++++++ 10 files changed, 137 insertions(+), 125 deletions(-) create mode 100644 nix/host-modules/lib/store-retry.nix diff --git a/nix/agent-modules/bao.nix b/nix/agent-modules/bao.nix index f484f87d..e74106e5 100644 --- a/nix/agent-modules/bao.nix +++ b/nix/agent-modules/bao.nix @@ -61,6 +61,8 @@ let # its default would decide whether the hive-side courier delivers into a # container that reads what it is given or into one that never looks. configured = cfg.addr != null; + + storeRetry = import ../host-modules/lib/store-retry.nix { }; in { options.services.hyperhive.agent.bao = { @@ -96,21 +98,14 @@ in pkgs.openbao pkgs.coreutils ]; - # Sized for a store that comes up around the same time this container - # does, not for one that is sealed: a few short attempts cover the race, - # and a longer window would only delay the report of a real failure. - # - # `StartLimit*` are `[Unit]` settings, so they go here and not in - # `serviceConfig` — systemd ignores them under `[Service]`. The window - # has to exceed `RestartSec` times the burst. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ../host-modules/lib/store-retry.nix. From an agent the store is reached + # through the gateway's stream passthrough, so either being down fails + # the login below. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; User = agentName; Group = agentName; # Bare ids, no paths: the terse `LoadCredential=` form that inherits a diff --git a/nix/agent-modules/forge-token.nix b/nix/agent-modules/forge-token.nix index 6268d0e5..c5f7d453 100644 --- a/nix/agent-modules/forge-token.nix +++ b/nix/agent-modules/forge-token.nix @@ -49,6 +49,8 @@ let # The store's address is the whole switch, as in ./bao.nix and # ./queue-identity.nix. configured = cfg.addr != null; + + storeRetry = import ../host-modules/lib/store-retry.nix { }; in { options.services.hyperhive.agent.forge.tokenFile = lib.mkOption { @@ -90,9 +92,9 @@ in pkgs.coreutils pkgs.diffutils ]; - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ../host-modules/lib/store-retry.nix. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; # Not `RemainAfterExit`: the timer below has to be able to start this # unit again, and an active unit cannot be started. @@ -100,8 +102,6 @@ in # in it, alive between runs instead. RemainAfterExit = false; TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; User = agentName; Group = agentName; RuntimeDirectory = runtimeDir; diff --git a/nix/host-modules/glue-matrix-bao-token.nix b/nix/host-modules/glue-matrix-bao-token.nix index 0bb4e716..f4b38d0c 100644 --- a/nix/host-modules/glue-matrix-bao-token.nix +++ b/nix/host-modules/glue-matrix-bao-token.nix @@ -82,6 +82,7 @@ let matrixMachine = "hive-matrix"; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + storeRetry = import ./lib/store-retry.nix { }; in { config = lib.mkMerge [ @@ -151,30 +152,18 @@ in deployCfg.bao.package pkgs.coreutils ]; - # Sized for the race this loses, not for an unseal. `swarm-bao` comes up - # seconds before this unit asks, and the cert-auth role it logs in - # against is written seconds after — so a few short attempts cover it. - # ⚠️ `swarm-bao-controller-policy`'s 2880 × 30s is NOT the model to copy. - # That unit blocks nothing; this one is `Before=` the homeserver's - # container, and whether that ordering waits across an auto-restart is - # unverified — so an hours-long window would be a bet on an unknown, - # where a minute is not. A store still sealed after it keeps the degrade - # below, as today. - # - # `StartLimit*` are `[Unit]` settings, so they go here and not in - # `serviceConfig` — systemd ignores them under `[Service]`. The window - # has to exceed `RestartSec × burst`. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ./lib/store-retry.nix. ⚠️ This unit is `Before=` the homeserver's + # container, and an auto-restarting unit is still `activating`, so the + # homeserver waits for as long as the login below keeps failing — up to + # the full 24h on a store that stays sealed. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; - # What actually bounds the read below. Stated here rather than + # What actually bounds each attempt below. Stated here rather than # left to systemd's default, so the number a boot waits on is in # the file that waits. TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; }; environment = { BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}"; @@ -193,9 +182,11 @@ in # A sealed or uninitialised store answers on the port and never # answers the read, so "the store is up" is not the same as "the - # store can answer". `TimeoutStartSec` above is the bound; the - # homeserver only `Wants=` this unit, so hitting it degrades to - # keeping the local token rather than holding up the container. + # store can answer". `TimeoutStartSec` above bounds each attempt and + # a timed-out attempt is retried like any other failure; the + # homeserver only `Wants=` this unit, so a retry budget spent + # degrades to keeping the local token rather than failing the + # container. # `bao`'s own message is the only thing separating a missing value # from a refused identity from an unreachable host. This unit's # degraded mode is correct for all three, so it reports which one diff --git a/nix/host-modules/glue-queue-agent-credential.nix b/nix/host-modules/glue-queue-agent-credential.nix index d2a48a89..cab3f3ad 100644 --- a/nix/host-modules/glue-queue-agent-credential.nix +++ b/nix/host-modules/glue-queue-agent-credential.nix @@ -79,6 +79,7 @@ let credentialPath = "secret/swarm/hives/${hyperhiveCfg.hiveName}/queue/agent"; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + storeRetry = import ./lib/store-retry.nix { }; in { options.services.hyperhive.deploy.hive-controller.queue = { @@ -172,10 +173,10 @@ in requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ]; # Ordered before hive-c0re, so no agent container renders ahead of an # attempt at its credential. `Wants=`, not `Requires=`: a store this - # unit can't reach delays hive-c0re's start by its own start-limit - # window (`TimeoutStartSec`, retried up to `startLimitBurst` times - # below) rather than failing it — hive-c0re starts once that window - # elapses, whatever credential is or isn't on disk by then. + # unit can't reach delays hive-c0re's start for as long as this unit + # retries (up to the 24h of ./lib/store-retry.nix) rather than failing + # it — hive-c0re starts once the retries end, whatever credential is + # or isn't on disk by then. before = [ "hive-c0re.service" ]; wantedBy = [ "multi-user.target" @@ -185,26 +186,15 @@ in baoDeploy.package pkgs.coreutils ]; - # Sized for the race this loses, not for an unseal: `swarm-bao` comes up - # seconds before this unit asks, and the cert-auth role it logs in - # against is written seconds after, so a few short attempts cover it. - # An hours-long window would be a bet on a store that is sealed, and the - # degrade below is already correct for that. - # - # `StartLimit*` are `[Unit]` settings, so they go here and not in - # `serviceConfig` — systemd ignores them under `[Service]`. The window - # has to exceed `RestartSec × burst`. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ./lib/store-retry.nix. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; - # What actually bounds the reads below. Stated here rather than left to - # systemd's default, so the number a boot waits on is in the file that - # waits. + # What actually bounds each attempt at the reads below. Stated here + # rather than left to systemd's default, so the number a boot waits on + # is in the file that waits. TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; }; environment = { BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}"; diff --git a/nix/host-modules/hive-tls.nix b/nix/host-modules/hive-tls.nix index cf3a3d38..ce744fb7 100644 --- a/nix/host-modules/hive-tls.nix +++ b/nix/host-modules/hive-tls.nix @@ -299,6 +299,7 @@ let // lib.optionalAttrs (baoDeploy.serverCaFile != null) { BAO_CACERT = baoDeploy.serverCaFile; }; + storeRetry = import ./lib/store-retry.nix { }; in { # Host-side TLS trust root for the self-signed gateway mode. @@ -751,21 +752,14 @@ in ]; wants = [ "container@${baoCfg.machine}.service" ]; path = servicesCertPath; - # Sized like the store's own granting units, and for the same - # reason: under `seal = "shamir"` an operator unseals BY HAND, and - # the login below fails for as long as that takes. 2880 × 30s is - # 24h inside a 25h window — `StartLimit*` are `[Unit]` settings, so - # the window must exceed `RestartSec × burst` or it closes between - # attempts and the burst is never reached. - startLimitBurst = 2880; - startLimitIntervalSec = 90000; - serviceConfig = { + # ./lib/store-retry.nix: the login below fails for as long as the + # store is sealed or down. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; UMask = "0077"; SyslogIdentifier = "swarm-services-cert"; - Restart = "on-failure"; - RestartSec = 30; }; environment = servicesCertEnvironment; script = '' @@ -926,18 +920,18 @@ in mv -f "$svcroot.new" "$svcroot" chmod 0644 "$svcroot" - # Propagation, for the retry and renewal paths only. Ordered before - # both of these, so on a normal boot they have not run yet, - # `is-active` is false, and ordering alone does the work. What this - # covers is the store coming up hours after the gateway did, and - # `swarm-services-cert-renew` rotating the leaf under a running - # gateway: nginx serves a *copy* of the leaf, so re-issuing the - # source changes nothing until the copy is remade. + # Propagation. nginx serves a *copy* of the leaf, so re-issuing the + # source changes nothing until the copy is remade. A running + # gateway (`swarm-services-cert-renew` rotating the leaf) gets the + # copy remade and a reload. A gateway that is not running gets + # started: a start job still waiting on this unit absorbs the + # request, and a start that already ended on this unit's + # `requiredBy` has nothing else that would start it again. # # ⚠️ `--no-block`, and it is not a preference. This unit declares - # `Before=` both of these, so a blocking `systemctl restart` - # enqueues a job that systemd will not start until this unit is - # active — and this unit is not active until its ExecStart + # `Before=` both of these, so a blocking `systemctl restart` or + # `start` enqueues a job that systemd will not start until this + # unit is active — and this unit is not active until its ExecStart # returns, which is waiting on that job. A deadlock, held until # the 24h retry window's `TimeoutStartSec` fires. Queueing the # job and letting it run once we exit is the only ordering that @@ -952,6 +946,10 @@ in echo "swarm-services leaf rotated — re-importing and reloading nginx" systemctl restart --no-block hive-gateway-self-signed-cert.service systemctl reload --no-block nginx.service + elif [ "$(systemctl show -P LoadState nginx.service)" = loaded ]; then + echo "swarm-services leaf landed with nginx not running — starting it" + systemctl reset-failed nginx.service + systemctl start --no-block nginx.service fi # The bundle is assembled by `hive-tls-ca`, which runs BEFORE this diff --git a/nix/host-modules/lib/store-retry.nix b/nix/host-modules/lib/store-retry.nix new file mode 100644 index 00000000..8034a4f4 --- /dev/null +++ b/nix/host-modules/lib/store-retry.nix @@ -0,0 +1,32 @@ +# Retry shape for a oneshot that fetches a credential or certificate from +# the secret store: every 30s for 24h. Under `seal = "shamir"` an operator +# unseals BY HAND, and a fetch fails for as long as that takes, or for as +# long as the store or the gateway in front of it is down. `start-limit-hit` +# does not clear when the store comes back, so a budget shorter than the +# outage leaves the unit failed until something starts it again. +# +# `StartLimit*` are `[Unit]` settings — systemd ignores them under +# `[Service]` — and the window must exceed `RestartSec × burst` or it closes +# between attempts and the burst is never reached: 2880 × 30s is 24h inside +# a 25h window. +# +# ⚠️ A unit in auto-restart is still `activating`, so its start job stays +# queued across attempts: anything ordered `After=` it waits for as long as +# it retries, up to the full 24h. +# +# Pure attrset — NOT a NixOS module. Use from a unit definition: +# +# storeRetry = import ./lib/store-retry.nix { }; +# systemd.services.foo = { +# inherit (storeRetry) startLimitBurst startLimitIntervalSec; +# serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; }; +# }; +{ }: +{ + startLimitBurst = 2880; + startLimitIntervalSec = 90000; + serviceConfig = { + Restart = "on-failure"; + RestartSec = 30; + }; +} diff --git a/nix/host-modules/swarm-bao.nix b/nix/host-modules/swarm-bao.nix index b7726500..104c7789 100644 --- a/nix/host-modules/swarm-bao.nix +++ b/nix/host-modules/swarm-bao.nix @@ -1188,6 +1188,7 @@ let # sharing the netns is what lets the store bind the host's own addresses # rather than a convenience for nginx. privateNetwork = false; + storeRetry = import ./lib/store-retry.nix { }; in { imports = [ @@ -2379,20 +2380,15 @@ in baoDeploy.package pkgs.coreutils ]; - # Sized for the race this loses: the secret is minted by authelia and - # copied in by the publisher, both of which may be a host away and - # neither of which this boot waits on. `StartLimit*` are `[Unit]` - # settings — systemd ignores them under `[Service]` — and the window - # has to exceed `RestartSec × burst`. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ./lib/store-retry.nix: the secret is minted by authelia and copied in + # by the publisher, both of which may be a host away and neither of + # which this boot waits on. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; SyslogIdentifier = "swarm-bao-forwarder-oidc"; TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; }; environment = { BAO_ADDR = "https://${cfg.domain}:${toString cfg.port}"; diff --git a/nix/host-modules/swarm-grafana.nix b/nix/host-modules/swarm-grafana.nix index a0a09561..89ac3314 100644 --- a/nix/host-modules/swarm-grafana.nix +++ b/nix/host-modules/swarm-grafana.nix @@ -243,6 +243,7 @@ let # Shared host netns, like every sibling swarm container: the gateway # reaches this at 127.0.0.1:. privateNetwork = false; + storeRetry = import ./lib/store-retry.nix { }; in { # `enable` moved to `services.hyperhive.deploy.grafana.enable` — see @@ -673,27 +674,17 @@ in baoDeploy.package pkgs.coreutils ]; - # Sized for the race this loses, not for an unseal: `swarm-bao` comes up - # seconds before this unit asks, and the cert-auth role it logs in - # against is written seconds after, so a few short attempts cover it. An - # hours-long window would be a bet on a sealed store, and the degrade - # below is already correct for that. - # - # `StartLimit*` are `[Unit]` settings, so they go here and not in - # `serviceConfig` — systemd ignores them under `[Service]`. The window - # has to exceed `RestartSec × burst`. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ./lib/store-retry.nix. Grafana's container is ordered after this unit, + # so it waits while the login below keeps failing. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; SyslogIdentifier = "swarm-bao-grafana-oidc"; - # What actually bounds the read below. Stated here rather than left to - # systemd's default, so the number a boot waits on is in the file that - # waits. + # What actually bounds each attempt below. Stated here rather than left + # to systemd's default, so the number a boot waits on is in the file + # that waits. TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; }; environment = { BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}"; diff --git a/nix/host-modules/swarm-otel.nix b/nix/host-modules/swarm-otel.nix index c67b1fa1..c68f38ef 100644 --- a/nix/host-modules/swarm-otel.nix +++ b/nix/host-modules/swarm-otel.nix @@ -310,6 +310,7 @@ let # store, without either crossing a network boundary that would need # its own trust material. privateNetwork = false; + storeRetry = import ./lib/store-retry.nix { }; in { # `enable` moved to `services.hyperhive.deploy.swarm-otel.enable` — see ./deploy.nix. @@ -790,27 +791,17 @@ in baoDeploy.package pkgs.coreutils ]; - # Sized for the race this loses, not for an unseal: `swarm-bao` comes up - # seconds before this unit asks, and the cert-auth role it logs in - # against is written seconds after, so a few short attempts cover it. An - # hours-long window would be a bet on a sealed store, and the degrade - # below is already correct for that. - # - # `StartLimit*` are `[Unit]` settings, so they go here and not in - # `serviceConfig` — systemd ignores them under `[Service]`. The window - # has to exceed `RestartSec × burst`. - startLimitBurst = 4; - startLimitIntervalSec = 300; - serviceConfig = { + # ./lib/store-retry.nix. The collector's container is ordered after + # this unit, so it waits while the login below keeps failing. + inherit (storeRetry) startLimitBurst startLimitIntervalSec; + serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; RemainAfterExit = true; SyslogIdentifier = "swarm-bao-otel-oidc"; - # What actually bounds the read below. Stated here rather than left to - # systemd's default, so the number a boot waits on is in the file that - # waits. + # What actually bounds each attempt below. Stated here rather than left + # to systemd's default, so the number a boot waits on is in the file + # that waits. TimeoutStartSec = 30; - Restart = "on-failure"; - RestartSec = 15; }; environment = { BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}"; diff --git a/nix/module-eval/bao-grants.nix b/nix/module-eval/bao-grants.nix index b99b3d3c..3fc14745 100644 --- a/nix/module-eval/bao-grants.nix +++ b/nix/module-eval/bao-grants.nix @@ -866,6 +866,34 @@ let && u.serviceConfig.Restart == "on-failure" ) grantingUnitNames; } + { + # A fetch from the store outlives a store or gateway that is down or + # sealed for hours: `start-limit-hit` does not clear when the store comes + # back, so a shorter budget leaves the credential missing until someone + # starts the unit by hand. + name = "every unit fetching a credential or certificate from the store retries 2880 times at 30s"; + ok = + lib.all + ( + unit: + let + s = baoGrantWithConsumers.systemd.services; + u = s.${unit}; + in + s ? ${unit} + && toString u.unitConfig.StartLimitBurst == "2880" + && toString u.unitConfig.StartLimitIntervalSec == "90000" + && toString u.serviceConfig.RestartSec == "30" + && u.serviceConfig.Restart == "on-failure" + ) + ( + [ + "swarm-services-cert" + "swarm-bao-forwarder-oidc" + ] + ++ policyReaders + ); + } { # The only unit left acting with the token, so the only one that may # skip on it.