From 7eb966fe2bfe70b4812d6aa7a3a9e9a7f647e35e Mon Sep 17 00:00:00 2001 From: atlas Date: Wed, 30 Sep 2026 00:00:28 +0200 Subject: [PATCH] credential units: restart consumers on a changed credential; fix the ordering claim The previous commit's comments said a unit in auto-restart keeps its start job, so anything ordered after it waits for the whole 24h retry window. That is wrong under the default RestartMode=normal: each failed attempt passes through `failed`, which ends that start job. `After=` dependents proceed after one attempt, `Requires=` dependents fail with `dependency`, and the retries continue as fresh start jobs. The 2026-09-24 journal shows it with the already-2880 swarm-services-cert: nginx got "Dependency failed" 1ms after the first failure, and switch-to-configuration exited before the first restart was scheduled. The comments in lib/store-retry.nix, glue-matrix-bao-token.nix, glue-queue-agent-credential.nix, swarm-otel.nix and swarm-grafana.nix now say that, and so does docs/swarm/credentials.md. Because dependents start after one attempt, a consumer that loads its credential at start never sees a value a later attempt lands, or a rotated one. nix/host-modules/lib/refresh-consumer.nix adds `secret_differs` and `refresh_consumer`, and the four fetch units whose consumers take a start-time copy call them after the write, only when the value changed: - swarm-bao-matrix-token -> tuwunel.service in hive-matrix - swarm-bao-otel-oidc -> opentelemetry-collector.service in swarm-otel - swarm-bao-grafana-oidc -> grafana.service in the grafana container - swarm-bao-forwarder-oidc -> opentelemetry-collector.service in swarm-bao A running consumer is try-restarted, a failed one is reset and started, all with --no-block. Inline in the fetch script rather than a PathChanged path unit because the fetch script is the only writer and already knows whether the value changed, and it is the same shape as this PR's nginx hook and swarm-bao-nats-tls's restart of nats. module-eval-bao-grants gains one case per consumer. Refs #4662 --- docs/swarm/credentials.md | 14 ++++- nix/host-modules/glue-matrix-bao-token.nix | 25 +++++--- .../glue-queue-agent-credential.nix | 8 +-- nix/host-modules/lib/refresh-consumer.nix | 57 +++++++++++++++++++ nix/host-modules/lib/store-retry.nix | 9 ++- nix/host-modules/swarm-bao.nix | 11 ++++ nix/host-modules/swarm-grafana.nix | 15 ++++- nix/host-modules/swarm-otel.nix | 16 +++++- nix/module-eval/bao-grants.nix | 45 ++++++++++++++- 9 files changed, 178 insertions(+), 22 deletions(-) create mode 100644 nix/host-modules/lib/refresh-consumer.nix diff --git a/docs/swarm/credentials.md b/docs/swarm/credentials.md index 728a01d3..a60443ce 100644 --- a/docs/swarm/credentials.md +++ b/docs/swarm/credentials.md @@ -174,9 +174,17 @@ needed the store first. The unit exists whenever sets from its own store address — the same all-or-nothing gate the per-hive queue credential beside it uses, and the reason the delivery above never lands in a container with nothing to read it. It fails loudly where the hive-side -readers degrade quietly, which is deliberate: a missing queue secret means a -swarm whose publisher has yet to run, while a refused certificate means an -agent that believes it reaches the store and never does. +readers treat a missing value as a normal state, which is deliberate: a +missing queue secret means a swarm whose publisher has yet to run, while a +refused certificate means an agent that believes it reaches the store and +never does. + +Both kinds of unit retry a failed login every 30 seconds for a day +(`nix/host-modules/lib/store-retry.nix`). A unit ordered after one waits for a +single attempt, not for the retries: an attempt that fails ends that start, so +the dependent starts with whatever is already on disk. When a later attempt +lands a changed value, the homeserver, Grafana and collector readers restart +the service that loaded it (`nix/host-modules/lib/refresh-consumer.nix`). ## Progressive enhancement diff --git a/nix/host-modules/glue-matrix-bao-token.nix b/nix/host-modules/glue-matrix-bao-token.nix index f4b38d0c..e7d585bf 100644 --- a/nix/host-modules/glue-matrix-bao-token.nix +++ b/nix/host-modules/glue-matrix-bao-token.nix @@ -82,6 +82,7 @@ let matrixMachine = "hive-matrix"; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + refreshConsumer = import ./lib/refresh-consumer.nix { }; storeRetry = import ./lib/store-retry.nix { }; in { @@ -151,11 +152,12 @@ in path = [ deployCfg.bao.package pkgs.coreutils + pkgs.systemd ]; - # ./lib/store-retry.nix. ⚠️ This unit is `Before=` the homeserver's - # container, and an auto-restarting unit is still `activating`, so the - # homeserver waits for as long as the login below keeps failing — up to - # the full 24h on a store that stays sealed. + # ./lib/store-retry.nix. This unit is `Before=` the homeserver's + # container, which waits for one attempt and then starts on the token it + # already has; a token a later attempt lands restarts the homeserver + # (below). inherit (storeRetry) startLimitBurst startLimitIntervalSec; serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; @@ -179,14 +181,14 @@ in set -euo pipefail ${atomicWriteSecret} + ${refreshConsumer} # A sealed or uninitialised store answers on the port and never # answers the read, so "the store is up" is not the same as "the # store can answer". `TimeoutStartSec` above bounds each attempt and # a timed-out attempt is retried like any other failure; the - # homeserver only `Wants=` this unit, so a retry budget spent - # degrades to keeping the local token rather than failing the - # container. + # homeserver only `Wants=` this unit, so a failed attempt lets it + # start on the local token rather than failing it. # `bao`'s own message is the only thing separating a missing value # from a refused identity from an unreachable host. This unit's # degraded mode is correct for all three, so it reports which one @@ -232,6 +234,8 @@ in exit 0 fi + changed=0 + if secret_differs ${lib.escapeShellArg (toString deployCfg.matrix.appserviceTokenFile)} "$token"; then changed=1; fi atomic_write_secret 0600 "" ${lib.escapeShellArg (toString deployCfg.matrix.appserviceTokenFile)} "$token" # Re-stamp the registration file from the token just written. The @@ -245,6 +249,13 @@ in # hive-matrix's own renderer rather than a `printf` here, so the # registration's shape has one home. ${deployCfg.matrix.appserviceRegistrationScript} + + # tuwunel loads the registration through `LoadCredential`, a copy + # taken at start, while hive-c0re reads the token file on every call: + # a changed token splits the two until the homeserver restarts. + if [ "$changed" = 1 ]; then + refresh_consumer ${lib.escapeShellArg matrixMachine} tuwunel.service + fi ''; }; }) diff --git a/nix/host-modules/glue-queue-agent-credential.nix b/nix/host-modules/glue-queue-agent-credential.nix index cab3f3ad..894ea8df 100644 --- a/nix/host-modules/glue-queue-agent-credential.nix +++ b/nix/host-modules/glue-queue-agent-credential.nix @@ -173,10 +173,10 @@ in requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ]; # Ordered before hive-c0re, so no agent container renders ahead of an # attempt at its credential. `Wants=`, not `Requires=`: a store this - # unit can't reach delays hive-c0re's start for as long as this unit - # retries (up to the 24h of ./lib/store-retry.nix) rather than failing - # it — hive-c0re starts once the retries end, whatever credential is - # or isn't on disk by then. + # unit can't reach delays hive-c0re's start by one attempt + # (`TimeoutStartSec` below) rather than failing it — hive-c0re starts + # after that attempt with whatever credential is on disk, while + # ./lib/store-retry.nix keeps retrying in the background. before = [ "hive-c0re.service" ]; wantedBy = [ "multi-user.target" diff --git a/nix/host-modules/lib/refresh-consumer.nix b/nix/host-modules/lib/refresh-consumer.nix new file mode 100644 index 00000000..a30386f7 --- /dev/null +++ b/nix/host-modules/lib/refresh-consumer.nix @@ -0,0 +1,57 @@ +# Shared shell steps for a host oneshot that fetches a credential into a +# file a service inside a container loads at start (`LoadCredential`, or a +# config file expanded once while parsing). Such a service never sees a +# value that lands after it started — a late fetch or a rotation — until it +# is restarted, so the fetch restarts it, and only when the value changed: +# a routine re-fetch of the same value leaves it alone. +# +# secret_differs +# True when is missing or holds something other than . +# Call it BEFORE `atomic_write_secret` writes . Compares with +# `$(< path)`, a bash builtin, so the value never becomes an argument in +# /proc; both sides lose their trailing newlines, which is how +# `atomic_write_secret` writes and `$(bao …)` reads. +# +# refresh_consumer +# Nothing while `container@` is not active: a container that +# starts later loads the file as it starts. Otherwise a running +# consumer is restarted, and a failed one — +# a consumer that found no credential may have hit its start limit — is +# reset and started. A consumer stopped on purpose stays stopped. +# `--no-block` throughout: a caller ordered `Before=` its consumer must +# not wait on a job that waits on the caller (the deadlock stated at +# `swarm-services-cert`'s propagation in ../hive-tls.nix). +# +# Pure function — NOT a NixOS module. Call it from a module's `let`: +# +# refreshConsumer = import ./lib/refresh-consumer.nix { }; +# script = '' +# ${refreshConsumer} +# changed=0 +# if secret_differs "$path" "$secret"; then changed=1; fi +# atomic_write_secret 0400 root:root "$path" "$secret" +# if [ "$changed" = 1 ]; then refresh_consumer my-machine my.service; fi +# ''; +# +# Requires `systemd` on the caller's `path`. +{ }: +'' + secret_differs() { + [ ! -f "$1" ] || [ "$(< "$1")" != "$2" ] + } + + refresh_consumer() { + local machine="$1" unit="$2" + if ! systemctl is-active --quiet "container@$machine.service"; then + return 0 + fi + if systemctl --machine="$machine" is-failed --quiet "$unit"; then + echo "the credential for $unit in $machine changed and $unit had failed — starting it" + systemctl --machine="$machine" reset-failed "$unit" + systemctl --machine="$machine" start --no-block "$unit" + else + echo "the credential for $unit in $machine changed — restarting it if it runs" + systemctl --machine="$machine" try-restart --no-block "$unit" + fi + } +'' diff --git a/nix/host-modules/lib/store-retry.nix b/nix/host-modules/lib/store-retry.nix index 8034a4f4..0072ebb6 100644 --- a/nix/host-modules/lib/store-retry.nix +++ b/nix/host-modules/lib/store-retry.nix @@ -10,9 +10,12 @@ # between attempts and the burst is never reached: 2880 × 30s is 24h inside # a 25h window. # -# ⚠️ A unit in auto-restart is still `activating`, so its start job stays -# queued across attempts: anything ordered `After=` it waits for as long as -# it retries, up to the full 24h. +# Under the default `RestartMode=normal` each failed attempt passes through +# `failed`, which ends that attempt's start job: a unit ordered `After=` it +# waits for ONE attempt, a unit that `Requires=` it fails with `dependency`, +# and the retries continue in the background as fresh start jobs. A consumer +# that loaded the credential before it landed does not see it without a +# restart — ./refresh-consumer.nix. # # Pure attrset — NOT a NixOS module. Use from a unit definition: # diff --git a/nix/host-modules/swarm-bao.nix b/nix/host-modules/swarm-bao.nix index 104c7789..b47be0ef 100644 --- a/nix/host-modules/swarm-bao.nix +++ b/nix/host-modules/swarm-bao.nix @@ -1129,6 +1129,7 @@ let forwarderHostSecretDir = builtins.dirOf forwarderHostSecretPath; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + refreshConsumer = import ./lib/refresh-consumer.nix { }; # The collector runs under `DynamicUser` and opens `client_secret_file` # itself, so there is no uid to hand a 0400 file to. `LoadCredential` reads # it as root before the sandbox exists and re-exposes it under a path that @@ -2379,6 +2380,7 @@ in path = [ baoDeploy.package pkgs.coreutils + pkgs.systemd ]; # ./lib/store-retry.nix: the secret is minted by authelia and copied in # by the publisher, both of which may be a host away and neither of @@ -2407,6 +2409,7 @@ in set -euo pipefail ${atomicWriteSecret} + ${refreshConsumer} # `bao`'s own message is the only thing separating a missing value # from a refused identity from an unreachable store, and this unit @@ -2448,7 +2451,15 @@ in # an argument in /proc the way `install <<<"$secret"` or an `echo` # from `path` would. install -d -m 0755 ${lib.escapeShellArg forwarderHostSecretDir} + changed=0 + if secret_differs ${lib.escapeShellArg forwarderHostSecretPath} "$secret"; then changed=1; fi atomic_write_secret 0400 root:root ${lib.escapeShellArg forwarderHostSecretPath} "$secret" + + # `LoadCredential` copies the file at start only: a rotated secret + # reaches the forwarder by restarting it. + if [ "$changed" = 1 ]; then + refresh_consumer ${lib.escapeShellArg cfg.machine} opentelemetry-collector.service + fi ''; }; }) diff --git a/nix/host-modules/swarm-grafana.nix b/nix/host-modules/swarm-grafana.nix index 89ac3314..a601c1db 100644 --- a/nix/host-modules/swarm-grafana.nix +++ b/nix/host-modules/swarm-grafana.nix @@ -239,6 +239,7 @@ let grafanaUid = config.ids.uids.grafana; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + refreshConsumer = import ./lib/refresh-consumer.nix { }; # Shared host netns, like every sibling swarm container: the gateway # reaches this at 127.0.0.1:. @@ -673,9 +674,11 @@ in path = [ baoDeploy.package pkgs.coreutils + pkgs.systemd ]; - # ./lib/store-retry.nix. Grafana's container is ordered after this unit, - # so it waits while the login below keeps failing. + # ./lib/store-retry.nix. Grafana's container is ordered after this unit + # and waits for one attempt; a secret a later attempt lands restarts + # Grafana (below). inherit (storeRetry) startLimitBurst startLimitIntervalSec; serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; @@ -700,6 +703,7 @@ in set -euo pipefail ${atomicWriteSecret} + ${refreshConsumer} # `bao`'s own message is the only thing separating a missing value from # a refused identity from an unreachable host. This unit's degraded @@ -757,7 +761,14 @@ in # grants the group nothing; if this mode ever widens, the gid has to be # discovered at runtime rather than assumed. install -d -m 0755 ${lib.escapeShellArg hostSecretDir} + changed=0 + if secret_differs ${lib.escapeShellArg hostSecretPath} "$secret"; then changed=1; fi atomic_write_secret 0400 ${lib.escapeShellArg "${toString config.ids.uids.grafana}:0"} ${lib.escapeShellArg hostSecretPath} "$secret" + + # `$__file{}` is expanded once, while Grafana parses its config. + if [ "$changed" = 1 ]; then + refresh_consumer ${lib.escapeShellArg cfg.machine} grafana.service + fi ''; }; diff --git a/nix/host-modules/swarm-otel.nix b/nix/host-modules/swarm-otel.nix index c68f38ef..ed32500f 100644 --- a/nix/host-modules/swarm-otel.nix +++ b/nix/host-modules/swarm-otel.nix @@ -167,6 +167,7 @@ let collectorSecretPath = "/run/credentials/opentelemetry-collector.service/${collectorCredentialId}"; atomicWriteSecret = import ./lib/atomic-write-secret.nix { }; + refreshConsumer = import ./lib/refresh-consumer.nix { }; # `attrNames` is sorted, so this is a function of the hive SET and not of # the order anyone wrote it in. @@ -790,9 +791,11 @@ in path = [ baoDeploy.package pkgs.coreutils + pkgs.systemd ]; - # ./lib/store-retry.nix. The collector's container is ordered after - # this unit, so it waits while the login below keeps failing. + # ./lib/store-retry.nix. The collector's container is ordered after this + # unit and waits for one attempt; a secret a later attempt lands restarts + # the collector (below). inherit (storeRetry) startLimitBurst startLimitIntervalSec; serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; @@ -817,6 +820,7 @@ in set -euo pipefail ${atomicWriteSecret} + ${refreshConsumer} # `bao`'s own message is the only thing separating a missing value # from a refused identity from an unreachable host. This unit's @@ -863,7 +867,15 @@ in # no uid to give this to — `LoadCredential` reads it as root before # the sandbox exists and re-exposes it to whichever uid the unit got. install -d -m 0755 ${lib.escapeShellArg collectorHostSecretDir} + changed=0 + if secret_differs ${lib.escapeShellArg collectorHostSecretPath} "$secret"; then changed=1; fi atomic_write_secret 0400 root:root ${lib.escapeShellArg collectorHostSecretPath} "$secret" + + # `LoadCredential` copies the file at start only, and a collector that + # started without it is in its start limit. + if [ "$changed" = 1 ]; then + refresh_consumer ${lib.escapeShellArg cfg.machine} opentelemetry-collector.service + fi ''; }; diff --git a/nix/module-eval/bao-grants.nix b/nix/module-eval/bao-grants.nix index 3fc14745..c5be65a0 100644 --- a/nix/module-eval/bao-grants.nix +++ b/nix/module-eval/bao-grants.nix @@ -1563,6 +1563,49 @@ let && lib.hasInfix ''"token_policies":["swarm-operator-viewer"]'' s && lib.hasInfix ''"allowed_redirect_uris":["https://bao-ui.t.local/ui/vault/auth/oidc/oidc/callback"]'' s; } - ]; + ] + # A consumer that loads its credential at start never sees one a later + # attempt lands, so the fetch that lands it restarts that consumer — only + # on a changed value, so a routine re-fetch leaves it running. + ++ + map + ( + { + fetch, + machine, + consumer, + }: + { + name = "${fetch} restarts ${consumer} in ${machine} when the value it fetched changed"; + ok = + let + s = baoGrantWithConsumers.systemd.services.${fetch}.script; + in + lib.hasInfix "secret_differs " s + && lib.hasInfix "refresh_consumer ${lib.escapeShellArg machine} ${consumer}" s; + } + ) + [ + { + fetch = "swarm-bao-matrix-token"; + machine = "hive-matrix"; + consumer = "tuwunel.service"; + } + { + fetch = "swarm-bao-otel-oidc"; + machine = baoGrantWithConsumers.services.hyperhive.swarm.otel.machine; + consumer = "opentelemetry-collector.service"; + } + { + fetch = "swarm-bao-grafana-oidc"; + machine = baoGrantWithConsumers.services.hyperhive.swarm.grafana.machine; + consumer = "grafana.service"; + } + { + fetch = "swarm-bao-forwarder-oidc"; + machine = baoGrantWithConsumers.services.hyperhive.swarm.bao.machine; + consumer = "opentelemetry-collector.service"; + } + ]; in runGroup "bao-grants" cases