Watch
0
0
Fork
You've already forked hyperhive
0

credential units: restart consumers on a changed credential; fix the ordering claim

The previous commit's comments said a unit in auto-restart keeps its
start job, so anything ordered after it waits for the whole 24h retry
window. That is wrong under the default RestartMode=normal: each failed
attempt passes through `failed`, which ends that start job. `After=`
dependents proceed after one attempt, `Requires=` dependents fail with
`dependency`, and the retries continue as fresh start jobs. The
2026-09-24 journal shows it with the already-2880 swarm-services-cert:
nginx got "Dependency failed" 1ms after the first failure, and
switch-to-configuration exited before the first restart was scheduled.
The comments in lib/store-retry.nix, glue-matrix-bao-token.nix,
glue-queue-agent-credential.nix, swarm-otel.nix and swarm-grafana.nix
now say that, and so does docs/swarm/credentials.md.

Because dependents start after one attempt, a consumer that loads its
credential at start never sees a value a later attempt lands, or a
rotated one. nix/host-modules/lib/refresh-consumer.nix adds
`secret_differs` and `refresh_consumer`, and the four fetch units whose
consumers take a start-time copy call them after the write, only when
the value changed:

- swarm-bao-matrix-token -> tuwunel.service in hive-matrix
- swarm-bao-otel-oidc -> opentelemetry-collector.service in swarm-otel
- swarm-bao-grafana-oidc -> grafana.service in the grafana container
- swarm-bao-forwarder-oidc -> opentelemetry-collector.service in swarm-bao

A running consumer is try-restarted, a failed one is reset and started,
all with --no-block. Inline in the fetch script rather than a
PathChanged path unit because the fetch script is the only writer and
already knows whether the value changed, and it is the same shape as
this PR's nginx hook and swarm-bao-nats-tls's restart of nats.

module-eval-bao-grants gains one case per consumer.

Refs #4662
This commit is contained in:
atlas 2026-09-30 00:00:28 +02:00 • committed by mara
commit 7eb966fe2b
9 changed files with 178 additions and 22 deletions

View file

@ -174,9 +174,17 @@ needed the store first. The unit exists whenever
sets from its own store address — the same all-or-nothing gate the per-hive
queue credential beside it uses, and the reason the delivery above never lands
in a container with nothing to read it. It fails loudly where the hive-side
readers degrade quietly, which is deliberate: a missing queue secret means a
swarm whose publisher has yet to run, while a refused certificate means an
agent that believes it reaches the store and never does.
readers treat a missing value as a normal state, which is deliberate: a
missing queue secret means a swarm whose publisher has yet to run, while a
refused certificate means an agent that believes it reaches the store and
never does.
Both kinds of unit retry a failed login every 30 seconds for a day
(`nix/host-modules/lib/store-retry.nix`). A unit ordered after one waits for a
single attempt, not for the retries: an attempt that fails ends that start, so
the dependent starts with whatever is already on disk. When a later attempt
lands a changed value, the homeserver, Grafana and collector readers restart
the service that loaded it (`nix/host-modules/lib/refresh-consumer.nix`).
## Progressive enhancement

View file

@ -82,6 +82,7 @@ let
matrixMachine = "hive-matrix";
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
refreshConsumer = import ./lib/refresh-consumer.nix { };
storeRetry = import ./lib/store-retry.nix { };
in
{
@ -151,11 +152,12 @@ in
path = [
deployCfg.bao.package
pkgs.coreutils
pkgs.systemd
];
# ./lib/store-retry.nix. ⚠️ This unit is `Before=` the homeserver's
# container, and an auto-restarting unit is still `activating`, so the
# homeserver waits for as long as the login below keeps failing — up to
# the full 24h on a store that stays sealed.
# ./lib/store-retry.nix. This unit is `Before=` the homeserver's
# container, which waits for one attempt and then starts on the token it
# already has; a token a later attempt lands restarts the homeserver
# (below).
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
serviceConfig = storeRetry.serviceConfig // {
Type = "oneshot";
@ -179,14 +181,14 @@ in
set -euo pipefail
${atomicWriteSecret}
${refreshConsumer}
# A sealed or uninitialised store answers on the port and never
# answers the read, so "the store is up" is not the same as "the
# store can answer". `TimeoutStartSec` above bounds each attempt and
# a timed-out attempt is retried like any other failure; the
# homeserver only `Wants=` this unit, so a retry budget spent
# degrades to keeping the local token rather than failing the
# container.
# homeserver only `Wants=` this unit, so a failed attempt lets it
# start on the local token rather than failing it.
# `bao`'s own message is the only thing separating a missing value
# from a refused identity from an unreachable host. This unit's
# degraded mode is correct for all three, so it reports which one
@ -232,6 +234,8 @@ in
exit 0
fi
changed=0
if secret_differs ${lib.escapeShellArg (toString deployCfg.matrix.appserviceTokenFile)} "$token"; then changed=1; fi
atomic_write_secret 0600 "" ${lib.escapeShellArg (toString deployCfg.matrix.appserviceTokenFile)} "$token"
# Re-stamp the registration file from the token just written. The
@ -245,6 +249,13 @@ in
# hive-matrix's own renderer rather than a `printf` here, so the
# registration's shape has one home.
${deployCfg.matrix.appserviceRegistrationScript}
# tuwunel loads the registration through `LoadCredential`, a copy
# taken at start, while hive-c0re reads the token file on every call:
# a changed token splits the two until the homeserver restarts.
if [ "$changed" = 1 ]; then
refresh_consumer ${lib.escapeShellArg matrixMachine} tuwunel.service
fi
'';
};
})

View file

@ -173,10 +173,10 @@ in
requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ];
# Ordered before hive-c0re, so no agent container renders ahead of an
# attempt at its credential. `Wants=`, not `Requires=`: a store this
# unit can't reach delays hive-c0re's start for as long as this unit
# retries (up to the 24h of ./lib/store-retry.nix) rather than failing
# it — hive-c0re starts once the retries end, whatever credential is
# or isn't on disk by then.
# unit can't reach delays hive-c0re's start by one attempt
# (`TimeoutStartSec` below) rather than failing it — hive-c0re starts
# after that attempt with whatever credential is on disk, while
# ./lib/store-retry.nix keeps retrying in the background.
before = [ "hive-c0re.service" ];
wantedBy = [
"multi-user.target"

View file

@ -0,0 +1,57 @@
# Shared shell steps for a host oneshot that fetches a credential into a
# file a service inside a container loads at start (`LoadCredential`, or a
# config file expanded once while parsing). Such a service never sees a
# value that lands after it started — a late fetch or a rotation — until it
# is restarted, so the fetch restarts it, and only when the value changed:
# a routine re-fetch of the same value leaves it alone.
#
# secret_differs <path> <value>
# True when <path> is missing or holds something other than <value>.
# Call it BEFORE `atomic_write_secret` writes <path>. Compares with
# `$(< path)`, a bash builtin, so the value never becomes an argument in
# /proc; both sides lose their trailing newlines, which is how
# `atomic_write_secret` writes and `$(bao …)` reads.
#
# refresh_consumer <machine> <unit>
# Nothing while `container@<machine>` is not active: a container that
# starts later loads the file as it starts. Otherwise a running
# consumer is restarted, and a failed one —
# a consumer that found no credential may have hit its start limit — is
# reset and started. A consumer stopped on purpose stays stopped.
# `--no-block` throughout: a caller ordered `Before=` its consumer must
# not wait on a job that waits on the caller (the deadlock stated at
# `swarm-services-cert`'s propagation in ../hive-tls.nix).
#
# Pure function — NOT a NixOS module. Call it from a module's `let`:
#
# refreshConsumer = import ./lib/refresh-consumer.nix { };
# script = ''
# ${refreshConsumer}
# changed=0
# if secret_differs "$path" "$secret"; then changed=1; fi
# atomic_write_secret 0400 root:root "$path" "$secret"
# if [ "$changed" = 1 ]; then refresh_consumer my-machine my.service; fi
# '';
#
# Requires `systemd` on the caller's `path`.
{ }:
''
secret_differs() {
[ ! -f "$1" ] || [ "$(< "$1")" != "$2" ]
}
refresh_consumer() {
local machine="$1" unit="$2"
if ! systemctl is-active --quiet "container@$machine.service"; then
return 0
fi
if systemctl --machine="$machine" is-failed --quiet "$unit"; then
echo "the credential for $unit in $machine changed and $unit had failed — starting it"
systemctl --machine="$machine" reset-failed "$unit"
systemctl --machine="$machine" start --no-block "$unit"
else
echo "the credential for $unit in $machine changed — restarting it if it runs"
systemctl --machine="$machine" try-restart --no-block "$unit"
fi
}
''

View file

@ -10,9 +10,12 @@
# between attempts and the burst is never reached: 2880 × 30s is 24h inside
# a 25h window.
#
# ⚠️ A unit in auto-restart is still `activating`, so its start job stays
# queued across attempts: anything ordered `After=` it waits for as long as
# it retries, up to the full 24h.
# Under the default `RestartMode=normal` each failed attempt passes through
# `failed`, which ends that attempt's start job: a unit ordered `After=` it
# waits for ONE attempt, a unit that `Requires=` it fails with `dependency`,
# and the retries continue in the background as fresh start jobs. A consumer
# that loaded the credential before it landed does not see it without a
# restart — ./refresh-consumer.nix.
#
# Pure attrset — NOT a NixOS module. Use from a unit definition:
#

View file

@ -1129,6 +1129,7 @@ let
forwarderHostSecretDir = builtins.dirOf forwarderHostSecretPath;
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
refreshConsumer = import ./lib/refresh-consumer.nix { };
# The collector runs under `DynamicUser` and opens `client_secret_file`
# itself, so there is no uid to hand a 0400 file to. `LoadCredential` reads
# it as root before the sandbox exists and re-exposes it under a path that
@ -2379,6 +2380,7 @@ in
path = [
baoDeploy.package
pkgs.coreutils
pkgs.systemd
];
# ./lib/store-retry.nix: the secret is minted by authelia and copied in
# by the publisher, both of which may be a host away and neither of
@ -2407,6 +2409,7 @@ in
set -euo pipefail
${atomicWriteSecret}
${refreshConsumer}
# `bao`'s own message is the only thing separating a missing value
# from a refused identity from an unreachable store, and this unit
@ -2448,7 +2451,15 @@ in
# an argument in /proc the way `install <<<"$secret"` or an `echo`
# from `path` would.
install -d -m 0755 ${lib.escapeShellArg forwarderHostSecretDir}
changed=0
if secret_differs ${lib.escapeShellArg forwarderHostSecretPath} "$secret"; then changed=1; fi
atomic_write_secret 0400 root:root ${lib.escapeShellArg forwarderHostSecretPath} "$secret"
# `LoadCredential` copies the file at start only: a rotated secret
# reaches the forwarder by restarting it.
if [ "$changed" = 1 ]; then
refresh_consumer ${lib.escapeShellArg cfg.machine} opentelemetry-collector.service
fi
'';
};
})

View file

@ -239,6 +239,7 @@ let
grafanaUid = config.ids.uids.grafana;
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
refreshConsumer = import ./lib/refresh-consumer.nix { };
# Shared host netns, like every sibling swarm container: the gateway
# reaches this at 127.0.0.1:<port>.
@ -673,9 +674,11 @@ in
path = [
baoDeploy.package
pkgs.coreutils
pkgs.systemd
];
# ./lib/store-retry.nix. Grafana's container is ordered after this unit,
# so it waits while the login below keeps failing.
# ./lib/store-retry.nix. Grafana's container is ordered after this unit
# and waits for one attempt; a secret a later attempt lands restarts
# Grafana (below).
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
serviceConfig = storeRetry.serviceConfig // {
Type = "oneshot";
@ -700,6 +703,7 @@ in
set -euo pipefail
${atomicWriteSecret}
${refreshConsumer}
# `bao`'s own message is the only thing separating a missing value from
# a refused identity from an unreachable host. This unit's degraded
@ -757,7 +761,14 @@ in
# grants the group nothing; if this mode ever widens, the gid has to be
# discovered at runtime rather than assumed.
install -d -m 0755 ${lib.escapeShellArg hostSecretDir}
changed=0
if secret_differs ${lib.escapeShellArg hostSecretPath} "$secret"; then changed=1; fi
atomic_write_secret 0400 ${lib.escapeShellArg "${toString config.ids.uids.grafana}:0"} ${lib.escapeShellArg hostSecretPath} "$secret"
# `$__file{}` is expanded once, while Grafana parses its config.
if [ "$changed" = 1 ]; then
refresh_consumer ${lib.escapeShellArg cfg.machine} grafana.service
fi
'';
};

View file

@ -167,6 +167,7 @@ let
collectorSecretPath = "/run/credentials/opentelemetry-collector.service/${collectorCredentialId}";
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
refreshConsumer = import ./lib/refresh-consumer.nix { };
# `attrNames` is sorted, so this is a function of the hive SET and not of
# the order anyone wrote it in.
@ -790,9 +791,11 @@ in
path = [
baoDeploy.package
pkgs.coreutils
pkgs.systemd
];
# ./lib/store-retry.nix. The collector's container is ordered after
# this unit, so it waits while the login below keeps failing.
# ./lib/store-retry.nix. The collector's container is ordered after this
# unit and waits for one attempt; a secret a later attempt lands restarts
# the collector (below).
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
serviceConfig = storeRetry.serviceConfig // {
Type = "oneshot";
@ -817,6 +820,7 @@ in
set -euo pipefail
${atomicWriteSecret}
${refreshConsumer}
# `bao`'s own message is the only thing separating a missing value
# from a refused identity from an unreachable host. This unit's
@ -863,7 +867,15 @@ in
# no uid to give this to — `LoadCredential` reads it as root before
# the sandbox exists and re-exposes it to whichever uid the unit got.
install -d -m 0755 ${lib.escapeShellArg collectorHostSecretDir}
changed=0
if secret_differs ${lib.escapeShellArg collectorHostSecretPath} "$secret"; then changed=1; fi
atomic_write_secret 0400 root:root ${lib.escapeShellArg collectorHostSecretPath} "$secret"
# `LoadCredential` copies the file at start only, and a collector that
# started without it is in its start limit.
if [ "$changed" = 1 ]; then
refresh_consumer ${lib.escapeShellArg cfg.machine} opentelemetry-collector.service
fi
'';
};

View file

@ -1563,6 +1563,49 @@ let
&& lib.hasInfix ''"token_policies":["swarm-operator-viewer"]'' s
&& lib.hasInfix ''"allowed_redirect_uris":["https://bao-ui.t.local/ui/vault/auth/oidc/oidc/callback"]'' s;
}
];
]
# A consumer that loads its credential at start never sees one a later
# attempt lands, so the fetch that lands it restarts that consumer — only
# on a changed value, so a routine re-fetch leaves it running.
++
map
(
{
fetch,
machine,
consumer,
}:
{
name = "${fetch} restarts ${consumer} in ${machine} when the value it fetched changed";
ok =
let
s = baoGrantWithConsumers.systemd.services.${fetch}.script;
in
lib.hasInfix "secret_differs " s
&& lib.hasInfix "refresh_consumer ${lib.escapeShellArg machine} ${consumer}" s;
}
)
[
{
fetch = "swarm-bao-matrix-token";
machine = "hive-matrix";
consumer = "tuwunel.service";
}
{
fetch = "swarm-bao-otel-oidc";
machine = baoGrantWithConsumers.services.hyperhive.swarm.otel.machine;
consumer = "opentelemetry-collector.service";
}
{
fetch = "swarm-bao-grafana-oidc";
machine = baoGrantWithConsumers.services.hyperhive.swarm.grafana.machine;
consumer = "grafana.service";
}
{
fetch = "swarm-bao-forwarder-oidc";
machine = baoGrantWithConsumers.services.hyperhive.swarm.bao.machine;
consumer = "opentelemetry-collector.service";
}
];
in
runGroup "bao-grants" cases