hive-tls: renew the swarm-services leaf on a daily timer

The store's `swarm-services` role issues the services leaf for 720h, and
`swarm-services-cert` only ever ran at boot or rebuild: it is a
`RemainAfterExit` oneshot wanted by `multi-user.target` and no timer
targeted it. A hive not rebuilt within 30 days served an expired leaf.

`swarm-services-cert-renew` runs the same script from a daily timer. It
is a unit of its own because a timer starting the `RemainAfterExit` unit
is a no-op, and restarting that unit instead would propagate through
`hive-gateway-self-signed-cert`'s `Requires=` to nginx, so a sealed store
would take the gateway down over a still-valid leaf. Nothing requires or
orders against the new unit; it has no `Restart=`, so a failure stays in
`systemctl --failed` until the next tick, and the script only moves files
into place after the store has answered.

The re-issue threshold was `checkend 2592000`, the whole 30-day
lifetime, so every run re-issued. It is now half the role's lifetime,
read from a new internal option `deploy.bao.servicesPkiLeafTtlHours`
that the role's `ttl`/`max_ttl` also read. Boot and timer share the
script and so the threshold. The services-root re-check reads the
same option, at the store's own replacement threshold (hours × 3600),
so the hive asks for a new leaf when the store replaces its root. A
`flock` keeps the two runs from interleaving one issuance's key with another's leaf.

`checks.module-eval-hive-tls` pins the timer, that the unit it starts
re-runs the issuance without `RemainAfterExit`, that nothing depends on
it, and that both the leaf and root thresholds move with the option.

Closes #4587
This commit is contained in:
atlas 2026-09-25 20:32:28 +02:00 • committed by mara
commit 0649673ebf
5 changed files with 283 additions and 50 deletions

View file

@ -247,6 +247,40 @@ let
covers "$leaf" "$hiveNames"
}
'';
# What `swarm-services-cert` and `swarm-services-cert-renew` both run the
# issuance with. Bound here rather than read back off the first unit: its
# merged `path` already carries systemd's defaults, so the second would
# render a different `PATH` and fail to merge.
servicesCertPath = [
baoDeploy.package
pkgs.jq
pkgs.openssl
pkgs.coreutils
pkgs.systemd
# `cmp`, which is NOT in coreutils. Without it the root-changed test
# below is `command not found` — 127, invisible under `if !`, and
# therefore always "changed", so the bundle rebuild it guards fired
# on every issuance instead of on a new root.
pkgs.diffutils
# `flock`, for the lock at the top of the script.
pkgs.util-linux
];
servicesCertEnvironment = {
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
}
// lib.optionalAttrs (cfg.baoClientCertFile != null) {
BAO_CLIENT_CERT = cfg.baoClientCertFile;
}
// lib.optionalAttrs (cfg.baoClientKeyFile != null) {
BAO_CLIENT_KEY = cfg.baoClientKeyFile;
}
# Absent means the system trust store, which is what a deployment
# with a real CA wants and what a self-signed one must not be left
# with.
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
BAO_CACERT = baoDeploy.serverCaFile;
};
in
{
# Host-side TLS trust root for the self-signed gateway mode.
@ -367,11 +401,14 @@ in
# rebuild of every hive that is not the CA host, about a fallback
# that no longer happens.
# Both run on the deploy and sit on the gateway's start path, so a
# failure in either takes TLS down for every service behind it.
# The first two run on the deploy and sit on the gateway's start path,
# so a failure in either takes TLS down for every service behind it.
# The renewal sits on no start path, and a failure there is a leaf that
# lapses days later unless someone reads why it failed now.
services.hyperhive.swarm.otel.journaldUnits = [
"hive-tls-ca"
"swarm-services-cert"
"swarm-services-cert-renew"
];
# Generate (and rotate) the hive CA + gateway leaf before anything
@ -598,8 +635,8 @@ in
#
# The services leaf is not one of them any longer: it is issued
# by the secret store, whose key this unit does not hold and
# cannot re-sign under. Renewing it is `swarm-services-cert`'s
# job, at boot, which is a cadence its own issue owns.
# cannot re-sign under. `swarm-services-cert-renew` renews it,
# on its own daily timer.
${leafCoverage}
# Re-sign only when a leaf is within half its validity of expiry.
@ -694,18 +731,7 @@ in
"swarm-bao-services-issuer-policy.service"
];
wants = [ "container@${baoCfg.machine}.service" ];
path = [
baoDeploy.package
pkgs.jq
pkgs.openssl
pkgs.coreutils
pkgs.systemd
# `cmp`, which is NOT in coreutils. Without it the root-changed test
# below is `command not found` — 127, invisible under `if !`, and
# therefore always "changed", so the bundle rebuild it guards fired
# on every issuance instead of on a new root.
pkgs.diffutils
];
path = servicesCertPath;
# Sized like the store's own granting units, and for the same
# reason: under `seal = "shamir"` an operator unseals BY HAND, and
# the login below fails for as long as that takes. 2880 × 30s is
@ -722,26 +748,21 @@ in
Restart = "on-failure";
RestartSec = 30;
};
environment = {
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
}
// lib.optionalAttrs (cfg.baoClientCertFile != null) {
BAO_CLIENT_CERT = cfg.baoClientCertFile;
}
// lib.optionalAttrs (cfg.baoClientKeyFile != null) {
BAO_CLIENT_KEY = cfg.baoClientKeyFile;
}
# Absent means the system trust store, which is what a deployment
# with a real CA wants and what a self-signed one must not be left
# with.
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
BAO_CACERT = baoDeploy.serverCaFile;
};
environment = servicesCertEnvironment;
script = ''
set -euo pipefail
d=${lib.escapeShellArg cfg.stateDir}
install -d -m 0755 "$d"
# `swarm-services-cert-renew` runs this same script, and two runs
# issuing at once can leave one's key beside the other's leaf. The
# second waits and then finds the leaf fresh.
exec 9>"$d/.swarm-services-cert.lock"
if ! flock -w 600 9; then
echo "another swarm-services issuance has held $d/.swarm-services-cert.lock for 10 minutes" >&2
exit 1
fi
${leafCoverage}
if [ "$want_svc" = 0 ]; then
@ -749,14 +770,20 @@ in
exit 0
fi
# Same rule as the hive leaf's, at a different cadence: re-issue
# when the file is missing, within 30 days of expiry, or no
# longer carrying every configured name. The name check is what
# makes adding a swarm service take effect on the rebuild that
# added it rather than whenever the certificate happens to lapse.
# Re-issue when the file is missing, past half the lifetime the
# store's role grants it, or no longer carrying every configured
# name. The name check is what makes adding a swarm service take
# effect on the rebuild that added it rather than whenever the
# certificate happens to lapse.
#
# Half, as `hive-tls-resign` does for the hive leaf, so the daily
# renewal has half the lifetime to retry through a sealed store.
# Not the whole lifetime: a leaf issued with exactly that much left
# is inside it the second after, so every run re-issued.
renewat=$(( ${toString baoDeploy.servicesPkiLeafTtlHours} * 3600 / 2 ))
reissue=0
{ [ -s "$svcleaf" ] && [ -s "$d/swarm-services-key.pem" ]; } || reissue=1
openssl x509 -in "$svcleaf" -noout -checkend 2592000 >/dev/null 2>&1 || reissue=1
openssl x509 -in "$svcleaf" -noout -checkend "$renewat" >/dev/null 2>&1 || reissue=1
covers "$svcleaf" "$svcNames" || reissue=1
[ -s "$svcroot" ] || reissue=1
@ -764,11 +791,14 @@ in
# sides of this from needing to talk. The store replaces an issuer
# that has less than a leaf's window left (./swarm-bao.nix's
# regeneration guard) — so a hive testing its own copy of that same
# certificate against the same threshold asks for a new leaf on the
# same activation, and gets the replacement root back with it.
# Without this a leaf stays "valid" while the anchor it chains to no
# longer exists in the mount, which no expiry check would ever catch.
openssl x509 -in "$svcroot" -noout -checkend 2592000 >/dev/null 2>&1 || reissue=1
# certificate against the same threshold, read from the same option,
# asks for a new leaf on the same activation, and gets the
# replacement root back with it. Without this a leaf stays "valid"
# while the anchor it chains to no longer exists in the mount, which
# no expiry check would ever catch.
openssl x509 -in "$svcroot" -noout -checkend ${
toString (baoDeploy.servicesPkiLeafTtlHours * 3600)
} >/dev/null 2>&1 || reissue=1
if [ "$reissue" = 0 ]; then
echo "swarm-services leaf valid and covering the configured names — leaving it alone"
@ -877,12 +907,13 @@ in
mv -f "$svcroot.new" "$svcroot"
chmod 0644 "$svcroot"
# Propagation, for the RETRY path only. Ordered before both of
# these, so on a normal boot they have not run yet, `is-active`
# is false, and ordering alone does the work. What this covers is
# the store coming up hours after the gateway did: nginx serves a
# *copy* of the leaf, so re-issuing the source changes nothing
# until the copy is remade.
# Propagation, for the retry and renewal paths only. Ordered before
# both of these, so on a normal boot they have not run yet,
# `is-active` is false, and ordering alone does the work. What this
# covers is the store coming up hours after the gateway did, and
# `swarm-services-cert-renew` rotating the leaf under a running
# gateway: nginx serves a *copy* of the leaf, so re-issuing the
# source changes nothing until the copy is remade.
#
# ⚠️ `--no-block`, and it is not a preference. This unit declares
# `Before=` both of these, so a blocking `systemctl restart`
@ -917,6 +948,37 @@ in
'';
};
# The services leaf's renewal: `swarm-services-cert`'s own script, run
# again from the daily timer below.
#
# ⚠️ A unit of its own, not a timer on `swarm-services-cert`. That unit
# is `RemainAfterExit`, so once it has run it stays active and a timer
# starting it does nothing. And it cannot be restarted instead:
# `hive-gateway-self-signed-cert` `Requires=` it, so a restart restarts
# the gateway's cert import and nginx behind it, and a store that is
# sealed at that moment fails the start and takes both down over a leaf
# that was still valid. Nothing requires or orders against this unit, so
# its failure is its own.
#
# Failure leaves the leaf alone: every file the script writes is a
# `.new` until the store has answered with all three fields. No
# `Restart=`: the unit stays failed until the next tick, which is how
# a sealed store shows up in `systemctl --failed`.
systemd.services.swarm-services-cert-renew = {
description = "Renew the swarm-services TLS leaf from the secret store's PKI";
# The timer's `Persistent=` can fire it during boot. Behind the boot
# issuance, so it finds the leaf that run just wrote.
after = [ "swarm-services-cert.service" ];
path = servicesCertPath;
environment = servicesCertEnvironment;
inherit (config.systemd.services.swarm-services-cert) script;
serviceConfig = {
Type = "oneshot";
UMask = "0077";
SyslogIdentifier = "swarm-services-cert-renew";
};
};
systemd.timers.hive-tls-resign = {
description = "Weekly gateway-leaf re-sign and propagation";
wantedBy = [ "timers.target" ];
@ -929,6 +991,19 @@ in
};
};
# Daily rather than `hive-tls-resign`'s weekly, for the reason
# ./swarm-nats.nix's `swarm-bao-nats-tls` timer is: this leaf needs the
# store, which can be sealed or unreachable, so there is a retry per day
# from half-life to expiry rather than two.
systemd.timers.swarm-services-cert-renew = {
description = "Daily renewal check for the swarm-services TLS leaf";
wantedBy = [ "timers.target" ];
timerConfig = {
OnCalendar = "daily";
Persistent = true;
};
};
# Signal the hive-c0re lifecycle that a hive CA exists: it bind-mounts
# this file (read-only, public certs ONLY — never a key) into each
# agent container so agents + their tools can trust the gateway's