hive-tls: renew the swarm-services leaf on a daily timer
The store's `swarm-services` role issues the services leaf for 720h, and `swarm-services-cert` only ever ran at boot or rebuild: it is a `RemainAfterExit` oneshot wanted by `multi-user.target` and no timer targeted it. A hive not rebuilt within 30 days served an expired leaf. `swarm-services-cert-renew` runs the same script from a daily timer. It is a unit of its own because a timer starting the `RemainAfterExit` unit is a no-op, and restarting that unit instead would propagate through `hive-gateway-self-signed-cert`'s `Requires=` to nginx, so a sealed store would take the gateway down over a still-valid leaf. Nothing requires or orders against the new unit; it has no `Restart=`, so a failure stays in `systemctl --failed` until the next tick, and the script only moves files into place after the store has answered. The re-issue threshold was `checkend 2592000`, the whole 30-day lifetime, so every run re-issued. It is now half the role's lifetime, read from a new internal option `deploy.bao.servicesPkiLeafTtlHours` that the role's `ttl`/`max_ttl` also read. Boot and timer share the script and so the threshold. The services-root re-check reads the same option, at the store's own replacement threshold (hours × 3600), so the hive asks for a new leaf when the store replaces its root. A `flock` keeps the two runs from interleaving one issuance's key with another's leaf. `checks.module-eval-hive-tls` pins the timer, that the unit it starts re-runs the issuance without `RemainAfterExit`, that nothing depends on it, and that both the leaf and root thresholds move with the option. Closes #4587
This commit is contained in:
parent
afde9f380f
commit
0649673ebf
5 changed files with 283 additions and 50 deletions
|
|
@ -247,6 +247,40 @@ let
|
|||
covers "$leaf" "$hiveNames"
|
||||
}
|
||||
'';
|
||||
|
||||
# What `swarm-services-cert` and `swarm-services-cert-renew` both run the
|
||||
# issuance with. Bound here rather than read back off the first unit: its
|
||||
# merged `path` already carries systemd's defaults, so the second would
|
||||
# render a different `PATH` and fail to merge.
|
||||
servicesCertPath = [
|
||||
baoDeploy.package
|
||||
pkgs.jq
|
||||
pkgs.openssl
|
||||
pkgs.coreutils
|
||||
pkgs.systemd
|
||||
# `cmp`, which is NOT in coreutils. Without it the root-changed test
|
||||
# below is `command not found` — 127, invisible under `if !`, and
|
||||
# therefore always "changed", so the bundle rebuild it guards fired
|
||||
# on every issuance instead of on a new root.
|
||||
pkgs.diffutils
|
||||
# `flock`, for the lock at the top of the script.
|
||||
pkgs.util-linux
|
||||
];
|
||||
servicesCertEnvironment = {
|
||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||
}
|
||||
// lib.optionalAttrs (cfg.baoClientCertFile != null) {
|
||||
BAO_CLIENT_CERT = cfg.baoClientCertFile;
|
||||
}
|
||||
// lib.optionalAttrs (cfg.baoClientKeyFile != null) {
|
||||
BAO_CLIENT_KEY = cfg.baoClientKeyFile;
|
||||
}
|
||||
# Absent means the system trust store, which is what a deployment
|
||||
# with a real CA wants and what a self-signed one must not be left
|
||||
# with.
|
||||
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
|
||||
BAO_CACERT = baoDeploy.serverCaFile;
|
||||
};
|
||||
in
|
||||
{
|
||||
# Host-side TLS trust root for the self-signed gateway mode.
|
||||
|
|
@ -367,11 +401,14 @@ in
|
|||
# rebuild of every hive that is not the CA host, about a fallback
|
||||
# that no longer happens.
|
||||
|
||||
# Both run on the deploy and sit on the gateway's start path, so a
|
||||
# failure in either takes TLS down for every service behind it.
|
||||
# The first two run on the deploy and sit on the gateway's start path,
|
||||
# so a failure in either takes TLS down for every service behind it.
|
||||
# The renewal sits on no start path, and a failure there is a leaf that
|
||||
# lapses days later unless someone reads why it failed now.
|
||||
services.hyperhive.swarm.otel.journaldUnits = [
|
||||
"hive-tls-ca"
|
||||
"swarm-services-cert"
|
||||
"swarm-services-cert-renew"
|
||||
];
|
||||
|
||||
# Generate (and rotate) the hive CA + gateway leaf before anything
|
||||
|
|
@ -598,8 +635,8 @@ in
|
|||
#
|
||||
# The services leaf is not one of them any longer: it is issued
|
||||
# by the secret store, whose key this unit does not hold and
|
||||
# cannot re-sign under. Renewing it is `swarm-services-cert`'s
|
||||
# job, at boot, which is a cadence its own issue owns.
|
||||
# cannot re-sign under. `swarm-services-cert-renew` renews it,
|
||||
# on its own daily timer.
|
||||
${leafCoverage}
|
||||
|
||||
# Re-sign only when a leaf is within half its validity of expiry.
|
||||
|
|
@ -694,18 +731,7 @@ in
|
|||
"swarm-bao-services-issuer-policy.service"
|
||||
];
|
||||
wants = [ "container@${baoCfg.machine}.service" ];
|
||||
path = [
|
||||
baoDeploy.package
|
||||
pkgs.jq
|
||||
pkgs.openssl
|
||||
pkgs.coreutils
|
||||
pkgs.systemd
|
||||
# `cmp`, which is NOT in coreutils. Without it the root-changed test
|
||||
# below is `command not found` — 127, invisible under `if !`, and
|
||||
# therefore always "changed", so the bundle rebuild it guards fired
|
||||
# on every issuance instead of on a new root.
|
||||
pkgs.diffutils
|
||||
];
|
||||
path = servicesCertPath;
|
||||
# Sized like the store's own granting units, and for the same
|
||||
# reason: under `seal = "shamir"` an operator unseals BY HAND, and
|
||||
# the login below fails for as long as that takes. 2880 × 30s is
|
||||
|
|
@ -722,26 +748,21 @@ in
|
|||
Restart = "on-failure";
|
||||
RestartSec = 30;
|
||||
};
|
||||
environment = {
|
||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||
}
|
||||
// lib.optionalAttrs (cfg.baoClientCertFile != null) {
|
||||
BAO_CLIENT_CERT = cfg.baoClientCertFile;
|
||||
}
|
||||
// lib.optionalAttrs (cfg.baoClientKeyFile != null) {
|
||||
BAO_CLIENT_KEY = cfg.baoClientKeyFile;
|
||||
}
|
||||
# Absent means the system trust store, which is what a deployment
|
||||
# with a real CA wants and what a self-signed one must not be left
|
||||
# with.
|
||||
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
|
||||
BAO_CACERT = baoDeploy.serverCaFile;
|
||||
};
|
||||
environment = servicesCertEnvironment;
|
||||
script = ''
|
||||
set -euo pipefail
|
||||
d=${lib.escapeShellArg cfg.stateDir}
|
||||
install -d -m 0755 "$d"
|
||||
|
||||
# `swarm-services-cert-renew` runs this same script, and two runs
|
||||
# issuing at once can leave one's key beside the other's leaf. The
|
||||
# second waits and then finds the leaf fresh.
|
||||
exec 9>"$d/.swarm-services-cert.lock"
|
||||
if ! flock -w 600 9; then
|
||||
echo "another swarm-services issuance has held $d/.swarm-services-cert.lock for 10 minutes" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
${leafCoverage}
|
||||
|
||||
if [ "$want_svc" = 0 ]; then
|
||||
|
|
@ -749,14 +770,20 @@ in
|
|||
exit 0
|
||||
fi
|
||||
|
||||
# Same rule as the hive leaf's, at a different cadence: re-issue
|
||||
# when the file is missing, within 30 days of expiry, or no
|
||||
# longer carrying every configured name. The name check is what
|
||||
# makes adding a swarm service take effect on the rebuild that
|
||||
# added it rather than whenever the certificate happens to lapse.
|
||||
# Re-issue when the file is missing, past half the lifetime the
|
||||
# store's role grants it, or no longer carrying every configured
|
||||
# name. The name check is what makes adding a swarm service take
|
||||
# effect on the rebuild that added it rather than whenever the
|
||||
# certificate happens to lapse.
|
||||
#
|
||||
# Half, as `hive-tls-resign` does for the hive leaf, so the daily
|
||||
# renewal has half the lifetime to retry through a sealed store.
|
||||
# Not the whole lifetime: a leaf issued with exactly that much left
|
||||
# is inside it the second after, so every run re-issued.
|
||||
renewat=$(( ${toString baoDeploy.servicesPkiLeafTtlHours} * 3600 / 2 ))
|
||||
reissue=0
|
||||
{ [ -s "$svcleaf" ] && [ -s "$d/swarm-services-key.pem" ]; } || reissue=1
|
||||
openssl x509 -in "$svcleaf" -noout -checkend 2592000 >/dev/null 2>&1 || reissue=1
|
||||
openssl x509 -in "$svcleaf" -noout -checkend "$renewat" >/dev/null 2>&1 || reissue=1
|
||||
covers "$svcleaf" "$svcNames" || reissue=1
|
||||
[ -s "$svcroot" ] || reissue=1
|
||||
|
||||
|
|
@ -764,11 +791,14 @@ in
|
|||
# sides of this from needing to talk. The store replaces an issuer
|
||||
# that has less than a leaf's window left (./swarm-bao.nix's
|
||||
# regeneration guard) — so a hive testing its own copy of that same
|
||||
# certificate against the same threshold asks for a new leaf on the
|
||||
# same activation, and gets the replacement root back with it.
|
||||
# Without this a leaf stays "valid" while the anchor it chains to no
|
||||
# longer exists in the mount, which no expiry check would ever catch.
|
||||
openssl x509 -in "$svcroot" -noout -checkend 2592000 >/dev/null 2>&1 || reissue=1
|
||||
# certificate against the same threshold, read from the same option,
|
||||
# asks for a new leaf on the same activation, and gets the
|
||||
# replacement root back with it. Without this a leaf stays "valid"
|
||||
# while the anchor it chains to no longer exists in the mount, which
|
||||
# no expiry check would ever catch.
|
||||
openssl x509 -in "$svcroot" -noout -checkend ${
|
||||
toString (baoDeploy.servicesPkiLeafTtlHours * 3600)
|
||||
} >/dev/null 2>&1 || reissue=1
|
||||
|
||||
if [ "$reissue" = 0 ]; then
|
||||
echo "swarm-services leaf valid and covering the configured names — leaving it alone"
|
||||
|
|
@ -877,12 +907,13 @@ in
|
|||
mv -f "$svcroot.new" "$svcroot"
|
||||
chmod 0644 "$svcroot"
|
||||
|
||||
# Propagation, for the RETRY path only. Ordered before both of
|
||||
# these, so on a normal boot they have not run yet, `is-active`
|
||||
# is false, and ordering alone does the work. What this covers is
|
||||
# the store coming up hours after the gateway did: nginx serves a
|
||||
# *copy* of the leaf, so re-issuing the source changes nothing
|
||||
# until the copy is remade.
|
||||
# Propagation, for the retry and renewal paths only. Ordered before
|
||||
# both of these, so on a normal boot they have not run yet,
|
||||
# `is-active` is false, and ordering alone does the work. What this
|
||||
# covers is the store coming up hours after the gateway did, and
|
||||
# `swarm-services-cert-renew` rotating the leaf under a running
|
||||
# gateway: nginx serves a *copy* of the leaf, so re-issuing the
|
||||
# source changes nothing until the copy is remade.
|
||||
#
|
||||
# ⚠️ `--no-block`, and it is not a preference. This unit declares
|
||||
# `Before=` both of these, so a blocking `systemctl restart`
|
||||
|
|
@ -917,6 +948,37 @@ in
|
|||
'';
|
||||
};
|
||||
|
||||
# The services leaf's renewal: `swarm-services-cert`'s own script, run
|
||||
# again from the daily timer below.
|
||||
#
|
||||
# ⚠️ A unit of its own, not a timer on `swarm-services-cert`. That unit
|
||||
# is `RemainAfterExit`, so once it has run it stays active and a timer
|
||||
# starting it does nothing. And it cannot be restarted instead:
|
||||
# `hive-gateway-self-signed-cert` `Requires=` it, so a restart restarts
|
||||
# the gateway's cert import and nginx behind it, and a store that is
|
||||
# sealed at that moment fails the start and takes both down over a leaf
|
||||
# that was still valid. Nothing requires or orders against this unit, so
|
||||
# its failure is its own.
|
||||
#
|
||||
# Failure leaves the leaf alone: every file the script writes is a
|
||||
# `.new` until the store has answered with all three fields. No
|
||||
# `Restart=`: the unit stays failed until the next tick, which is how
|
||||
# a sealed store shows up in `systemctl --failed`.
|
||||
systemd.services.swarm-services-cert-renew = {
|
||||
description = "Renew the swarm-services TLS leaf from the secret store's PKI";
|
||||
# The timer's `Persistent=` can fire it during boot. Behind the boot
|
||||
# issuance, so it finds the leaf that run just wrote.
|
||||
after = [ "swarm-services-cert.service" ];
|
||||
path = servicesCertPath;
|
||||
environment = servicesCertEnvironment;
|
||||
inherit (config.systemd.services.swarm-services-cert) script;
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
UMask = "0077";
|
||||
SyslogIdentifier = "swarm-services-cert-renew";
|
||||
};
|
||||
};
|
||||
|
||||
systemd.timers.hive-tls-resign = {
|
||||
description = "Weekly gateway-leaf re-sign and propagation";
|
||||
wantedBy = [ "timers.target" ];
|
||||
|
|
@ -929,6 +991,19 @@ in
|
|||
};
|
||||
};
|
||||
|
||||
# Daily rather than `hive-tls-resign`'s weekly, for the reason
|
||||
# ./swarm-nats.nix's `swarm-bao-nats-tls` timer is: this leaf needs the
|
||||
# store, which can be sealed or unreachable, so there is a retry per day
|
||||
# from half-life to expiry rather than two.
|
||||
systemd.timers.swarm-services-cert-renew = {
|
||||
description = "Daily renewal check for the swarm-services TLS leaf";
|
||||
wantedBy = [ "timers.target" ];
|
||||
timerConfig = {
|
||||
OnCalendar = "daily";
|
||||
Persistent = true;
|
||||
};
|
||||
};
|
||||
|
||||
# Signal the hive-c0re lifecycle that a hive CA exists: it bind-mounts
|
||||
# this file (read-only, public certs ONLY — never a key) into each
|
||||
# agent container so agents + their tools can trust the gateway's
|
||||
|
|
|
|||
Loading…
Reference in a new issue