hyperhive/nix/host-modules/hive-tls.nix
atlas 41f0d7e036 fix(#3462): apply the name check in the unit that runs on the deploy
hive-tls-ca re-signs at service activation -- the rebuild itself --
while hive-tls-resign only fires from a weekly timer. The previous commit
put the coverage check in the timer unit, so a corrected serviceDomains
would not have taken effect until up to a week after the deploy that
changed it. Same bug one level along: found a trigger, not the trigger.

Two guards now share one definition rather than each carrying their own,
because a rule enforced in one and not the other is worse than one
enforced in neither -- it looks fixed and only fires on whichever path
you did not take.

Also widens hive-tls-ca's condition to the services leaf. Both leaves are
signed inside that block but only the hive leaf gated it, so a fresh
gateway.pem suppressed the re-signing of a swarm-services.pem that was
missing or stale.
2026-08-18 21:54:38 +02:00

674 lines
32 KiB
Nix
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

{
lib,
config,
pkgs,
...
}:
let
cfg = config.services.hyperhive.tls;
hyperhiveCfg = config.services.hyperhive;
gatewayCfg = config.services.hyperhive.gateway;
swarmCaCfg = config.services.hyperhive.swarm.ca;
domain = hyperhiveCfg.domain;
# Derived once in ./swarm.nix and read here + in ./swarm-ca.nix, so
# the names this leaf carries as SANs and the names the sub-CA is
# constrained to cannot disagree.
swarmServiceDomains = hyperhiveCfg.swarm.serviceDomains;
# True for the names `signHiveLeaf` below actually covers: the hive
# domain itself, or ONE label under it. `*.<domain>` is a single-label
# wildcard — `a.b.<domain>` does not match it — so the depth check is
# the whole point rather than a nicety.
coveredByHiveLeaf =
n:
n == domain
|| (lib.hasSuffix ".${domain}" n && !lib.hasInfix "." (lib.removeSuffix ".${domain}" n));
# Service names this host can serve a *matching* certificate for, and
# the ones it can't. A name is issuable here when the hive leaf covers
# it, or when this host issues the swarm-services leaf — which needs
# the swarm root's private key, i.e. `swarm.ca.autoConfigure`.
uncoveredServiceDomains = lib.filter (n: !coveredByHiveLeaf n) swarmServiceDomains;
# The host-managed hive CA is the trust anchor for self-signed mode.
# It is only stood up when the gateway actually serves a self-signed
# cert: the gateway must be in self-signed mode. `domain` is required
# (asserted in hive-network.nix), so the leaf SANs always have a
# domain to derive from. The self-signed condition is the gateway
# module's single source of truth (`gateway.useSelfSigned`): true when
# neither an operator cert (`tls.certDir`) nor ACME is set.
active = hyperhiveCfg.enable && gatewayCfg.useSelfSigned;
# How this hive's CA comes into existence when it is missing — and it
# is one of exactly two things, chosen by config rather than by what
# happens to be on disk.
#
# Issuing a sub-CA requires the swarm root's PRIVATE key, so it can
# only happen where that key legitimately lives: the single-host
# deployment that `swarm.ca.autoConfigure` describes. A host cannot
# infer that it is that host — a swarm's services and its hives can
# sit anywhere — so with the flag off this hive self-signs exactly as
# it always has, and an operator who wants it in the hierarchy
# installs the CA themselves. Falling back to self-signed rather than
# failing keeps a plain hive working out of the box; what it loses is
# membership of a swarm's trust, which is the correct thing to lose
# for a hive nobody has federated.
caGenScript =
if swarmCaCfg.autoConfigure then
''
# The root should exist `swarm-ca.service` runs before this and
# is required by it. If it doesn't, something upstream failed and
# signing with a half-provisioned root would be worse than stopping.
if [ ! -s "$root" ] || [ ! -s "$rootk" ]; then
echo "swarm.ca.autoConfigure is set but there is no root CA key at $rootk" >&2
echo "(swarm-ca.service should have generated it) refusing to issue a hive CA." >&2
exit 1
fi
echo "issuing fresh hive CA at $ca under the swarm root"
cacsr="$(mktemp "$d/ca.csr.XXXXXX")"
caext="$(mktemp "$d/ca.ext.XXXXXX")"
trap 'rm -f "$cacsr" "$caext"' EXIT
openssl req -newkey rsa:4096 -nodes -sha256 \
-keyout "$cak" -out "$cacsr" \
-subj "/CN=hive-ca ${domain}"
# printf (not a heredoc) so the ext-file lines carry no leading
# whitespace once nix has stripped the indented-string indent.
{
printf 'basicConstraints=critical,CA:TRUE,pathlen:0\n'
printf 'keyUsage=critical,keyCertSign,cRLSign\n'
printf 'subjectKeyIdentifier=hash\n'
printf 'authorityKeyIdentifier=keyid:always\n'
# The constraint is the point of the hierarchy, not a
# flourish: without it a leaked hive CA mints any name in
# the swarm, and it is verifiers that enforce this, not our
# good behaviour. The IP exclusions are not redundant a
# DNS constraint says nothing about an iPAddress SAN, and a
# name type nobody constrained is a name type this CA is
# unconstrained for.
printf 'nameConstraints=critical,permitted;DNS:%s,excluded;IP:0.0.0.0/0.0.0.0,excluded;IP:0:0:0:0:0:0:0:0/0:0:0:0:0:0:0:0\n' \
${lib.escapeShellArg domain}
} > "$caext"
openssl x509 -req -in "$cacsr" -CA "$root" -CAkey "$rootk" \
-CAcreateserial -days ${toString cfg.caValidityDays} -sha256 \
-extfile "$caext" -out "$ca"
rm -f "$cacsr" "$caext"
trap - EXIT
''
else
''
echo "generating fresh self-signed hive CA at $ca"
openssl req -x509 -newkey rsa:4096 -nodes -sha256 \
-days ${toString cfg.caValidityDays} \
-keyout "$cak" -out "$ca" \
-subj "/CN=hive-ca ${domain}" \
-addext "basicConstraints=critical,CA:TRUE,pathlen:0" \
-addext "keyUsage=critical,keyCertSign,cRLSign"
'';
# Sign one leaf. Parameterised rather than hardcoded to `gateway.*`
# because there are now two: the hive's own leaf, issued by the hive
# CA, and the swarm-services leaf, issued by the services sub-CA that
# ./swarm-ca.nix maintains. Same ceremony, different issuer and names
# — and one script means the two cannot drift in how they are built.
#
# $1 stateDir $2 basename $3 CN $4 SAN list $5 issuer cert $6 issuer key
signLeafScript = pkgs.writeShellScript "hive-tls-sign-leaf" ''
set -euo pipefail
d="$1"
base="$2"
cn="$3"
sans="$4"
ca="$5"
cak="$6"
leaf="$d/$base.pem"
leafk="$d/$base-key.pem"
csr="$(mktemp "$d/$base.csr.XXXXXX")"
ext="$(mktemp "$d/$base.ext.XXXXXX")"
only="$(mktemp "$d/$base.leaf.XXXXXX")"
trap 'rm -f "$csr" "$ext" "$only"' EXIT
openssl req -newkey rsa:4096 -nodes -sha256 \
-keyout "$leafk" -out "$csr" \
-subj "/CN=$cn"
# printf (not a heredoc) so the ext-file lines carry no leading
# whitespace once nix has stripped the indented-string indent.
{
printf 'subjectAltName=%s\n' "$sans"
printf 'basicConstraints=critical,CA:FALSE\n'
printf 'keyUsage=critical,digitalSignature,keyEncipherment\n'
printf 'extendedKeyUsage=serverAuth\n'
} > "$ext"
openssl x509 -req -in "$csr" -CA "$ca" -CAkey "$cak" \
-CAcreateserial -days ${toString cfg.leafValidityDays} -sha256 \
-extfile "$ext" -out "$only"
# nginx serves this file verbatim, so it must carry the leaf AND its
# issuer: the hive CA is an intermediate under the swarm root, and a
# client that anchors on the root cannot build the middle of the
# chain by itself. Agents anchor on the hive CA directly and
# validated either way the appended cert is what makes a swarm
# peer, or anything else holding only the root, work.
# Leaf first: both `ssl_certificate` and the `openssl x509 -in`
# expiry checks read the first cert in the file.
cat "$only" "$ca" > "$leaf"
chmod 0600 "$leafk"
chmod 0644 "$leaf"
'';
# The hive's own leaf: signed by the hive CA, covering the hive domain
# and its sub-domains.
#
# `forge.` and `matrix.` were listed here explicitly and both were
# redundant with the wildcard beside them — a wildcard covers exactly
# one label, and those are one label. Naming them read as policy
# ("these two services live under this leaf"), which is the reason they
# outlived it: a configured service name is no longer necessarily under
# this domain, and when it isn't, this is the one list it cannot join.
# The hive CA is `nameConstraints=permitted;DNS:<hive domain>`, and a
# violation invalidates the CERTIFICATE rather than the offending name
# — measured with openssl, not reasoned about — so one sibling name
# added here takes the leaf down for the dashboard and every other
# vhost sharing it. Configured service names get the swarm-services
# leaf below, whose sub-CA is constrained to exactly those names.
signHiveLeaf = ''
${signLeafScript} "$d" gateway ${lib.escapeShellArg domain} \
${lib.escapeShellArg "DNS:${domain},DNS:*.${domain}"} \
"$d/ca.pem" "$d/ca-key.pem"
'';
# The swarm-services leaf: signed by the services sub-CA, covering the
# swarm's service names. Those are *siblings* of the hive domain, not
# children, so the hive CA is name-constrained out of them and cannot
# sign this however its SAN list is written.
#
# Skipped when the sub-CA isn't on disk: it only exists where the
# swarm CA is autoconfigured, and a hive that gets its certs from its
# operator has nothing for this to do.
signServicesLeaf = lib.optionalString (swarmServiceDomains != [ ]) ''
servicesCa=${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca.pem"}
servicesCaKey=${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca-key.pem"}
if [ -s "$servicesCa" ] && [ -s "$servicesCaKey" ]; then
${signLeafScript} "$d" swarm-services \
${lib.escapeShellArg (builtins.head swarmServiceDomains)} \
${lib.escapeShellArg (lib.concatMapStringsSep "," (n: "DNS:${n}") swarmServiceDomains)} \
"$servicesCa" "$servicesCaKey"
else
echo "no swarm-services sub-CA at $servicesCa skipping the services leaf"
fi
'';
# Shared by BOTH units that decide whether to re-sign. It lives here
# rather than in one of them because the two guards have to agree: they
# answer the same question at different times (`hive-tls-ca` at service
# activation, i.e. on the deploy; `hive-tls-resign` from a weekly timer
# for a host that stays up long enough to drift). A rule implemented in
# one and not the other is worse than one implemented in neither — it
# looks fixed and only fires on whichever path you did not take, which
# is exactly how a corrected `serviceDomains` still served a stale leaf.
#
# Expects `$d` (state dir) to be set; defines `$leaf`-adjacent names and
# `covers`.
leafCoverage = ''
svcleaf="$d/swarm-services.pem"
hiveNames=${lib.escapeShellArg "${domain} *.${domain}"}
svcNames=${lib.escapeShellArg (lib.concatStringsSep " " swarmServiceDomains)}
# The services leaf is only expected where the sub-CA exists;
# elsewhere its absence is the correct state, not a stale leaf.
want_svc=${if swarmServiceDomains == [ ] then "0" else "1"}
if [ ! -s ${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca.pem"} ]; then
want_svc=0
fi
# Expiry is not the only way a leaf goes wrong. One signed when the
# configured name set was smaller stays valid for its whole lifetime
# while omitting every name added since so a config change can
# evaluate, build and deploy cleanly while the gateway keeps serving a
# certificate that does not cover the new service. The services sub-CA
# already reconciles this way (its `.names` comparison in
# ../swarm-ca.nix); the leaves did not.
#
# Reads the names out of the CERTIFICATE, not a sidecar file: the pem
# is what nginx serves, and a bookkeeping file drifts from it the
# moment anyone replaces a leaf by hand.
covers() { # $1 = leaf, $2 = space-separated names it must carry
[ -s "$1" ] || return 1
_have="$(openssl x509 -in "$1" -noout -ext subjectAltName 2>/dev/null \
| tr ',' '\n' | sed -n 's/^[[:space:]]*DNS:\(.*\)$/\1/p' || true)"
[ -n "$_have" ] || return 1
_missing=0
# The hive leaf carries `*.<domain>`; without this the shell
# expands it against the cwd and the comparison silently tests a
# filename instead of a name.
set -f
for _n in $2; do
printf '%s\n' "$_have" | grep -qxF "$_n" || _missing=1
done
set +f
[ "$_missing" = 0 ]
}
# True when every leaf this host is supposed to hold is present and
# carries its configured names. Expiry is the callers' own business
# they use different windows.
leavesCoverNames() {
covers "$leaf" "$hiveNames" || return 1
[ "$want_svc" = 0 ] && return 0
covers "$svcleaf" "$svcNames"
}
'';
in
{
# Host-side TLS trust root for the self-signed gateway mode.
#
# A bare self-signed leaf would be its own trust anchor, so every
# regeneration would be a new anchor and every consumer (agents,
# federation peers) would have to re-trust on each rotation — and a
# runtime-generated, in-container cert can't be wired into an agent's
# build-time trust store at all.
#
# So the issuer is a long-lived **hive CA** held on the host. The
# gateway serves a **leaf** signed by that CA (via the `tls.certDir`
# bind-mount path); agents and federation peers trust the *CA* once,
# and leaf rotation never re-breaks them. See `docs/gateway.md`
# ("Self-signed TLS").
#
# That CA is self-signed by default. Under
# `swarm.ca.autoConfigure` — the all-on-one-host case — it is instead
# an intermediate issued under the swarm root (./swarm-ca.nix owns the
# why) and name-constrained to this hive's domain, so it stays the
# anchor agents pin while a peer holding only the root can validate
# everything this hive serves. Either way an existing CA is left
# alone; see the issuance comment below.
options.services.hyperhive.tls = {
stateDir = lib.mkOption {
type = lib.types.str;
default = "/var/lib/hive-tls";
description = ''
Host directory holding the hive CA + gateway leaf cert for the
self-signed gateway mode. `ca.pem` (the anchor agents and
federation peers trust), `ca-key.pem` (0600, never leaves the
host), `gateway.pem` / `gateway-key.pem` (the leaf the gateway
container bind-mounts and nginx serves). Persistent so the CA
survives reboots re-deriving it would re-break every consumer.
'';
};
caValidityDays = lib.mkOption {
type = lib.types.int;
default = 7300;
description = ''
Validity window of the hive CA in days (default ~20y). Kept long
and well beyond `leafValidityDays` so the CA outlives many leaf
rotations the whole point of the CA is to be a stable anchor
that consumers trust once. The CA is regenerated only if missing
or already expired.
'';
};
leafValidityDays = lib.mkOption {
type = lib.types.int;
default = 30;
description = ''
Validity window of the gateway leaf cert in days (default 30).
Short-lived by design ahead of the CA/Browser-Forum's move
toward ~47-day max lifetimes which bounds the blast radius of a
leaf-key compromise. The leaf is re-signed by the (stable) CA
when it is missing or near expiry; because it shares the CA
anchor, a rotation does not disturb consumer trust. Agents and
federation peers validate against the CA, not browser CA/B-forum
limits. The weekly `hive-tls-resign` timer re-signs the leaf once
it is within half its validity of expiry and propagates the new
leaf into the running gateway, so a long-uptime host renews
automatically without a reboot.
'';
};
};
config = lib.mkIf active {
# A swarm service name that is a sibling of the hive domain rather
# than a child needs the swarm-services leaf, and this host only
# signs that one when it holds the swarm root key. Without it nginx
# falls back to the hive leaf on those names and every client sees a
# name mismatch — on a config that evaluates and deploys cleanly.
#
# ⚠️ A WARNING, NOT AN ASSERTION, and the distinction is the point:
# this module can see what *it* can issue; it cannot see an
# operator-installed services sub-CA, an external ACME setup, or a
# cert delivered by any other means. A rebuild must not be blocked
# by a conclusion this host is not in a position to reach. Report
# the observation and let the operator judge it.
#
# (`certDir` / ACME modes don't reach here at all: `active` is
# self-signed-only, and there the operator's cert decides.)
warnings = lib.optional (uncoveredServiceDomains != [ ] && !swarmCaCfg.autoConfigure) ''
services.hyperhive: these swarm service names are outside this
hive's domain (${domain}), so the hive CA's leaf does not cover
them:
${lib.concatMapStringsSep "\n" (n: " ${n}") uncoveredServiceDomains}
services.hyperhive.swarm.ca.autoConfigure is false, so this host
does not sign the swarm-services leaf either, and the gateway will
serve the hive leaf on those names a certificate-name mismatch
for browsers and for agents' git-over-https.
If you have already arranged certificates for them an
operator-installed services sub-CA in ${swarmCaCfg.stateDir}, or
services.hyperhive.gateway.tls.{certDir,acme} this is expected
and you can ignore it. Otherwise pin the names back under
${domain} (services.hyperhive.swarm.{forge.domain,
matrix.gatewayHost, authelia.domain}) or install the sub-CA.
'';
# Generate (and rotate) the hive CA + gateway leaf before anything
# serves it. Idempotent: the CA is created once and reused; the leaf
# is re-signed on expiry under the same CA so the anchor is stable.
systemd.services.hive-tls-ca = {
description = "Generate hive CA + gateway leaf TLS cert (self-signed mode)";
wantedBy = [ "multi-user.target" ];
# The consumer is `hive-gateway-self-signed-cert`, which copies the
# leaf into the gateway's state dir at the mode nginx can read, and
# which nginx in turn `Requires=`. So this must run first or that
# copy fails under `set -eu` and takes nginx down with it — order
# against the unit that reads the file.
before = [ "hive-gateway-self-signed-cert.service" ];
requiredBy = [ "hive-gateway-self-signed-cert.service" ];
# The issuance below needs the swarm root key on disk, and (for the
# services leaf) the services sub-CA it signs under. When this host
# generates them (single-host swarm) both units must have run first;
# when the operator provides the material there is no unit to wait
# for, so the dependency is conditional rather than a unit that
# exists and does nothing.
#
# Without waiting for swarm-services-ca specifically, this unit races
# it: if hive-tls-ca finishes first, it finds no services-ca.pem yet,
# silently skips signing the services leaf (the same as "operator
# hasn't set one up"), and the gateway comes up with a vhost pointed
# at a cert that was never written.
after = lib.optionals swarmCaCfg.autoConfigure [
"swarm-ca.service"
"swarm-services-ca.service"
];
requires = lib.optionals swarmCaCfg.autoConfigure [
"swarm-ca.service"
"swarm-services-ca.service"
];
path = [ pkgs.openssl ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
UMask = "0077";
# Pin the journal identity (else it's the `script` store-path wrapper).
SyslogIdentifier = "hive-tls-ca";
};
script = ''
set -euo pipefail
d=${lib.escapeShellArg cfg.stateDir}
install -d -m 0755 "$d"
ca="$d/ca.pem"
cak="$d/ca-key.pem"
leaf="$d/gateway.pem"
leafk="$d/gateway-key.pem"
root=${lib.escapeShellArg "${swarmCaCfg.stateDir}/root.pem"}
rootk=${lib.escapeShellArg "${swarmCaCfg.stateDir}/root-key.pem"}
prev="$d/ca-previous.pem"
marker="$d/.swarm-ca-adopted"
# --- Adoption: a hive whose CA predates the swarm root.
#
# Runs ONLY where this host also owns the root (`autoConfigure`),
# because adoption invalidates an anchor that consumers already
# trust and they refresh on their own schedule. On one box that
# schedule is knowable; across hosts it is not, so there the
# operator does it and this unit only says so, loudly.
#
# Guarded by a marker rather than by the state of the world: the
# marker's ABSENCE is the trigger, so this fires once per hive
# instead of re-deciding every activation. That is the difference
# between a migration and a kicking machine.
if [ -s "$root" ] && [ -s "$ca" ] && [ ! -e "$marker" ] \
&& ! openssl verify -CAfile "$root" "$ca" >/dev/null 2>&1; then
${
if swarmCaCfg.autoConfigure then
''
echo "adopting the swarm CA: $ca does not chain to $root" >&2
# Keep the old CA as an anchor across the overlap. Consumers
# read the bundle, so adoption is additive before it is
# subtractive agents pick up new trust only when their
# container restarts, which is a window even on one host.
cp "$ca" "$prev"
chmod 0644 "$prev"
rm -f "$ca" "$cak"
touch "$marker"''
else
''
echo "hive-tls: this hive's CA does not chain to the swarm root." >&2
echo " hive CA: $ca" >&2
echo " swarm root: $root" >&2
echo "Adopting the hierarchy is not automatic here: it invalidates an" >&2
echo "anchor that peers and agents on OTHER hosts still trust, and they" >&2
echo "refresh on their own schedule only you know when that is safe." >&2
echo "To adopt: rm $ca $cak && systemctl restart hive-tls-ca.service" >&2
echo "To keep the current CA deliberately: touch $marker" >&2
exit 1''
}
fi
# --- CA: generated once and reused across leaf rotations, in
# whichever of the two shapes `caGenScript` selected.
# Regenerated only if missing or already expired (checkend 0).
# A new CA means every consumer must re-trust, so the leaf is
# dropped to force a re-sign under the fresh CA.
#
# An existing CA is never re-rooted here. A hive predating the
# swarm root carries a self-signed `ca.pem`, and swapping it for
# a swarm-issued one would break every consumer already trusting
# it including agents, whose trust is refreshed only when their
# container restarts. Adopting the hierarchy on such a hive is
# therefore an operator step (drop the CA, let this unit
# re-issue), which is also what keeps this change non-disruptive
# to hives that never adopt it.
if [ ! -s "$ca" ] || [ ! -s "$cak" ] \
|| ! openssl x509 -in "$ca" -noout -checkend 0 >/dev/null 2>&1; then
${caGenScript}
chmod 0600 "$cak"
chmod 0644 "$ca"
rm -f "$leaf" "$leafk"
fi
# --- Leaf: (re)sign when missing, within 30 days of expiry, or no
# longer covering the configured names, always under the current
# (stable) CA.
${leafCoverage}
# This is the guard that runs ON THE DEPLOY `hive-tls-resign`
# only fires from a weekly timer, so a rule enforced only there is
# up to a week late and does nothing for the rebuild that changed
# the names in the first place.
#
# The name check has to include the SERVICES leaf even though the
# condition is written around the hive one, because both are signed
# in this block: a fresh `gateway.pem` otherwise suppresses the
# re-sign of a `swarm-services.pem` that is missing or stale, which
# is what left a corrected `serviceDomains` still mis-served.
resign=0
{ [ -s "$leaf" ] && [ -s "$leafk" ]; } || resign=1
openssl x509 -in "$leaf" -noout -checkend 2592000 >/dev/null 2>&1 || resign=1
leavesCoverNames || resign=1
if [ "$want_svc" = 1 ] && [ ! -s "$svcleaf" ]; then
resign=1
fi
if [ "$resign" = 1 ]; then
echo "signing gateway leaf at $leaf (missing, near expiry, or missing a configured name)"
${signHiveLeaf}
${signServicesLeaf}
fi
# --- Trust bundle: what a consumer must TRUST, as opposed to
# `ca.pem`, which is what this host SIGNS with. Why they stopped
# being the same file, and why `ca-previous.pem` stays in the set
# after an adoption, are in docs/swarm/ca.md. Three constraints
# the code cannot state:
#
# Written IN PLACE, never renamed into position containers
# bind-mount this file and a bind mount follows the inode, so a
# rename leaves every consumer holding the old one.
#
# Dropping `ca-previous.pem` is deliberately NOT done here:
# "long enough" is a deployment fact, not a unit's call.
#
# Built as an array with `if` guards, not
# `[ -s f ] && anchors+=(f)`: under `set -e` an && list whose test
# fails IS a failing command and kills the unit on exactly the
# hive where the optional file is legitimately absent. The
# reflexive `|| true` is worse; it also swallows a real failure to
# read the hive CA.
bundle="$d/trust-bundle.pem"
anchors=("$ca")
if [ -s "$prev" ]; then anchors+=("$prev"); fi
if [ -s "$root" ]; then anchors+=("$root"); fi
cat "''${anchors[@]}" > "$bundle"
chmod 0644 "$bundle"
'';
};
# Weekly re-sign of the gateway leaf so short-lived leaves renew
# without depending on a reboot.
#
# `hive-tls-ca` only re-signs at service activation (boot/rebuild); a
# long-uptime host would otherwise let a 30-day leaf lapse silently.
# This service re-signs the leaf directly (not by bouncing hive-tls-ca)
# and propagates the new leaf to nginx when the file actually changed.
#
# Propagation mechanism: nginx serves a *copy* of the leaf, written by
# `hive-gateway-self-signed-cert` at the mode nginx can read. Re-signing
# the source therefore changes nothing on its own — the copy has to be
# remade and nginx reloaded, which is what the two calls below do.
#
# ⚠️ Neither call is `|| true`: a swallowed propagation failure means
# the leaf rotates on disk while nginx keeps serving the old copy
# until it expires, with this unit reporting success the whole time.
# A failure here must fail the timer.
systemd.services.hive-tls-resign = {
description = "Re-sign the gateway TLS leaf and reload nginx";
# hive-tls-ca must have run first so the CA key exists before we try
# to re-sign under it. On first boot `Persistent=true` on the weekly
# timer fires immediately; without this ordering the resign could race
# the CA initialisation and fail with "no such file" on the CA key.
after = [ "hive-tls-ca.service" ];
path = [
pkgs.openssl
pkgs.coreutils
pkgs.systemd
];
serviceConfig = {
Type = "oneshot";
UMask = "0077";
SyslogIdentifier = "hive-tls-resign";
};
script = ''
set -euo pipefail
d=${lib.escapeShellArg cfg.stateDir}
leaf="$d/gateway.pem"
# EVERY leaf this host issues must be covered here. A leaf that
# first-boot issuance creates and this unit does not know about
# looks perfect for its entire validity and then expires with no
# warning the failure is invisible until it is total. `svcleaf`
# and the name checks come from the shared snippet, so this unit
# and `hive-tls-ca` cannot disagree about what a good leaf is.
${leafCoverage}
# Re-sign only when a leaf is within half its validity of expiry.
# The weekly cadence catches this window well before one lapses.
halflife=$(( ${toString cfg.leafValidityDays} * 86400 / 2 ))
fresh() { # a leaf is fresh if it exists and is not near expiry
[ -s "$1" ] && openssl x509 -in "$1" -noout -checkend "$halflife" >/dev/null 2>&1
}
if fresh "$leaf" && { [ "$want_svc" = 0 ] || fresh "$svcleaf"; } \
&& leavesCoverNames; then
echo "leaves valid, and covering the configured names no resign needed"
exit 0
fi
echo "a leaf is missing, near expiry, or missing a configured name re-signing"
before="$(sha256sum "$leaf" "$svcleaf" 2>/dev/null || true)"
${signHiveLeaf}
${signServicesLeaf}
after="$(sha256sum "$leaf" "$svcleaf" 2>/dev/null || true)"
if [ "$before" != "$after" ]; then
echo "gateway leaf rotated re-importing and reloading nginx"
systemctl restart hive-gateway-self-signed-cert.service
systemctl reload nginx.service
else
echo "gateway leaf unchanged (already up to date)"
fi
'';
};
systemd.timers.hive-tls-resign = {
description = "Weekly gateway-leaf re-sign and propagation";
wantedBy = [ "timers.target" ];
timerConfig = {
# Run weekly; Persistent=true fires a missed run on next boot if
# the timer was not active (e.g. the host was off on the scheduled
# day), preventing a dormant timer from letting the leaf lapse.
OnCalendar = "weekly";
Persistent = true;
};
};
# Signal the hive-c0re lifecycle that a hive CA exists: it bind-mounts
# this file (read-only, public certs ONLY — never a key) into each
# agent container so agents + their tools can trust the gateway's
# self-signed leaf, and the meta flake wires the per-agent trust
# bundle. It is the ANCHOR bundle rather than `ca.pem` for the reason
# spelled out where the bundle is written above; no key path is ever
# exposed (an agent that could read one could mint trusted certs).
systemd.services.hive-c0re.environment.HIVE_TLS_CA_PATH = "${cfg.stateDir}/trust-bundle.pem";
# The same anchor, named for the swarm-queue clients that need it when
# they mint a token from authelia over TLS. Declared HERE, beside the
# bundle, rather than in each consumer's module: the path is this
# module's fact, and two consumers re-deriving `${stateDir}/…` would be
# two places to fix the day it moves.
#
# ⚠️ This is what was missing. `swarm-queue-client` built a bare
# `reqwest::Client`, so it trusted only the platform roots and died at
# `invalid peer certificate: UnknownIssuer` against a swarm whose
# authelia is signed by the swarm CA — while this very bundle sat on
# disk, already assembled, already handed to hive-c0re under a different
# variable name. The anchor was never missing; nothing pointed the queue
# client at it.
#
# Set unconditionally within this module's `active` guard, exactly like
# the line above: where there is no hive CA this module contributes
# nothing at all, and the clients then fall back to the platform roots —
# which is correct for a swarm fronted by a public certificate.
systemd.services.hive-c0re.environment.HIVE_C0RE_OIDC_CA_FILE = "${cfg.stateDir}/trust-bundle.pem";
# ⚠️ Gated, where the hive-c0re line above is not, and the asymmetry is
# the point: hive-c0re runs on every hive, the controller runs on one.
# Defining an environment key on a unit that does not exist CREATES a
# unit fragment for it — inert (no `ExecStart`, empty `wantedBy`, never
# activated) but present on every non-controller hive with a CA. Caught
# in review on this PR; it evaluates and builds clean either way, which
# is exactly why it needed a reviewer rather than a check.
systemd.services.swarm-controller.environment.SWARM_CONTROLLER_OIDC_CA_FILE =
lib.mkIf hyperhiveCfg.swarm.controller.enable "${cfg.stateDir}/trust-bundle.pem";
};
}