{ lib, config, pkgs, ... }: let cfg = config.services.hyperhive.tls; hyperhiveCfg = config.services.hyperhive; gatewayCfg = config.services.hyperhive.gateway; swarmCaCfg = config.services.hyperhive.swarm.ca; domain = hyperhiveCfg.domain; # Derived once in ./swarm.nix and read here + in ./swarm-ca.nix, so # the names this leaf carries as SANs and the names the sub-CA is # constrained to cannot disagree. swarmServiceDomains = hyperhiveCfg.swarm.serviceDomains; # True for the names `signHiveLeaf` below actually covers: the hive # domain itself, or ONE label under it. `*.` is a single-label # wildcard — `a.b.` does not match it — so the depth check is # the whole point rather than a nicety. coveredByHiveLeaf = n: n == domain || (lib.hasSuffix ".${domain}" n && !lib.hasInfix "." (lib.removeSuffix ".${domain}" n)); # Service names this host can serve a *matching* certificate for, and # the ones it can't. A name is issuable here when the hive leaf covers # it, or when this host issues the swarm-services leaf — which needs # the swarm root's private key, i.e. `swarm.ca.autoConfigure`. uncoveredServiceDomains = lib.filter (n: !coveredByHiveLeaf n) swarmServiceDomains; # The host-managed hive CA is the trust anchor for self-signed mode. # It is only stood up when the gateway actually serves a self-signed # cert: the gateway must be in self-signed mode. `domain` is required # (asserted in hive-network.nix), so the leaf SANs always have a # domain to derive from. The self-signed condition is the gateway # module's single source of truth (`gateway.useSelfSigned`): true when # neither an operator cert (`tls.certDir`) nor ACME is set. active = hyperhiveCfg.enable && gatewayCfg.useSelfSigned; # How this hive's CA comes into existence when it is missing — and it # is one of exactly two things, chosen by config rather than by what # happens to be on disk. # # Issuing a sub-CA requires the swarm root's PRIVATE key, so it can # only happen where that key legitimately lives: the single-host # deployment that `swarm.ca.autoConfigure` describes. A host cannot # infer that it is that host — a swarm's services and its hives can # sit anywhere — so with the flag off this hive self-signs exactly as # it always has, and an operator who wants it in the hierarchy # installs the CA themselves. Falling back to self-signed rather than # failing keeps a plain hive working out of the box; what it loses is # membership of a swarm's trust, which is the correct thing to lose # for a hive nobody has federated. caGenScript = if swarmCaCfg.autoConfigure then '' # The root should exist — `swarm-ca.service` runs before this and # is required by it. If it doesn't, something upstream failed and # signing with a half-provisioned root would be worse than stopping. if [ ! -s "$root" ] || [ ! -s "$rootk" ]; then echo "swarm.ca.autoConfigure is set but there is no root CA key at $rootk" >&2 echo "(swarm-ca.service should have generated it) — refusing to issue a hive CA." >&2 exit 1 fi echo "issuing fresh hive CA at $ca under the swarm root" cacsr="$(mktemp "$d/ca.csr.XXXXXX")" caext="$(mktemp "$d/ca.ext.XXXXXX")" trap 'rm -f "$cacsr" "$caext"' EXIT openssl req -newkey rsa:4096 -nodes -sha256 \ -keyout "$cak" -out "$cacsr" \ -subj "/CN=hive-ca ${domain}" # printf (not a heredoc) so the ext-file lines carry no leading # whitespace once nix has stripped the indented-string indent. { printf 'basicConstraints=critical,CA:TRUE,pathlen:0\n' printf 'keyUsage=critical,keyCertSign,cRLSign\n' printf 'subjectKeyIdentifier=hash\n' printf 'authorityKeyIdentifier=keyid:always\n' # The constraint is the point of the hierarchy, not a # flourish: without it a leaked hive CA mints any name in # the swarm, and it is verifiers that enforce this, not our # good behaviour. The IP exclusions are not redundant — a # DNS constraint says nothing about an iPAddress SAN, and a # name type nobody constrained is a name type this CA is # unconstrained for. printf 'nameConstraints=critical,permitted;DNS:%s,excluded;IP:0.0.0.0/0.0.0.0,excluded;IP:0:0:0:0:0:0:0:0/0:0:0:0:0:0:0:0\n' \ ${lib.escapeShellArg domain} } > "$caext" openssl x509 -req -in "$cacsr" -CA "$root" -CAkey "$rootk" \ -CAcreateserial -days ${toString cfg.caValidityDays} -sha256 \ -extfile "$caext" -out "$ca" rm -f "$cacsr" "$caext" trap - EXIT '' else '' echo "generating fresh self-signed hive CA at $ca" openssl req -x509 -newkey rsa:4096 -nodes -sha256 \ -days ${toString cfg.caValidityDays} \ -keyout "$cak" -out "$ca" \ -subj "/CN=hive-ca ${domain}" \ -addext "basicConstraints=critical,CA:TRUE,pathlen:0" \ -addext "keyUsage=critical,keyCertSign,cRLSign" ''; # Sign one leaf. Parameterised rather than hardcoded to `gateway.*` # because there are now two: the hive's own leaf, issued by the hive # CA, and the swarm-services leaf, issued by the services sub-CA that # ./swarm-ca.nix maintains. Same ceremony, different issuer and names # — and one script means the two cannot drift in how they are built. # # $1 stateDir $2 basename $3 CN $4 SAN list $5 issuer cert $6 issuer key signLeafScript = pkgs.writeShellScript "hive-tls-sign-leaf" '' set -euo pipefail d="$1" base="$2" cn="$3" sans="$4" ca="$5" cak="$6" leaf="$d/$base.pem" leafk="$d/$base-key.pem" csr="$(mktemp "$d/$base.csr.XXXXXX")" ext="$(mktemp "$d/$base.ext.XXXXXX")" only="$(mktemp "$d/$base.leaf.XXXXXX")" trap 'rm -f "$csr" "$ext" "$only"' EXIT openssl req -newkey rsa:4096 -nodes -sha256 \ -keyout "$leafk" -out "$csr" \ -subj "/CN=$cn" # printf (not a heredoc) so the ext-file lines carry no leading # whitespace once nix has stripped the indented-string indent. { printf 'subjectAltName=%s\n' "$sans" printf 'basicConstraints=critical,CA:FALSE\n' printf 'keyUsage=critical,digitalSignature,keyEncipherment\n' printf 'extendedKeyUsage=serverAuth\n' } > "$ext" openssl x509 -req -in "$csr" -CA "$ca" -CAkey "$cak" \ -CAcreateserial -days ${toString cfg.leafValidityDays} -sha256 \ -extfile "$ext" -out "$only" # nginx serves this file verbatim, so it must carry the leaf AND its # issuer: the hive CA is an intermediate under the swarm root, and a # client that anchors on the root cannot build the middle of the # chain by itself. Agents anchor on the hive CA directly and # validated either way — the appended cert is what makes a swarm # peer, or anything else holding only the root, work. # Leaf first: both `ssl_certificate` and the `openssl x509 -in` # expiry checks read the first cert in the file. cat "$only" "$ca" > "$leaf" chmod 0600 "$leafk" chmod 0644 "$leaf" ''; # The hive's own leaf: signed by the hive CA, covering the hive domain # and its sub-domains. # # `forge.` and `matrix.` were listed here explicitly and both were # redundant with the wildcard beside them — a wildcard covers exactly # one label, and those are one label. Naming them read as policy # ("these two services live under this leaf"), which is the reason they # outlived it: a configured service name is no longer necessarily under # this domain, and when it isn't, this is the one list it cannot join. # The hive CA is `nameConstraints=permitted;DNS:`, and a # violation invalidates the CERTIFICATE rather than the offending name # — measured with openssl, not reasoned about — so one sibling name # added here takes the leaf down for the dashboard and every other # vhost sharing it. Configured service names get the swarm-services # leaf below, whose sub-CA is constrained to exactly those names. signHiveLeaf = '' ${signLeafScript} "$d" gateway ${lib.escapeShellArg domain} \ ${lib.escapeShellArg "DNS:${domain},DNS:*.${domain}"} \ "$d/ca.pem" "$d/ca-key.pem" ''; # The swarm-services leaf: signed by the services sub-CA, covering the # swarm's service names. Those are *siblings* of the hive domain, not # children, so the hive CA is name-constrained out of them and cannot # sign this however its SAN list is written. # # Skipped when the sub-CA isn't on disk: it only exists where the # swarm CA is autoconfigured, and a hive that gets its certs from its # operator has nothing for this to do. signServicesLeaf = lib.optionalString (swarmServiceDomains != [ ]) '' servicesCa=${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca.pem"} servicesCaKey=${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca-key.pem"} if [ -s "$servicesCa" ] && [ -s "$servicesCaKey" ]; then ${signLeafScript} "$d" swarm-services \ ${lib.escapeShellArg (builtins.head swarmServiceDomains)} \ ${lib.escapeShellArg (lib.concatMapStringsSep "," (n: "DNS:${n}") swarmServiceDomains)} \ "$servicesCa" "$servicesCaKey" else echo "no swarm-services sub-CA at $servicesCa — skipping the services leaf" fi ''; # Shared by BOTH units that decide whether to re-sign. It lives here # rather than in one of them because the two guards have to agree: they # answer the same question at different times (`hive-tls-ca` at service # activation, i.e. on the deploy; `hive-tls-resign` from a weekly timer # for a host that stays up long enough to drift). A rule implemented in # one and not the other is worse than one implemented in neither — it # looks fixed and only fires on whichever path you did not take, which # is exactly how a corrected `serviceDomains` still served a stale leaf. # # Expects `$d` (state dir) to be set; defines `$leaf`-adjacent names and # `covers`. leafCoverage = '' svcleaf="$d/swarm-services.pem" hiveNames=${lib.escapeShellArg "${domain} *.${domain}"} svcNames=${lib.escapeShellArg (lib.concatStringsSep " " swarmServiceDomains)} # The services leaf is only expected where the sub-CA exists; # elsewhere its absence is the correct state, not a stale leaf. want_svc=${if swarmServiceDomains == [ ] then "0" else "1"} if [ ! -s ${lib.escapeShellArg "${swarmCaCfg.stateDir}/services-ca.pem"} ]; then want_svc=0 fi # Expiry is not the only way a leaf goes wrong. One signed when the # configured name set was smaller stays valid for its whole lifetime # while omitting every name added since — so a config change can # evaluate, build and deploy cleanly while the gateway keeps serving a # certificate that does not cover the new service. The services sub-CA # already reconciles this way (its `.names` comparison in # ../swarm-ca.nix); the leaves did not. # # Reads the names out of the CERTIFICATE, not a sidecar file: the pem # is what nginx serves, and a bookkeeping file drifts from it the # moment anyone replaces a leaf by hand. covers() { # $1 = leaf, $2 = space-separated names it must carry [ -s "$1" ] || return 1 _have="$(openssl x509 -in "$1" -noout -ext subjectAltName 2>/dev/null \ | tr ',' '\n' | sed -n 's/^[[:space:]]*DNS:\(.*\)$/\1/p' || true)" [ -n "$_have" ] || return 1 _missing=0 # ⚠️ The hive leaf carries `*.`; without this the shell # expands it against the cwd and the comparison silently tests a # filename instead of a name. set -f for _n in $2; do printf '%s\n' "$_have" | grep -qxF "$_n" || _missing=1 done set +f [ "$_missing" = 0 ] } # True when every leaf this host is supposed to hold is present and # carries its configured names. Expiry is the callers' own business — # they use different windows. leavesCoverNames() { covers "$leaf" "$hiveNames" || return 1 [ "$want_svc" = 0 ] && return 0 covers "$svcleaf" "$svcNames" } ''; in { # Host-side TLS trust root for the self-signed gateway mode. # # A bare self-signed leaf would be its own trust anchor, so every # regeneration would be a new anchor and every consumer (agents, # federation peers) would have to re-trust on each rotation — and a # runtime-generated, in-container cert can't be wired into an agent's # build-time trust store at all. # # So the issuer is a long-lived **hive CA** held on the host. The # gateway serves a **leaf** signed by that CA (via the `tls.certDir` # bind-mount path); agents and federation peers trust the *CA* once, # and leaf rotation never re-breaks them. See `docs/gateway.md` # ("Self-signed TLS"). # # That CA is self-signed by default. Under # `swarm.ca.autoConfigure` — the all-on-one-host case — it is instead # an intermediate issued under the swarm root (./swarm-ca.nix owns the # why) and name-constrained to this hive's domain, so it stays the # anchor agents pin while a peer holding only the root can validate # everything this hive serves. Either way an existing CA is left # alone; see the issuance comment below. options.services.hyperhive.tls = { stateDir = lib.mkOption { type = lib.types.str; default = "/var/lib/hive-tls"; description = '' Host directory holding the hive CA + gateway leaf cert for the self-signed gateway mode. `ca.pem` (the anchor agents and federation peers trust), `ca-key.pem` (0600, never leaves the host), `gateway.pem` / `gateway-key.pem` (the leaf the gateway container bind-mounts and nginx serves). Persistent so the CA survives reboots — re-deriving it would re-break every consumer. ''; }; caValidityDays = lib.mkOption { type = lib.types.int; default = 7300; description = '' Validity window of the hive CA in days (default ~20y). Kept long and well beyond `leafValidityDays` so the CA outlives many leaf rotations — the whole point of the CA is to be a stable anchor that consumers trust once. The CA is regenerated only if missing or already expired. ''; }; leafValidityDays = lib.mkOption { type = lib.types.int; default = 30; description = '' Validity window of the gateway leaf cert in days (default 30). Short-lived by design — ahead of the CA/Browser-Forum's move toward ~47-day max lifetimes — which bounds the blast radius of a leaf-key compromise. The leaf is re-signed by the (stable) CA when it is missing or near expiry; because it shares the CA anchor, a rotation does not disturb consumer trust. Agents and federation peers validate against the CA, not browser CA/B-forum limits. The weekly `hive-tls-resign` timer re-signs the leaf once it is within half its validity of expiry and propagates the new leaf into the running gateway, so a long-uptime host renews automatically without a reboot. ''; }; }; config = lib.mkIf active { # A swarm service name that is a sibling of the hive domain rather # than a child needs the swarm-services leaf, and this host only # signs that one when it holds the swarm root key. Without it nginx # falls back to the hive leaf on those names and every client sees a # name mismatch — on a config that evaluates and deploys cleanly. # # ⚠️ A WARNING, NOT AN ASSERTION, and the distinction is the point: # this module can see what *it* can issue; it cannot see an # operator-installed services sub-CA, an external ACME setup, or a # cert delivered by any other means. A rebuild must not be blocked # by a conclusion this host is not in a position to reach. Report # the observation and let the operator judge it. # # (`certDir` / ACME modes don't reach here at all: `active` is # self-signed-only, and there the operator's cert decides.) warnings = lib.optional (uncoveredServiceDomains != [ ] && !swarmCaCfg.autoConfigure) '' services.hyperhive: these swarm service names are outside this hive's domain (${domain}), so the hive CA's leaf does not cover them: ${lib.concatMapStringsSep "\n" (n: " ${n}") uncoveredServiceDomains} services.hyperhive.swarm.ca.autoConfigure is false, so this host does not sign the swarm-services leaf either, and the gateway will serve the hive leaf on those names — a certificate-name mismatch for browsers and for agents' git-over-https. If you have already arranged certificates for them — an operator-installed services sub-CA in ${swarmCaCfg.stateDir}, or services.hyperhive.gateway.tls.{certDir,acme} — this is expected and you can ignore it. Otherwise pin the names back under ${domain} (services.hyperhive.swarm.{forge.domain, matrix.gatewayHost, authelia.domain}) or install the sub-CA. ''; # Generate (and rotate) the hive CA + gateway leaf before anything # serves it. Idempotent: the CA is created once and reused; the leaf # is re-signed on expiry under the same CA so the anchor is stable. systemd.services.hive-tls-ca = { description = "Generate hive CA + gateway leaf TLS cert (self-signed mode)"; wantedBy = [ "multi-user.target" ]; # The consumer is `hive-gateway-self-signed-cert`, which copies the # leaf into the gateway's state dir at the mode nginx can read, and # which nginx in turn `Requires=`. So this must run first or that # copy fails under `set -eu` and takes nginx down with it — order # against the unit that reads the file. before = [ "hive-gateway-self-signed-cert.service" ]; requiredBy = [ "hive-gateway-self-signed-cert.service" ]; # The issuance below needs the swarm root key on disk, and (for the # services leaf) the services sub-CA it signs under. When this host # generates them (single-host swarm) both units must have run first; # when the operator provides the material there is no unit to wait # for, so the dependency is conditional rather than a unit that # exists and does nothing. # # Without waiting for swarm-services-ca specifically, this unit races # it: if hive-tls-ca finishes first, it finds no services-ca.pem yet, # silently skips signing the services leaf (the same as "operator # hasn't set one up"), and the gateway comes up with a vhost pointed # at a cert that was never written. after = lib.optionals swarmCaCfg.autoConfigure [ "swarm-ca.service" "swarm-services-ca.service" ]; requires = lib.optionals swarmCaCfg.autoConfigure [ "swarm-ca.service" "swarm-services-ca.service" ]; path = [ pkgs.openssl ]; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; UMask = "0077"; # Pin the journal identity (else it's the `script` store-path wrapper). SyslogIdentifier = "hive-tls-ca"; }; script = '' set -euo pipefail d=${lib.escapeShellArg cfg.stateDir} install -d -m 0755 "$d" ca="$d/ca.pem" cak="$d/ca-key.pem" leaf="$d/gateway.pem" leafk="$d/gateway-key.pem" root=${lib.escapeShellArg "${swarmCaCfg.stateDir}/root.pem"} rootk=${lib.escapeShellArg "${swarmCaCfg.stateDir}/root-key.pem"} prev="$d/ca-previous.pem" marker="$d/.swarm-ca-adopted" # --- Adoption: a hive whose CA predates the swarm root. # # Runs ONLY where this host also owns the root (`autoConfigure`), # because adoption invalidates an anchor that consumers already # trust and they refresh on their own schedule. On one box that # schedule is knowable; across hosts it is not, so there the # operator does it and this unit only says so, loudly. # # Guarded by a marker rather than by the state of the world: the # marker's ABSENCE is the trigger, so this fires once per hive # instead of re-deciding every activation. That is the difference # between a migration and a kicking machine. if [ -s "$root" ] && [ -s "$ca" ] && [ ! -e "$marker" ] \ && ! openssl verify -CAfile "$root" "$ca" >/dev/null 2>&1; then ${ if swarmCaCfg.autoConfigure then '' echo "adopting the swarm CA: $ca does not chain to $root" >&2 # Keep the old CA as an anchor across the overlap. Consumers # read the bundle, so adoption is additive before it is # subtractive — agents pick up new trust only when their # container restarts, which is a window even on one host. cp "$ca" "$prev" chmod 0644 "$prev" rm -f "$ca" "$cak" touch "$marker"'' else '' echo "hive-tls: this hive's CA does not chain to the swarm root." >&2 echo " hive CA: $ca" >&2 echo " swarm root: $root" >&2 echo "Adopting the hierarchy is not automatic here: it invalidates an" >&2 echo "anchor that peers and agents on OTHER hosts still trust, and they" >&2 echo "refresh on their own schedule — only you know when that is safe." >&2 echo "To adopt: rm $ca $cak && systemctl restart hive-tls-ca.service" >&2 echo "To keep the current CA deliberately: touch $marker" >&2 exit 1'' } fi # --- CA: generated once and reused across leaf rotations, in # whichever of the two shapes `caGenScript` selected. # Regenerated only if missing or already expired (checkend 0). # A new CA means every consumer must re-trust, so the leaf is # dropped to force a re-sign under the fresh CA. # # An existing CA is never re-rooted here. A hive predating the # swarm root carries a self-signed `ca.pem`, and swapping it for # a swarm-issued one would break every consumer already trusting # it — including agents, whose trust is refreshed only when their # container restarts. Adopting the hierarchy on such a hive is # therefore an operator step (drop the CA, let this unit # re-issue), which is also what keeps this change non-disruptive # to hives that never adopt it. if [ ! -s "$ca" ] || [ ! -s "$cak" ] \ || ! openssl x509 -in "$ca" -noout -checkend 0 >/dev/null 2>&1; then ${caGenScript} chmod 0600 "$cak" chmod 0644 "$ca" rm -f "$leaf" "$leafk" fi # --- Leaf: (re)sign when missing, within 30 days of expiry, or no # longer covering the configured names, always under the current # (stable) CA. ${leafCoverage} # ⚠️ This is the guard that runs ON THE DEPLOY — `hive-tls-resign` # only fires from a weekly timer, so a rule enforced only there is # up to a week late and does nothing for the rebuild that changed # the names in the first place. # # The name check has to include the SERVICES leaf even though the # condition is written around the hive one, because both are signed # in this block: a fresh `gateway.pem` otherwise suppresses the # re-sign of a `swarm-services.pem` that is missing or stale, which # is what left a corrected `serviceDomains` still mis-served. resign=0 { [ -s "$leaf" ] && [ -s "$leafk" ]; } || resign=1 openssl x509 -in "$leaf" -noout -checkend 2592000 >/dev/null 2>&1 || resign=1 leavesCoverNames || resign=1 if [ "$want_svc" = 1 ] && [ ! -s "$svcleaf" ]; then resign=1 fi if [ "$resign" = 1 ]; then echo "signing gateway leaf at $leaf (missing, near expiry, or missing a configured name)" ${signHiveLeaf} ${signServicesLeaf} fi # --- Trust bundle: what a consumer must TRUST, as opposed to # `ca.pem`, which is what this host SIGNS with. Why they stopped # being the same file, and why `ca-previous.pem` stays in the set # after an adoption, are in docs/swarm/ca.md. Three constraints # the code cannot state: # # Written IN PLACE, never renamed into position — containers # bind-mount this file and a bind mount follows the inode, so a # rename leaves every consumer holding the old one. # # Dropping `ca-previous.pem` is deliberately NOT done here: # "long enough" is a deployment fact, not a unit's call. # # Built as an array with `if` guards, not # `[ -s f ] && anchors+=(f)`: under `set -e` an && list whose test # fails IS a failing command and kills the unit — on exactly the # hive where the optional file is legitimately absent. The # reflexive `|| true` is worse; it also swallows a real failure to # read the hive CA. bundle="$d/trust-bundle.pem" anchors=("$ca") if [ -s "$prev" ]; then anchors+=("$prev"); fi if [ -s "$root" ]; then anchors+=("$root"); fi cat "''${anchors[@]}" > "$bundle" chmod 0644 "$bundle" ''; }; # Weekly re-sign of the gateway leaf so short-lived leaves renew # without depending on a reboot. # # `hive-tls-ca` only re-signs at service activation (boot/rebuild); a # long-uptime host would otherwise let a 30-day leaf lapse silently. # This service re-signs the leaf directly (not by bouncing hive-tls-ca) # and propagates the new leaf to nginx when the file actually changed. # # Propagation mechanism: nginx serves a *copy* of the leaf, written by # `hive-gateway-self-signed-cert` at the mode nginx can read. Re-signing # the source therefore changes nothing on its own — the copy has to be # remade and nginx reloaded, which is what the two calls below do. # # ⚠️ Neither call is `|| true`: a swallowed propagation failure means # the leaf rotates on disk while nginx keeps serving the old copy # until it expires, with this unit reporting success the whole time. # A failure here must fail the timer. systemd.services.hive-tls-resign = { description = "Re-sign the gateway TLS leaf and reload nginx"; # hive-tls-ca must have run first so the CA key exists before we try # to re-sign under it. On first boot `Persistent=true` on the weekly # timer fires immediately; without this ordering the resign could race # the CA initialisation and fail with "no such file" on the CA key. after = [ "hive-tls-ca.service" ]; path = [ pkgs.openssl pkgs.coreutils pkgs.systemd ]; serviceConfig = { Type = "oneshot"; UMask = "0077"; SyslogIdentifier = "hive-tls-resign"; }; script = '' set -euo pipefail d=${lib.escapeShellArg cfg.stateDir} leaf="$d/gateway.pem" # ⚠️ EVERY leaf this host issues must be covered here. A leaf that # first-boot issuance creates and this unit does not know about # looks perfect for its entire validity and then expires with no # warning — the failure is invisible until it is total. `svcleaf` # and the name checks come from the shared snippet, so this unit # and `hive-tls-ca` cannot disagree about what a good leaf is. ${leafCoverage} # Re-sign only when a leaf is within half its validity of expiry. # The weekly cadence catches this window well before one lapses. halflife=$(( ${toString cfg.leafValidityDays} * 86400 / 2 )) fresh() { # a leaf is fresh if it exists and is not near expiry [ -s "$1" ] && openssl x509 -in "$1" -noout -checkend "$halflife" >/dev/null 2>&1 } if fresh "$leaf" && { [ "$want_svc" = 0 ] || fresh "$svcleaf"; } \ && leavesCoverNames; then echo "leaves valid, and covering the configured names — no resign needed" exit 0 fi echo "a leaf is missing, near expiry, or missing a configured name — re-signing" before="$(sha256sum "$leaf" "$svcleaf" 2>/dev/null || true)" ${signHiveLeaf} ${signServicesLeaf} after="$(sha256sum "$leaf" "$svcleaf" 2>/dev/null || true)" if [ "$before" != "$after" ]; then echo "gateway leaf rotated — re-importing and reloading nginx" systemctl restart hive-gateway-self-signed-cert.service systemctl reload nginx.service else echo "gateway leaf unchanged (already up to date)" fi ''; }; systemd.timers.hive-tls-resign = { description = "Weekly gateway-leaf re-sign and propagation"; wantedBy = [ "timers.target" ]; timerConfig = { # Run weekly; Persistent=true fires a missed run on next boot if # the timer was not active (e.g. the host was off on the scheduled # day), preventing a dormant timer from letting the leaf lapse. OnCalendar = "weekly"; Persistent = true; }; }; # Signal the hive-c0re lifecycle that a hive CA exists: it bind-mounts # this file (read-only, public certs ONLY — never a key) into each # agent container so agents + their tools can trust the gateway's # self-signed leaf, and the meta flake wires the per-agent trust # bundle. It is the ANCHOR bundle rather than `ca.pem` for the reason # spelled out where the bundle is written above; no key path is ever # exposed (an agent that could read one could mint trusted certs). systemd.services.hive-c0re.environment.HIVE_TLS_CA_PATH = "${cfg.stateDir}/trust-bundle.pem"; # The same anchor, named for the swarm-queue clients that need it when # they mint a token from authelia over TLS. Declared HERE, beside the # bundle, rather than in each consumer's module: the path is this # module's fact, and two consumers re-deriving `${stateDir}/…` would be # two places to fix the day it moves. # # ⚠️ This is what was missing. `swarm-queue-client` built a bare # `reqwest::Client`, so it trusted only the platform roots and died at # `invalid peer certificate: UnknownIssuer` against a swarm whose # authelia is signed by the swarm CA — while this very bundle sat on # disk, already assembled, already handed to hive-c0re under a different # variable name. The anchor was never missing; nothing pointed the queue # client at it. # # Set unconditionally within this module's `active` guard, exactly like # the line above: where there is no hive CA this module contributes # nothing at all, and the clients then fall back to the platform roots — # which is correct for a swarm fronted by a public certificate. systemd.services.hive-c0re.environment.HIVE_C0RE_OIDC_CA_FILE = "${cfg.stateDir}/trust-bundle.pem"; # ⚠️ Gated, where the hive-c0re line above is not, and the asymmetry is # the point: hive-c0re runs on every hive, the controller runs on one. # Defining an environment key on a unit that does not exist CREATES a # unit fragment for it — inert (no `ExecStart`, empty `wantedBy`, never # activated) but present on every non-controller hive with a CA. Caught # in review on this PR; it evaluates and builds clean either way, which # is exactly why it needed a reviewer rather than a check. systemd.services.swarm-controller.environment.SWARM_CONTROLLER_OIDC_CA_FILE = lib.mkIf hyperhiveCfg.deploy.swarm-controller.enable "${cfg.stateDir}/trust-bundle.pem"; }; }