# Glue: the matrix registration token comes from the secret store. # # The store's first reader, and deliberately a small one. It fetches an opaque # 32-byte value and writes it where ./hive-matrix.nix already looks — the # homeserver never learns the store exists, and its config is unchanged. # # ⚠️ Why this credential first. It has no second file and no format: authelia's # OIDC secret needs a `.secret` *and* a matching `.digest`, so shipping that # one first would debug "can a reader authenticate and get bytes back" and # "did we write authelia's file format right" at the same time, with an SSO # outage as the failure mode. Here the failure is narrow — new agent accounts # cannot be provisioned, existing ones are untouched, nothing crash-loops. # # ⚠️ The fallback is today's behaviour, not a new one. `hive-matrix.nix`'s # activation script still mints a token when the file is absent; this unit # overwrites it with the swarm's copy when the store has one. A store that is # empty or unreachable leaves a working hive with a local token. # # 📌 This runs wherever a client identity is configured, NOT only where the # store is. `deploy.bao.enable` would have been the co-location assumption # itself; the reader needs a certificate, not a neighbour. On the store's own # host ./glue-bao-tls.nix supplies one as a `mkDefault` and nothing changes; # elsewhere an operator places the leaf and names it, and the same unit works. { pkgs, lib, config, ... }: let hyperhiveCfg = config.services.hyperhive; deployCfg = hyperhiveCfg.deploy; baoCfg = hyperhiveCfg.swarm.bao; baoDeploy = deployCfg.bao; # What decides whether this unit exists at all. A reader is defined by holding # a certificate the store accepts, and that is true on the store's own host # and on a hive three networks away for exactly the same reason. haveClientIdentity = baoDeploy.clientCertFile != null && baoDeploy.clientKeyFile != null; # Where the token lives in the store. A path, not a convention to guess at: # whoever writes it and whoever reads it must agree, and the agreement # belongs in one visible place. tokenPath = "secret/swarm/matrix/registration-token"; # A literal, not an option — ./hive-matrix.nix names its container # `containers.hive-matrix` directly and declares no `machine` to derive it # from, which the trust-bundle call in that file already says out loud. # ⚠️ A `swarm.matrix.machine` read parses fine and fails at module-system # resolution, so this is the kind of mistake only reading the target module # catches. matrixMachine = "hive-matrix"; in { config = lib.mkIf (hyperhiveCfg.enable && haveClientIdentity && deployCfg.matrix.enable) { # Same rule as the unit's own gate: this reader exists on a host that has a # client identity and a homeserver, which is not every host that runs the # store, so the store's module cannot name it. services.hyperhive.swarm.otel.journaldUnits = [ "swarm-bao-matrix-token" ]; systemd.services.swarm-bao-matrix-token = { description = "fetch the matrix registration token from the swarm secret store"; # Every one of these names a unit that exists only where the store runs. # `Requires=` on an absent unit fails the job outright, so the ordering is # conditional even though the read is not: off-host there is nothing local # to wait for, and the timeout below is what bounds the attempt instead. after = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" "container@${baoCfg.machine}.service" ]; wants = lib.optionals baoDeploy.enable [ "container@${baoCfg.machine}.service" ]; requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ]; before = [ "container@${matrixMachine}.service" ]; wantedBy = [ "container@${matrixMachine}.service" ]; path = [ deployCfg.bao.package pkgs.coreutils ]; # Sized for the race this loses, not for an unseal. `swarm-bao` comes up # seconds before this unit asks, and the cert-auth role it logs in # against is written seconds after — so a few short attempts cover it. # ⚠️ `swarm-bao-controller-policy`'s 2880 × 30s is NOT the model to copy. # That unit blocks nothing; this one is `Before=` the homeserver's # container, and whether that ordering waits across an auto-restart is # unverified — so an hours-long window would be a bet on an unknown, # where a minute is not. A store still sealed after it keeps the degrade # below, as today. # # `StartLimit*` are `[Unit]` settings, so they go here and not in # `serviceConfig` — systemd ignores them under `[Service]`. The window # has to exceed `RestartSec × burst`. startLimitBurst = 4; startLimitIntervalSec = 300; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; # What actually bounds the read below. Stated here rather than # left to systemd's default, so the number a boot waits on is in # the file that waits. TimeoutStartSec = 30; Restart = "on-failure"; RestartSec = 15; }; environment = { BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}"; BAO_CLIENT_CERT = baoDeploy.clientCertFile; BAO_CLIENT_KEY = baoDeploy.clientKeyFile; } # Absent means the system trust store, which is what a deployment with a # real CA wants and what a self-signed one must not be left with. // lib.optionalAttrs (baoDeploy.serverCaFile != null) { BAO_CACERT = baoDeploy.serverCaFile; }; script = '' set -euo pipefail # A sealed or uninitialised store answers on the port and never # answers the read, so "the store is up" is not the same as "the # store can answer". `TimeoutStartSec` above is the bound; the # homeserver only `Wants=` this unit, so hitting it degrades to # keeping the local token rather than holding up the container. # `bao`'s own message is the only thing separating a missing value # from a refused identity from an unreachable host. This unit's # degraded mode is correct for all three, so it reports which one # rather than asserting all three in a sentence of ours — a reader # that cannot say why it read nothing is indistinguishable from a # broken one. err="$(mktemp)" trap 'rm -f "$err"' EXIT # Cert auth is a login, not a transport setting. The `BAO_CLIENT_*` # variables above only decide which certificate the TLS handshake # presents; without a token `bao` asks its token helper instead, and # that is a `sh` this unit's `path` does not carry. `-token-only` # answers on stdout and skips the helper on both sides. # # Fails LOUDLY, unlike the read below: the three states a login failure # covers — store not up, sealed, role not written yet — are all things # a retry fixes, and `Restart=on-failure` above is what retries. Exiting # 0 here spends the whole boot on a condition that was seconds old. if ! BAO_TOKEN="$(bao login -method=cert -token-only 2>"$err")"; then echo "could not log in to swarm-bao with this host's certificate; keeping the token hive-matrix already has." >&2 if [ -s "$err" ]; then cat "$err" >&2 else echo "bao failed without writing a diagnostic." >&2 fi exit 1 fi export BAO_TOKEN if ! token="$(bao kv get -field=value ${lib.escapeShellArg tokenPath} 2>"$err")"; then echo "swarm-bao did not return ${tokenPath}; keeping the token hive-matrix already has." >&2 if [ -s "$err" ]; then cat "$err" >&2 else echo "bao failed without writing a diagnostic." >&2 fi exit 0 fi if [ -z "$token" ]; then echo "swarm-bao returned an empty ${tokenPath}; keeping the local token." >&2 exit 0 fi umask 077 printf '%s\n' "$token" > ${lib.escapeShellArg (toString deployCfg.matrix.registrationTokenFile)} chmod 0600 ${lib.escapeShellArg (toString deployCfg.matrix.registrationTokenFile)} ''; }; }; }