{ pkgs, lib, config, ... }: let cfg = config.services.hyperhive.swarm.nats; autheliaCfg = config.services.hyperhive.swarm.authelia; autheliaUrl = autheliaCfg.url; networkCfg = config.services.hyperhive.network; # Read even when the controller runs on a different host: what is needed # is the client id that module *declares*, which is the same string # everywhere, not whether the daemon happens to be enabled here. controllerCfg = config.services.hyperhive.swarm.controller; # The account the callout responder authenticates as, and the account # authorized clients are placed in. Two accounts rather than one: an # account is NATS' isolation boundary, so a responder that shares an # account with its clients can be published to by the things it # authorizes. calloutAccount = "AUTH"; clientAccount = "APP"; machine = "swarm-nats"; tlsCfg = config.services.hyperhive.tls; gatewayCfg = config.services.hyperhive.gateway; caTrust = import ./lib/hive-ca-trust.nix { inherit lib tlsCfg gatewayCfg; }; # The responder introspects authelia over https BY NAME. Its HTTP client is # reqwest/rustls, and `rustls-platform-verifier` resolves roots through # `rustls-native-certs`, which reads `SSL_CERT_FILE` — so the same assembled # bundle the Go containers use applies here. Without it the handshake fails # `UnknownIssuer`, introspection fails, and the responder denies *every* # client: one missing trust anchor surfacing as `authorization violation` at # every would-be queue user. caBundleModule = caTrust.trustBundle { inherit pkgs; name = machine; consumers = [ "swarm-nats-auth" ]; }; # Where the responder's credentials live *inside* the container, and the # host path that resolves to. Two names for one location, because the # host is the only place both filesystems are addressable. secretDirInContainer = "/var/lib/swarm-nats-auth"; inContainer = name: "${secretDirInContainer}/${name}"; secretDir = "/var/lib/nixos-containers/${machine}${secretDirInContainer}"; hostPath = name: "${secretDir}/${name}"; # The responder needs all three credentials. Gating on them rather than # on `cfg.enable` keeps a half-configured hive at "queue up, denying # everyone" instead of "unit crash-looping on a missing file". # # In auto mode the seeds are minted on this host before the container # starts, so they are configured by construction. responderConfigured = cfg.autoGenerateCallout || (cfg.calloutUserSeedFile != "" && cfg.calloutIssuerSeedFile != ""); clientSecretSource = "${autheliaCfg.hostClientSecretDir}/${cfg.clientId}.secret"; introspectionUrl = "${toString autheliaUrl}/api/oidc/introspection"; # Where the responder's seeds actually come from. One name for two # origins, so everything downstream stops caring which mode it is in. userSeedFile = if cfg.autoGenerateCallout then autoUserSeed else cfg.calloutUserSeedFile; issuerSeedFile = if cfg.autoGenerateCallout then autoIssuerSeed else cfg.calloutIssuerSeedFile; # Seeds stay on the host at 0600 and never enter the container or the # store: only the responder needs them, and it reads them by # `LoadCredential` from here. autoSeedDir = "/var/lib/swarm-nats-callout"; autoUserSeed = "${autoSeedDir}/callout-user.seed"; autoIssuerSeed = "${autoSeedDir}/issuer.seed"; # The runtime config directory: wrapper, settings symlink and fragment. # World-readable is correct (everything in it is public) and it is not in # the 0700 responder secret dir, which the `nats` user cannot traverse. # # ⚠️ ONE DIRECTORY IS FORCED, NOT TIDINESS. NATS resolves an `include` with # `filepath.Join(configDir, path)`, which strips a leading slash, so an # absolute include silently becomes relative and is never found. The # includes must therefore be bare filenames, i.e. siblings. Invisible to # eval: the wrapper renders perfectly and the server refuses to start. runtimeDir = "/var/lib/nats-callout"; runtimeWrapper = "${runtimeDir}/nats.conf"; hostRuntimeDir = "/var/lib/nixos-containers/${machine}${runtimeDir}"; natsFormat = pkgs.formats.json { }; # Upstream's rendered settings, re-rendered with the same generator on the # same value — the artifact `services.nats` would have used, not a # transcription. ⚠️ This reference is also the only thing keeping it alive: # `ExecStart` names a runtime path, so nothing else in the closure names # the store file, and its string context is what stops it being # garbage-collected out from under a running server. renderedSettings = natsFormat.generate "nats.conf" ( config.containers.${machine}.config.services.nats.settings ); # Defined once, rendered twice: into `settings` with the operator's values, # and into the runtime fragment with placeholders the generator fills in. # # 🪤 Do not inline these and hand-write the fragment instead. Both files are # loaded and the fragment is the LATER definition, so it wins — a # hand-written copy would make a future edit to `settings` silently # ineffective on exactly the hives that use auto mode. calloutBlocks = { userKey, issuerKey, }: { # Two accounts, and the callout user lives in neither of the accounts # it authorizes into. accounts = { # ⚠️ The nkey is not decoration and its absence was a real hole: a # `users` entry carrying only a `user` name has no credential, and # `CONNECT {"user":"auth"}` is then accepted with no password at # all. Since the name is a literal in this public module, that made # the callout-exempt identity walk-in-able from every container on # the shared netns — the same class of hole this module exists to # close, moved rather than fixed. Caught in review on the first # version of this file. An nkey and NOTHING else, both halves # measured against a running server rather than reasoned about: # # { user = "auth"; } → `CONNECT {"user":"auth"}` # is accepted with no # credential at all # { user = "auth"; nkey = "U…"; } → refuses to START: # "Nkey users do not take # usernames or passwords" # { nkey = "U…"; } → what this is # # A malformed key is fail-closed too: the server exits with # "Not a valid public nkey for a user" rather than starting with a # hole. So the only way to get a live server here is a real key # whose seed nobody but the responder holds. # # 🔒 That property is what makes auto mode safe: it renders as "" # there, so any field the fragment fails to override keeps a value # the server refuses to start on. An incomplete merge cannot leave # a walk-in-able server. ${calloutAccount}.users = [ { nkey = userKey; } ]; # ⚠️ `services.nats.jetstream = true` gives the SERVER JetStream; # an account gets it only from its own grant. Measured against a # running 2.14.1 with this exact two-account shape, because the # failure is invisible to any config-rendering check: # # global jetstream only → `nats kv add` from this account fails # `code=503 err_code=10039 jetstream not enabled for account`, # while the server starts cleanly and logs "Starting JetStream" # + this line → the same command succeeds # # The grant is per-account by design, and that is worth keeping: # the callout account above deliberately does NOT get it. The # responder mints credentials; it has no business holding stream # state. # # ⚠️ Also why the fragment renders the COMPLETE accounts block: if a # later definition replaced rather than merged, a partial one would # drop this grant and every KV op would fail on a healthy server. ${clientAccount} = { jetstream = "enabled"; }; }; authorization = { timeout = "2s"; # 🔒 THIS BLOCK IS THE FAIL-CLOSED STATE, and it is the measured # one rather than the obvious one. # # Measured on the pinned nats-server 2.14.1: both # `authorization { }` and `authorization { users: [] }` accept an # anonymous client and answer PONG — they read like "authorize # nobody" and are wide open. An auth_callout block sets # `auth_required` and refuses every client whose credential no # responder has approved, so a config whose responder does not # exist yet denies everyone. # # `nats-server -t` calls all three valid; it parses, it does not # authenticate. Only running them tells the difference. # # ⇒ this is both the safe interim state and the final shape. # Nothing here has to be swapped out when the responder lands # beside it — it only starts being able to say yes. auth_callout = { issuer = issuerKey; auth_users = [ userKey ]; account = calloutAccount; }; }; }; # The fragment as nix renders it, with placeholders where the runtime # values go. Rendered by the same JSON generator upstream uses, so the # fragment is generated rather than transcribed — NATS' config parser # accepts JSON, and an `include` of it merges (measured). calloutTemplate = natsFormat.generate "swarm-nats-callout-template.conf" (calloutBlocks { userKey = "@USER_PUBKEY@"; issuerKey = "@ISSUER_PUBKEY@"; }); in { # The swarm's message queue: one NATS server, reached by every hive. # # ⚠️ There is deliberately NO gateway vhost here, and this is the first # swarm service where that is true — the next reader will go looking for # one. NATS speaks its own TCP protocol rather than HTTP, so nginx # cannot front it the way it fronts the forge, matrix and authelia. # Cross-hive reach is the wireguard mesh; `gateway.localNames` and the # per-service vhost pattern do not apply. options.services.hyperhive.swarm.nats = { enable = lib.mkOption { type = lib.types.bool; default = false; description = '' Run the swarm's message queue in a `swarm-nats` container on this host. A swarm has one queue, so this belongs on the same host as the rest of the shared services. Off by default, and off means *absent*: no container is created and nothing else in the evaluated config changes. ''; }; # ⚠️ Deliberately NO `package` option, unlike this module's siblings. # `services.nats` upstream does not expose one — it resolves # `pkgs.nats-server` itself — so an option here would either be # ignored or need an overlay to mean anything, and an option that # does not control what it names is worse than its absence. Pin the # build with `nixpkgs.overlays` if you need to. port = lib.mkOption { type = lib.types.port; default = 4222; description = '' TCP port the queue listens on. 4222 is upstream's default and sits outside hyperhive's claimed ranges (dashboard 7000, forge 3000, matrix 8008, every agent in 8100..8999 via FNV-1a hash). ''; }; monitorPort = lib.mkOption { type = lib.types.port; default = 8222; description = '' Port NATS serves its **monitoring** endpoint on, bound to loopback. Not a metrics endpoint: the server has no Prometheus format of its own. This serves `/varz`, `/connz`, `/routez` as JSON, and the exporter below is what translates it — which is why enabling the exporter without this produces a process that starts cleanly and scrapes nothing. ⚠️ Loopback, and the exporter is the only intended reader. The endpoint is unauthenticated and `/connz` names every connected client, so the address it binds is the whole access control. Do not widen it, and do not put it behind a gateway vhost expecting that to add one. ''; }; metricsPort = lib.mkOption { type = lib.types.port; default = 7777; description = '' Port the Prometheus exporter serves NATS's metrics on, bound to loopback for the swarm collector to scrape. ⚠️ This and {option}`monitorPort` are two more claims on a port space every swarm container shares — they run in this container but `privateNetwork = false`, so a collision with any other hyperhive service is a runtime coin toss over which process gets the port, with nothing in any log saying so. Both defaults are upstream's own (`nats-server` 8222, `prometheus-nats-exporter` 7777) and neither is claimed elsewhere in this repo, checked when they were added. ''; }; clientId = lib.mkOption { type = lib.types.str; default = "swarm-nats"; description = '' OAuth2 client id the queue's authentication path identifies itself with. Must match the `id` of the corresponding entry in `services.hyperhive.swarm.authelia.oidc.clients` — which this module contributes for you when both run on this host. ''; }; autoGenerateCallout = lib.mkOption { type = lib.types.bool; default = false; example = true; description = '' Generate the auth-callout nkeys on this host instead of taking them from `calloutUserPublicKey` / `calloutIssuerPublicKey`. A first-boot unit mints both keypairs if absent, keeps the seeds host-side at `0600`, and writes only the public halves into a fragment the server reads. Nothing secret is evaluated, so nothing secret reaches the nix store. Leave it off wherever the queue and its clients are not the same operator's problem: the seeds must reach whoever runs the responder, and minting them here only moves that distribution somewhere less visible. `enableAllLocalDefaults` turns it on. ''; }; calloutUserPublicKey = lib.mkOption { type = lib.types.str; default = ""; example = "UDXU4RCSJNZOIQHZNWXHXORDPRTGNJAHAHFRGZNEEJCPQTT2M7NLCBBQ"; description = '' Public half of the **user** nkey the auth-callout responder authenticates as. `auth_callout.auth_users` exempts this identity from needing callout approval — it is the one that answers auth requests, so it cannot wait for itself. **That exemption is exactly why it needs a credential of its own**: without one the escape hatch is an open door, and on a container sharing the host netns it is an open door reachable from every agent container. An nkey rather than a password for the same reason `calloutIssuerPublicKey` is: only the public half appears here, and nix renders it into the world-readable store harmlessly. The seed reaches the responder and nothing else, so until the responder exists **nobody can authenticate as this user at all** — which is what makes a hive with no responder genuinely closed rather than merely gated. Required when `enable` is set. ''; }; calloutIssuerPublicKey = lib.mkOption { type = lib.types.str; default = ""; example = "ACYR44YO3XZRZBJIYLI5SL6LOPIW37JTD52LNOBHUE34XMH7N5ABMFJH"; description = '' Public half of the account nkey whose signature the server accepts on a user JWT minted by the auth-callout responder. **A public key, and therefore a value rather than a path** — the deliberate exception to the rule that credentials are `*File` options. It is published to every client that connects and its whole job is to be widely known; the matching *seed* is the secret, is never named here, and reaches only the responder. Required when `enable` is set. Without it the server has no issuer to trust and no client can be authorized — which is the fail-closed state described below, but arrived at by accident rather than on purpose, so it fails at eval instead. ''; }; authPackage = lib.mkOption { type = lib.types.package; defaultText = lib.literalExpression "hyperhive.packages.\${system}.swarm-nats-auth"; description = '' The auth-callout responder package. ⚠️ Named `authPackage`, not `package`, on purpose: this module deliberately has **no** `package` option for the server itself (see the note above — upstream's `services.nats` resolves `pkgs.nats-server` on its own), so a bare `package` here would read as "the NATS package" and mean something else entirely. ''; }; calloutUserSeedFile = lib.mkOption { type = lib.types.str; default = ""; example = "/run/secrets/swarm-nats-callout-user.seed"; description = '' Absolute host path to the **seed** whose public half is `calloutUserPublicKey`. The responder authenticates to the queue with it. A `str` rather than a `path`, and the reason is not style: a `path`-typed literal is hash-copied into the world-readable nix store at eval time, which is the opposite of what a seed wants. Same discipline as `otel.headersCredential`. Until this is set the responder cannot start, and the queue stays in its fail-closed state — which is the correct behaviour, not a gap. ''; }; calloutIssuerSeedFile = lib.mkOption { type = lib.types.str; default = ""; example = "/run/secrets/swarm-nats-issuer.seed"; description = '' Absolute host path to the **account** seed whose public half is `calloutIssuerPublicKey`. The responder signs the user JWTs it issues with it, so possession of this file is the authority to admit anyone to the queue. A `str` for the same store-leak reason as `calloutUserSeedFile`. ''; }; }; config = lib.mkIf cfg.enable { assertions = [ { # Fail at EVAL, not at boot: a queue that comes up unable to # authenticate anyone presents as every client hanging, which is # several layers from "the operator never set the issuer". assertion = cfg.autoGenerateCallout || cfg.calloutIssuerPublicKey != ""; message = '' services.hyperhive.swarm.nats.enable requires nats.calloutIssuerPublicKey — the public half of the account nkey that signs user JWTs for this queue. It is public and belongs in config; the matching seed is a secret and is delivered to the callout responder instead. See docs/swarm/secrets.md for which is which. ''; } { # Without this the callout-exempt user has no credential, and # NATS accepts `CONNECT {"user":"auth"}` from anyone. An eval # failure is the only place to catch that: the rendered config is # valid, the server starts, and the hole is invisible until # somebody connects. assertion = cfg.autoGenerateCallout || cfg.calloutUserPublicKey != ""; message = '' services.hyperhive.swarm.nats.enable requires nats.calloutUserPublicKey — the public half of the user nkey the auth-callout responder authenticates as. It is exempt from callout approval by design, which is exactly why it needs its own credential: a `users` entry with a name and no key authenticates anyone who sends that name. ''; } { assertion = autheliaUrl != null; message = '' services.hyperhive.swarm.nats.enable requires services.hyperhive.swarm.authelia.url — the queue authenticates clients by validating tokens that authelia issued. It defaults to this host's own instance only when this host runs authelia. A hive that federates with a swarm sets it explicitly to wherever that provider lives. ''; } ]; # One declaration, two readers. The queue knows which client id it # authenticates under; making the operator restate it in authelia's # client list would be a second source of truth for a string whose # mismatch is an opaque 401 from the token endpoint. # # `client_credentials` rather than an authorization-code flow: a # queue's clients are daemons with nobody to redirect, so a # non-interactive grant is what makes a hive able to authenticate at # all. No redirect URI exists or is wanted. # # ⚠️ `kind` says that, and an empty `redirectUris` does NOT. This # declaration carried the comment above while rendering as an # interactive client, because authelia permits only the grants a # client names and an omitted `grant_types` means authorization-code # alone. Measured against 4.39.20: the token endpoint answered # `unauthorized_client: The OAuth 2.0 Client is not allowed to use # authorization grant 'client_credentials'`. Introspection — which # is all the responder needs today — worked throughout, which is why # nothing was visibly broken while the comment was untrue. services.hyperhive.swarm.authelia.oidc.clients = lib.mkIf autheliaCfg.enable [ { id = cfg.clientId; description = "HyperHive swarm queue"; kind = "machine"; } ]; # Declared here rather than in the collector's module: an entry then # exists only where the service that named it runs. Gated on a # collector, because a target nobody reads asserts a collection that is # not happening. # # ⚠️ KNOWN LIMITATION, and it is SILENT. That rule constrains the # target, not the scraper — nothing places the collector on this host. # `swarm-required-services.nix` derives `nats.enable` and `otel.enable` # from one `lib.mkDefault`, so they are co-located by *default*, and an # operator may split them. Split, NATS is never scraped: this host # declares an entry no local collector reads, the collector's host never # enabled this module. No error, no warning — a healthy exporter and an # empty dashboard. # # Not guardable: separate hosts are separate evaluations with no shared # context, so this one cannot see what that one runs. Saying so is the # only mechanism there is. # # ⚠️ `swarm.otel`, not `hyperhive.otel` — two collectors one word apart, # and only this one reads `scrapeTargets`. Written in full so the gate # and the option it gates are visibly the same path. services.hyperhive.swarm.otel.scrapeTargets = lib.mkIf config.services.hyperhive.swarm.otel.enable { nats = "127.0.0.1:${toString cfg.metricsPort}"; }; containers.swarm-nats = { autoStart = true; ephemeral = false; # Shared host netns, like every sibling container. # # ⚠️ Which is exactly why the server below must refuse everyone # until the callout responder exists: on this netns the queue is # reachable from every agent container on the hive, so an # unauthenticated interim state would be a hole rather than a # rough edge. privateNetwork = false; # Binds only the public trust bundle, read-only. Empty when the gateway # is not self-signed, so the whole trust path drops out cleanly. bindMounts = caTrust.bindMount; config = { ... }: { imports = [ (import ./swarm-container-resolver.nix { inherit (networkCfg) bridgeIp; # The responder introspects authelia BY NAME on every auth # request, and a responder that cannot resolve it denies # every client — so it must not start before the resolver # file exists. dnsConsumers = [ "swarm-nats-auth.service" ]; }) caBundleModule ]; system.stateVersion = "26.05"; # Shared host netns: this container's own firewall.service # would rewrite the HOST ruleset at every boot. The host # firewall owns all filtering. networking.firewall.enable = false; # resolvconf stays off because the resolver unit imported above # owns /etc/resolv.conf. Leaving it on would let host-tracking # regenerate the file empty, since the host's copy does not # cross the boundary after start. networking.resolvconf.enable = lib.mkForce false; services.nats = { enable = true; # Retention, so a reader can ask "what did this hive last # say?" without anyone keeping a second copy. The upstream # option also wires `settings.jetstream.store_dir = dataDir`; # the container is `ephemeral = false`, so that survives a # restart with no bind mount. # # ⚠️ Losing the store is not a correctness problem here: a # reader then sees nothing for every hive, which is the true # answer until each one publishes again. It degrades to # honesty rather than to a stale "healthy". jetstream = true; serverName = "swarm-nats"; port = cfg.port; # In auto mode the keys are empty until the generator runs, # and `nats-server -t` rejects that ("Expected callout user to # be a valid public account nkey, got \"\""), so leaving this # on fails the BUILD of every all-local hive. Upstream's own # description names the case: disable it when the config # includes other files. The check moves to server start. validateConfig = !cfg.autoGenerateCallout; settings = calloutBlocks { userKey = cfg.calloutUserPublicKey; issuerKey = cfg.calloutIssuerPublicKey; } // { # The monitoring endpoint, which is what the exporter below # reads. `//` adds a key `calloutBlocks` does not produce # (`accounts`, `authorization`) — checked, because a shallow # merge that collided here would drop the auth config while # rendering a config the server starts on. # # ⚠️ Loopback: unauthenticated, and `/connz` lists every # connected client. http = "127.0.0.1:${toString cfg.monitorPort}"; }; }; # The translation layer. NATS has no Prometheus format of its # own, so this reads the JSON monitoring endpoint above and # re-serves it in the format the collector scrapes. # # Upstream's exporter module rather than a hand-rolled unit, for # the same reason `services.nats`'s own `settings` is kept: the # reviewed reasoning about flags and hardening lives there. services.prometheus.exporters.nats = { enable = true; listenAddress = "127.0.0.1"; port = cfg.metricsPort; url = "http://127.0.0.1:${toString cfg.monitorPort}"; }; # A wrapper that includes upstream's rendered settings verbatim # plus the runtime fragment; rendering the config ourselves # instead would throw away upstream's `settings`, where the # reviewed reasoning lives. The generator writes it, because the # includes must be siblings of the fragment (see `runtimeDir`). # # `mkForce`: upstream defines ExecStart inside an `mkMerge`, so a # plain override conflicts rather than wins. systemd.services.nats.serviceConfig.ExecStart = lib.mkIf cfg.autoGenerateCallout ( lib.mkForce "${pkgs.nats-server}/bin/nats-server -c ${runtimeWrapper}" ); # The auth-callout responder: the half that lets the server # above say *yes*. Without it the `auth_callout` block is a # door nobody can open, which is the deliberate interim state. # # ⚠️ It is gated on the seeds being configured rather than on # `cfg.enable`, so a half-configured hive gets a running, # refusing queue instead of a unit that crash-loops on a # missing file. A queue that denies everyone is a legible # failure; a restart loop is not. systemd.services.swarm-nats-auth = lib.mkIf responderConfigured { description = "swarm queue auth-callout responder"; after = [ "nats.service" ]; requires = [ "nats.service" ]; wantedBy = [ "multi-user.target" ]; serviceConfig = { ExecStart = lib.concatStringsSep " " [ "${cfg.authPackage}/bin/swarm-nats-auth" "--nats-url nats://127.0.0.1:${toString cfg.port}" "--user-seed-file \${CREDENTIALS_DIRECTORY}/callout-user.seed" "--issuer-seed-file \${CREDENTIALS_DIRECTORY}/issuer.seed" "--client-secret-file \${CREDENTIALS_DIRECTORY}/oidc-client.secret" "--client-id ${lib.escapeShellArg cfg.clientId}" # The account admitted clients land in, by NAME: in # server-config mode the server resolves `aud` against its # own `accounts` block, so this and the block above have to # be the same string — which is why both come from one let. "--account ${lib.escapeShellArg clientAccount}" "--introspection-url ${lib.escapeShellArg introspectionUrl}" # Both of these name a principal some OTHER module mints, # so both are read out of that module rather than spelled # again here — same argument as `--account` above, one # level wider. The responder denies a client id it does # not recognise, and a NATS denial arrives as a timeout, # so a drift here is silent at the point of change and # misattributed at the point of failure. "--hive-client-prefix ${lib.escapeShellArg autheliaCfg.hiveClientPrefix}" "--reader-client ${lib.escapeShellArg controllerCfg.queueClientId}" ]; # Every credential arrives by `LoadCredential` and is named # on the command line only as a **path** — `argv` is # world-readable via /proc//cmdline, so a value there # would be readable by every process on the host netns. LoadCredential = [ "callout-user.seed:${inContainer "callout-user.seed"}" "issuer.seed:${inContainer "issuer.seed"}" "oidc-client.secret:${inContainer "oidc-client.secret"}" ]; DynamicUser = true; Restart = "on-failure"; RestartSec = "5s"; SyslogIdentifier = "swarm-nats-auth"; }; }; # The server binary, so an operator with a shell in here can # run `nats-server -t` against the generated config. The unit # resolves ExecStart through the store path and puts nothing # on PATH. environment.systemPackages = [ pkgs.nats-server ]; }; }; # Deliver the responder's three credentials into the container before # it starts. Same shape as `hive-matrix-oidc-secret`, and for the same # reason it is a copy rather than a `bindMounts` entry: nixos-container # refuses to start when a bind source is missing, so one absent seed # would take down the **whole container including the queue**, not # merely the responder. A far larger blast radius than the fault. # Order the container after the host CA generator, so the bind source # exists before nspawn sets the mount up. Without it a late CA fails the # container start outright rather than degrading. systemd.services."container@${machine}" = caTrust.containerOrdering; systemd.services.swarm-nats-auth-secrets = lib.mkIf responderConfigured { description = "deliver the swarm queue responder's credentials"; before = [ "container@swarm-nats.service" ]; wantedBy = [ "container@swarm-nats.service" ]; # In auto mode the seeds this copies do not exist until the generator # has run. `requires` as well as `after`: if minting fails there is # nothing to deliver, and a copy that silently succeeds with a stale # or absent seed is worse than not running. after = lib.optional cfg.autoGenerateCallout "swarm-nats-callout-keys.service" # The third credential does not come from the generator above — it is # minted by authelia's FIRST BOOT, inside its own container. Ordering # after that container is necessary and NOT sufficient: the container # being up says nothing about whether its in-container secrets unit # has finished. The wait in the script is what actually closes it; # this only stops us spinning for the full timeout on every boot. ++ lib.optional autheliaCfg.enable "container@${autheliaCfg.machine}.service"; requires = lib.optional cfg.autoGenerateCallout "swarm-nats-callout-keys.service"; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; # Must exceed the script's own wait below, and is set rather than # left to the default for exactly that reason: systemd's # `DefaultTimeoutStartSec` is 90s, so a 120s wait would be killed at # 90 and the operator would get a generic unit timeout instead of # the message that names the missing file and says what to do. # The bound that matters belongs in one place, and this makes the # two visibly related. TimeoutStartSec = "180s"; SyslogIdentifier = "swarm-nats-auth-secrets"; }; path = [ pkgs.coreutils ]; script = '' set -euo pipefail install -d -m 0700 ${lib.escapeShellArg secretDir} install -m 0400 ${lib.escapeShellArg userSeedFile} \ ${lib.escapeShellArg (hostPath "callout-user.seed")} install -m 0400 ${lib.escapeShellArg issuerSeedFile} \ ${lib.escapeShellArg (hostPath "issuer.seed")} # Wait for authelia's minted secret rather than failing the instant # it is absent. On a fresh boot this unit and authelia's first-boot # generator race, and losing that race used to cost the WHOLE QUEUE: # this exits 1, the responder never starts, and `auth_callout` with # no responder refuses every client — fail-closed by design, so the # symptom appears on every queue client and nowhere near the cause. # # Bounded, not indefinite. Where authelia runs on another host the # file is never going to appear, and blocking the queue container # forever would replace a clear failure with a hang. After the # timeout this fails exactly as it did before, having first given # the co-located case the seconds it actually needs. secret=${lib.escapeShellArg clientSecretSource} deadline=$(( SECONDS + 120 )) while [ ! -s "$secret" ]; do if [ "$SECONDS" -ge "$deadline" ]; then echo "the swarm queue responder's OIDC secret never appeared at $secret" >&2 echo "(authelia mints it on first boot; if authelia runs on another host," >&2 echo " copy the secret there and this unit will pick it up)" >&2 exit 1 fi sleep 2 done install -m 0400 "$secret" \ ${lib.escapeShellArg (hostPath "oidc-client.secret")} ''; }; # ⚠️ Minted on the HOST, not in the container, because the responder is # a separate unit that needs the user seed: generating it inside would # trap it there and require a secret-export path back out — the exact # mechanism this is meant to avoid inventing. Only public halves cross. systemd.services.swarm-nats-callout-keys = lib.mkIf cfg.autoGenerateCallout { description = "mint the swarm queue's auth-callout nkeys"; before = [ "container@swarm-nats.service" ]; wantedBy = [ "container@swarm-nats.service" ]; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; SyslogIdentifier = "swarm-nats-callout-keys"; }; path = [ pkgs.coreutils pkgs.nkeys ]; script = '' set -euo pipefail umask 077 install -d -m 0700 ${lib.escapeShellArg autoSeedDir} # Mint iff absent. Idempotence is the whole contract: this runs on # every boot, and regenerating would silently invalidate every # credential the responder has already issued against the old # issuer. if [ ! -s ${lib.escapeShellArg autoUserSeed} ]; then nk -gen user > ${lib.escapeShellArg autoUserSeed}.tmp mv ${lib.escapeShellArg autoUserSeed}.tmp ${lib.escapeShellArg autoUserSeed} fi if [ ! -s ${lib.escapeShellArg autoIssuerSeed} ]; then nk -gen account > ${lib.escapeShellArg autoIssuerSeed}.tmp mv ${lib.escapeShellArg autoIssuerSeed}.tmp ${lib.escapeShellArg autoIssuerSeed} fi chmod 0600 ${lib.escapeShellArg autoUserSeed} ${lib.escapeShellArg autoIssuerSeed} user_pub="$(nk -inkey ${lib.escapeShellArg autoUserSeed} -pubout)" issuer_pub="$(nk -inkey ${lib.escapeShellArg autoIssuerSeed} -pubout)" # Public halves only — world-readable on purpose, since the server # publishes them to every client that connects. install -d -m 0755 ${lib.escapeShellArg hostRuntimeDir} sed -e "s|@USER_PUBKEY@|$user_pub|g" \ -e "s|@ISSUER_PUBKEY@|$issuer_pub|g" \ ${calloutTemplate} > ${lib.escapeShellArg hostRuntimeDir}/callout.conf.tmp # Mode BEFORE the rename: a rename publishes whatever the file # already is, so setting it afterwards leaves a window where the # live path has the wrong mode. Same trap as the gateway's # atomic-publish path. chmod 0444 ${lib.escapeShellArg hostRuntimeDir}/callout.conf.tmp mv ${lib.escapeShellArg hostRuntimeDir}/callout.conf.tmp \ ${lib.escapeShellArg hostRuntimeDir}/callout.conf # Upstream's rendered settings, as a sibling the wrapper can name # without a leading slash. Refreshed unconditionally — unlike the # seeds, this one MUST track the current system, and a stale copy # would silently run yesterday's config. ln -sfn ${renderedSettings} ${lib.escapeShellArg hostRuntimeDir}/settings.conf # The wrapper. Bare filenames — see `runtimeDir`. printf rather than # a heredoc, whose terminator would depend on nix's indentation # stripping and break the next time `nix fmt` touched this block. printf 'include "settings.conf"\ninclude "callout.conf"\n' \ > ${lib.escapeShellArg hostRuntimeDir}/nats.conf.tmp chmod 0444 ${lib.escapeShellArg hostRuntimeDir}/nats.conf.tmp mv ${lib.escapeShellArg hostRuntimeDir}/nats.conf.tmp \ ${lib.escapeShellArg hostRuntimeDir}/nats.conf ''; }; }; }