# The swarm's secret store: one OpenBao for the whole swarm, in a # `swarm-bao` nixos-container. # # Today every credential in ./swarm-*.nix is minted where it is read or copied # there by a delivery unit (docs/swarm/secrets.md), which ties each secret's # lifetime to its container's. # # ⚠️ It authenticates hive clients with a CLIENT CERTIFICATE, not with the # swarm's SSO — a boot-order fact rather than a preference. This store holds # authelia's own OIDC client secret, so a client that had to obtain an authelia # token first could never start from cold. `swarm-nats` can lean on authelia # precisely because it does not store authelia's credentials. # # ⚠️ NO GATEWAY VHOST, and unlike `swarm-nats` that is not because this speaks # a non-HTTP protocol. It speaks HTTPS, so nginx *could* front it: **the client # certificate IS the authentication**, and a terminating proxy strips it, # leaving bao seeing nginx as the client for every hive in the swarm — one # identity where there must be many. Reach is loopback plus whatever # `deploy.bao.extraListenAddresses` names. # # ⚠️ THIS MODULE HAS NO OPINION ABOUT WHERE THE STORE'S IDENTITY COMES FROM. # A store must not take its certificates from an authority it will itself # distribute: reach the store to get the CA material, need a cert from that CA # to reach the store. Service↔store mTLS is therefore its own trust domain, # separate from the gateway's HTTPS certificates and from both CAs in this # tree. The cert paths are inputs this module declares no default for and never # fills in; a glue module mints that identity and points them at it. { pkgs, lib, config, ... }: let cfg = config.services.hyperhive.swarm.bao; hyperhiveCfg = config.services.hyperhive; deployCfg = hyperhiveCfg.deploy; baoDeploy = deployCfg.bao; networkCfg = hyperhiveCfg.network; swarmDomain = hyperhiveCfg.swarm.domain; # Upstream's own default, kept so its documentation matches. The raft data # lives INSIDE the container on `ephemeral = false`, the same way # ./swarm-grafana.nix keeps its sqlite database — no sibling service binds # its state out to the host. # # ⚠️ Do not bind-mount this path. Upstream pairs `StateDirectory=` with # `DynamicUser=`, which makes systemd hold the state at # `/var/lib/private/openbao` and symlink this to it; that relocation is a # rename, and a rename of an active mount point fails `EBUSY` at # `STATE_DIRECTORY` — the unit then dies before `bao` runs at all. stateDir = "/var/lib/openbao"; # The TLS material is its own small bind, on both sides of the boundary at # the same path. Separate from `stateDir` because a HOST unit writes it and # the container only reads it, and because systemd owns `stateDir` and moves # it around; this directory is ours. tlsDir = "/var/lib/swarm-bao-tls"; # Where the key surfaces for openbao to open. NOT `${tlsDir}/server-key.pem`: # that file is root-owned 0600 on the host, and the service runs as a # `DynamicUser`, so it cannot read the bind-mounted original. `LoadCredential` # is systemd's answer to exactly this — PID 1 reads the source as root and # re-exposes it inside the unit, owned by the service's own account. # `$CREDENTIALS_DIRECTORY` is `/run/credentials/`, and the config file # is rendered ahead of time, so the path is spelled out rather than read from # the environment. serverKeyCredential = "server-key"; serverKeyCredentialPath = "/run/credentials/openbao.service/${serverKeyCredential}"; # The PIN is deliberately absent here. It arrives as `BAO_HSM_PIN` from an # EnvironmentFile the provisioning unit writes, because a value interpolated # into a nix expression renders world-readable into the store. # # With no seal stanza openbao falls back to Shamir, so this attrset being # empty is the difference between a store that unseals itself and one that # needs a human after every restart. # The PKCS11 token store and its PINs live on the HOST and are bind-mounted # in. Losing them loses the sealed store — the raft data is worth nothing # without the key that unseals it — so unlike the state directory these are # deliberately a host-level fact an operator can back up. tokenStoreDir = "/var/lib/swarm-bao-token"; pinEnvFile = "${tokenStoreDir}/pin.env"; # openbao runs as a `DynamicUser`: its uid is allocated at start by the # container's PID 1, so nothing can name it ahead of time — not this # expression, and not a unit on the host. A group is the handle that # survives that, which is why provisioning happens inside the container and # hands the store over by group rather than by owner. tokenGroup = "swarm-bao-token"; # The store's half of the TPM device's ownership. Declared on BOTH sides of # the container boundary with the same pinned gid: the node is the host's and # carries a number, while the unit that opens it lives in the container, so a # name alone resolves to two different groups. See `deploy.bao.tpmGid`. tpmGroup = "swarm-bao-tpm"; # RSA rather than AES, and the label says so because a store may still hold # the AES key an earlier version of this module created: openbao's seal # accepts only AEAD mechanisms — AES-GCM or RSA-OAEP — and a TPM 2.0 offers # no GCM, so RSA-OAEP is the only one both sides have. A label resolving to # both keys at once is an error, hence a distinct one. sealKeyLabel = "swarm-bao-seal-rsa"; sealSettings = lib.optionalAttrs (baoDeploy.seal == "pkcs11") { seal.pkcs11 = { lib = "${pkgs.tpm2-pkcs11}/lib/libtpm2_pkcs11.so"; token_label = "swarm-bao"; key_label = sealKeyLabel; mechanism = "CKM_RSA_PKCS_OAEP"; }; }; # Total on a null swarm domain for the same reason every sibling module is: # the required-domain assertion in hive-network.nix should be what an operator # sees, not a coercion error from here. domainBase = if swarmDomain == null then "invalid" else swarmDomain; # Where the leaf lands for openbao to read. `tlsDir` is bind-mounted at the # same path on both sides, so the delivery below needs no second mount, and # nothing has to bind `deploy.hive-controller.tls.stateDir`, which holds the # hive CA's private key. # # The certificate and the client CA are public material and are read straight # off the mount; only the key takes the credential path above. serverCertPath = "${tlsDir}/server.pem"; serverKeyPath = "${tlsDir}/server-key.pem"; # The host-side sources, verbatim from the options — no fallback, because a # fallback is exactly the CA opinion this module must not hold. The units # below only exist when both are set (see `haveServerTls`), so these are # never forced while null. serverCertSrc = baoDeploy.serverCertFile; serverKeySrc = baoDeploy.serverKeyFile; # Both or neither: a certificate without its key configures a listener that # cannot start, and the failure would surface as openbao refusing to boot # rather than as the missing setting it is. haveServerTls = baoDeploy.serverCertFile != null && baoDeploy.serverKeyFile != null; # Every listener serves the same identity: they differ in which address # they answer on, not in who they are. Client verification is separate and # optional — a store with no `clientCaFile` still serves TLS, it just does # not authenticate the far end, which is the honest rendering of "nobody # has said what to trust yet". listenerTls = { tls_cert_file = serverCertPath; tls_key_file = serverKeyCredentialPath; } // lib.optionalAttrs (baoDeploy.clientCaFile != null) { tls_client_ca_file = clientCaPath; tls_require_and_verify_client_cert = true; }; clientCaPath = "${tlsDir}/client-ca.pem"; extraListeners = lib.listToAttrs ( lib.imap1 ( i: addr: lib.nameValuePair "extra-${toString i}" ( { type = "tcp"; address = "${addr}:${toString cfg.port}"; } // listenerTls ) ) baoDeploy.extraListenAddresses ); # Loopback is unconditional and everything else is declared, which is not # symmetry for its own sake: # # A reader on this host reaches the store through loopback, and the host # running the store is always one of its readers — so loopback is a property # of what the store IS, not of where it sits. Every other address depends on # which network the hives that read it share, and that is a deployment fact. # Bind only loopback and no remote hive can reach the store; bind only a # shared-network address and an all-local swarm cannot reach its own. # # Neither is a superset of the other, which is why this is not one address # with a conditional value. # ⚠️ openbao logs `unknown or unsupported field ` for each key here. Its # unknown-field check does not know about named listener blocks; the parser # does, honours `type`, and configures every one. Not the cause of a store # that fails to start — look at `advertise` below. listeners = { loopback = { type = "tcp"; address = "127.0.0.1:${toString cfg.port}"; } // listenerTls; } // extraListeners // metricsListener; # Metrics get their own listener rather than a flag on the one above, and # that follows from what a scraper can express: `swarm.otel.scrapeTargets` # carries no scheme and no credential, while the API listener is TLS and — # once a client CA is set — demands a client certificate. The collector # cannot reach it at all. # # `metrics_only` narrows this one to the metrics path (every other path 404s) # and the unauthenticated access is confined to loopback. **Deliberately # unauthenticated for now** — a tracked follow-up owns giving the collector a # credential, since no scrape option can carry one today. # # Exists only where a collector does: an endpoint with no reader is exposure # bought for nothing. metricsListener = lib.optionalAttrs scrapeHere { metrics = { type = "tcp"; address = "127.0.0.1:${toString baoDeploy.metricsPort}"; tls_disable = true; telemetry = { unauthenticated_metrics_access = true; metrics_only = true; }; }; }; scrapeHere = deployCfg.swarm-otel.enable; # Non-zero is what SERVES the endpoint at all — the switch is a duration, not # a boolean, so a zero here is an openbao that answers 404 on a listener # configured to do nothing else. telemetry = lib.optionalAttrs scrapeHere { telemetry = { prometheus_retention_time = "24h"; disable_hostname = true; }; }; # Raft REFUSES TO START without `cluster_addr`, and the message names neither # the setting nor the stanza: "cluster address must be set when using raft # storage". # # By name and not by address: this is the URL a reader dials, and the name the # server certificate has to carry anyway. Cluster traffic is one port up, # upstream's own convention. advertise = { api_addr = "https://${cfg.domain}:${toString cfg.port}"; cluster_addr = "https://${cfg.domain}:${toString (cfg.port + 1)}"; }; in { # One service, two namespaces, and the split decides who may set what. # # `deploy.bao.*` is mostly what the host RUNNING the store decides: whether # to run it (`enable`, declared in ./deploy.nix with its siblings), which # build, how the root key is sealed, what it listens on. # # ⚠️ Except the reader's own identity. `clientCertFile`, `clientKeyFile` and # `serverCaFile` live here too, and they are set by whoever authenticates TO # the store — which includes a host that runs none of it. A reader is defined # by holding a certificate the store accepts, never by sharing a host with # one; ./glue-matrix-bao-token.nix gates on exactly that and not on `enable`. # Reading this block as store-runner-only is what makes an off-host reader # look inexpressible when it is already supported. # # `swarm.bao.*` below is what every host in the swarm has to agree on — the # name the store answers to, its port, its container. A host that is purely # a *client* needs all of that, because it is how the client finds the store. options.services.hyperhive.deploy.bao = { package = lib.mkOption { type = lib.types.package; default = pkgs.openbao; defaultText = lib.literalExpression "pkgs.openbao"; description = '' OpenBao package to run. ⚠️ An assertion below refuses 2.7.0 or newer, which drops the built-in PKCS11 seal. ''; }; seal = lib.mkOption { type = lib.types.enum [ "pkcs11" "shamir" ]; default = "pkcs11"; example = "shamir"; description = '' How the store's root key is sealed. `pkcs11` is the default and binds the key to the host's TPM: the store unseals itself at boot, and an attacker with the disk does not get the secrets. `shamir` is openbao's own default — unseal keys held by whoever ran `bao operator init`, entered by hand after every restart — and is the honest choice for a host with no TPM. ⚠️ This is a **declaration**, and nothing at evaluation time can check it: nix runs on the build machine and cannot see the target's TPM. Saying `pkcs11` on a host without one fails when the store's container starts and its provisioning unit cannot create the token — later than activation, and on the container's journal. That is deliberate — a store that comes up sealed by software while the config says hardware is weaker than it reads, and silently so. ''; }; tpmGid = lib.mkOption { type = lib.types.int; default = 31337; example = 4242; description = '' Numeric group id owning `/dev/tpmrm0`, so the store's seal can open it. ⚠️ A **number**, and it has a default, because a name cannot do this job. The store runs as a `DynamicUser` whose uid is allocated inside its container, and NixOS allocates `tss` — and every other system group — at *activation*, per machine. A group declared on the host and a group of the same name declared in the container therefore get different ids, and the device node carries the number. Pinning one value is what makes the two sides agree. The default sits above the range NixOS auto-assigns system groups from (400–999) and above the normal-user range (1000–29999), and below the range systemd allocates `DynamicUser` ids from (61184–65519), so it collides with nothing those allocate. Override it if it collides with something this module cannot see. Only read when {option}`services.hyperhive.deploy.bao.seal` is `pkcs11`; a shamir store never touches the TPM. ''; }; serverCertFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-bao/server.pem"; description = '' Certificate the store serves, covering {option}`services.hyperhive.swarm.bao.domain`. This module declares no default and deliberately does not know what could provide one — for the same reason {option}`services.hyperhive.deploy.bao.clientCaFile` doesn't: the store never reaches for an authority. On a hive that runs the store, a glue module supplies a path as a `mkDefault`, so naming your own here wins over it. A path, never a value. ''; }; serverKeyFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-bao/server-key.pem"; description = '' Private key for {option}`services.hyperhive.deploy.bao.serverCertFile`. Both or neither — a certificate with no key is a listener that cannot start. ''; }; clientCaFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-ca/root.pem"; description = '' Authority the store validates hive **client** certificates against. This module declares no default and does not reach for the hive CA: the hive CA is a future *consumer* of the store, so a store that authenticated against it could not come up before the thing it issues. On a hive that runs the store, a glue module supplies the CA it minted for exactly this, as a `mkDefault`. Point this at something else — the swarm root, an operator's own CA — and yours wins. `null` leaves client-certificate verification off, which is only appropriate where something else authenticates the connection. ''; }; clientCertFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-bao/client.pem"; description = '' Certificate a **reader on this machine** presents to the store. The counterpart to {option}`services.hyperhive.deploy.bao.clientCaFile`, which is the store's side of the same handshake. On a hive that runs the store, a glue module supplies the leaf it minted, as a `mkDefault`. Everywhere else this is the credential an operator places by hand — the one secret that cannot come out of the store, because it is what opens it. A path, never a value. ''; }; clientKeyFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-bao/client-key.pem"; description = '' Private key for {option}`services.hyperhive.deploy.bao.clientCertFile`. Both or neither — a certificate with no key authenticates nothing. ''; }; serverCaFile = lib.mkOption { type = lib.types.nullOr lib.types.str; default = null; example = "/var/lib/swarm-bao/ca.pem"; description = '' Authority a **reader on this machine** validates the store's certificate against. Not {option}`services.hyperhive.deploy.bao.clientCaFile` with the words rearranged: that one is the store choosing which readers to trust, this one is a reader choosing which store to trust. A deployment that self-signs both ends points them at the same file and reads that as confirmation they are interchangeable — they are not, and they diverge the moment either end gets a real CA. ''; }; metricsPort = lib.mkOption { type = lib.types.port; default = 8202; description = '' Loopback port the store serves its Prometheus metrics on, scraped by a collector on this same host. Here and not beside {option}`services.hyperhive.swarm.bao.port` because no *client* of the store ever needs it: reaching this port means being the local collector, which is a property of the host running the store. ⚠️ Cannot be {option}`services.hyperhive.swarm.bao.port` **+ 1** — openbao derives every listener's cluster address as its own port plus one, so the API listener already holds that number. The default sits one above it, and its own derived cluster address one above that. ''; }; extraListenAddresses = lib.mkOption { type = lib.types.listOf lib.types.str; default = [ ]; example = [ "10.100.0.1" ]; description = '' Addresses the store listens on **in addition to loopback**, each on {option}`services.hyperhive.swarm.bao.port`. Loopback is unconditional and not listed here: the host running the store is always one of its readers. Every other address depends on which network the reading hives share with this one, and that is a deployment fact no other module's config can be read to infer — a swarm meshed over wireguard names its mesh address, one on a trusted LAN names that interface, and an all-local swarm names nothing at all. Addresses only, no port: a store reachable on two ports is a misconfiguration rather than a topology. ''; }; }; options.services.hyperhive.swarm.bao = { machine = lib.mkOption { type = lib.types.str; readOnly = true; default = "swarm-bao"; description = '' Container name. Read-only: the name appears in host paths and in `machinectl`, so it is a fact other modules may read rather than a knob. ''; }; domain = lib.mkOption { type = lib.types.str; default = "bao.${domainBase}"; defaultText = lib.literalExpression ''"bao.''${services.hyperhive.swarm.domain}"''; description = '' Name the store is reached on. A **sibling** of the swarm's other service names, not a child of any hive domain: an authority whose `nameConstraints` permit one hive's domain cannot issue for a sibling of it, so the shape of this name decides which authorities could ever sign for the store. That is a property of the name, not a choice of issuer — this module makes no such choice. ''; }; port = lib.mkOption { type = lib.types.port; default = 8200; description = '' TCP port the store listens on. Upstream's own default, kept so an operator reading OpenBao documentation finds what they expect. Swarm-wide because a client has to know it to reach the store, and the same port on every listener: which *addresses* the store answers on is the running host's business ({option}`services.hyperhive.deploy.bao.extraListenAddresses`), but which port it answers on is something the whole swarm agrees. ''; }; }; # ⚠️ Gated on `deploy.bao.enable`, and that is load-bearing rather than # tidiness: an unconditional `config` block would evaluate the seal # assertion on EVERY hive, so a hive that runs no secret store at all # would fail to build the day nixpkgs moves openbao past 2.7.0. A check # about running this service has no business firing where it is not run. config = lib.mkMerge [ # Assertions sit in their own arm, gated only on running the store, so # they still fire when the cert paths are unset — the arm below is not # evaluated in that case, and an assertion that disappears exactly when # its subject is broken would be worse than none. (lib.mkIf (hyperhiveCfg.enable && deployCfg.bao.enable) { assertions = [ { assertion = lib.versionOlder baoDeploy.package.version "2.7.0"; message = '' The swarm secret store needs openbao older than 2.7.0 (this is ${baoDeploy.package.version}). 2.7.0 moves the PKCS11 seal out of the distribution into a plugin nixpkgs does not package, so the store would come up sealed by software without saying so. See https://openbao.org/community/deprecation/ ''; } { assertion = baoDeploy.metricsPort != cfg.port + 1; message = '' services.hyperhive.deploy.bao.metricsPort is ${toString baoDeploy.metricsPort}, which openbao already uses as the API listener's cluster address (services.hyperhive.swarm.bao.port + 1). Two listeners would claim the same port, and which one wins is a race with nothing in any log about it. Pick any other free port. ''; } { assertion = haveServerTls; message = '' The swarm secret store has no server certificate: set both services.hyperhive.deploy.bao.serverCertFile and .serverKeyFile. This module defaults neither, on purpose — a store must not take its identity from an authority it will itself distribute, and service-to-store mTLS is a separate trust domain from the gateway's certificates and from either CA in this tree. A hive that runs the store normally gets both from a glue module, so reaching this means that glue is absent or something set these back to null. ''; } ]; # The name every reader dials, made resolvable where the store runs. # Cross-hive traffic always goes via the domain; only what it resolves # to varies, and a multi-host swarm is the operator's upstream DNS. This # covers the case that has no upstream record to configure. # # ⚠️ DNS only. Bao is deliberately absent from `swarm.serviceDomains` # and gets no vhost: nginx terminating TLS would strip the client # certificate, which is how the store authenticates every hive — see # this file's header. `localNames` is the one half of the sibling # pattern that applies. services.hyperhive.gateway.localNames = [ cfg.domain ]; }) (lib.mkIf (hyperhiveCfg.enable && deployCfg.bao.enable) { # The in-container unit plus the two host-side ones this module defines. # `swarm-bao-pki` and `swarm-bao-matrix-token` are declared by the glue # modules that create them, per the option's own rule — and a name # nothing defines is silently ignored, so naming them from here would # read as coverage on hives that have neither. services.hyperhive.swarm.otel.journaldUnits = [ "openbao" "swarm-bao-certs" "swarm-bao-token" ]; services.hyperhive.swarm.otel.scrapeTargets = lib.mkIf scrapeHere { # Path and query, not just `host:port`: openbao serves no `/metrics` # at all, and `/v1/sys/metrics` answers JSON unless the format is # asked for. A scrape of the default path 404s, which reads as a # dead exporter rather than a wrong address. bao = "127.0.0.1:${toString baoDeploy.metricsPort}/v1/sys/metrics?format=prometheus"; }; }) (lib.mkIf (hyperhiveCfg.enable && deployCfg.bao.enable && haveServerTls) { # Provisions the TPM-backed token the seal above names. One-shot and # idempotent on ABSENCE, never on content: regenerating a PIN would # orphan an already-sealed store, so a rebuild must not rotate it. # # ⚠️ Untestable without a TPM, and the failure is deliberately at # activation — nix evaluates on the build machine and cannot see the # target's hardware, so `seal = "pkcs11"` is a declaration this unit # either makes true or fails on. # The store's server certificate, delivered rather than bind-mounted. # # ⚠️ A copy, for three separate reasons — the last one is the one that # matters most and is the least obvious: # 1. `nixos-container` refuses to start when a bind source is missing, # and a certificate minted on this same boot does not exist yet when # the container is ordered. Same trap `hostClientSecretDir` # documents in ./swarm-authelia.nix. Ordering against whatever # mints it belongs with whatever named the path, not here. # 2. `tlsDir` is already mounted at the same path inside, so a copy # needs no second mount. # 3. A directory holding a leaf usually holds the CA's private key # beside it. Binding that directory to reach one file inside it # would hand the container authority to mint any name that CA can — # which is why this takes a path to a FILE and copies it. systemd.services.swarm-bao-certs = { description = "deliver the swarm secret store's server certificate"; before = [ "container@${cfg.machine}.service" ]; requiredBy = [ "container@${cfg.machine}.service" ]; path = [ pkgs.coreutils ]; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; }; script = '' set -euo pipefail # 0755, not 0700: the container reads the certificate and the client # CA straight off this mount as a non-root user, so it has to be able # to traverse the directory. The key inside stays 0600 and reaches # the service through `LoadCredential` instead. install -d -m 0755 ${tlsDir} # Fail loudly rather than start a store that cannot serve. The path # is configured, so a missing file means whatever was supposed to # produce it did not run or failed — either way this is where it is # cheapest to notice. Otherwise it surfaces at the TLS handshake, # several layers from the setting that caused it. for f in ${lib.escapeShellArg serverCertSrc} ${lib.escapeShellArg serverKeySrc}; do if [ ! -s "$f" ]; then echo "swarm-bao has no server certificate: $f is missing or empty." >&2 echo "That path comes from deploy.bao.serverCertFile/serverKeyFile." >&2 exit 1 fi done install -m 0644 ${lib.escapeShellArg serverCertSrc} ${tlsDir}/server.pem install -m 0600 ${lib.escapeShellArg serverKeySrc} ${tlsDir}/server-key.pem '' + lib.optionalString (baoDeploy.clientCaFile != null) '' if [ ! -s ${lib.escapeShellArg baoDeploy.clientCaFile} ]; then echo "deploy.bao.clientCaFile names ${baoDeploy.clientCaFile}, which is missing or empty." >&2 exit 1 fi install -m 0644 ${lib.escapeShellArg baoDeploy.clientCaFile} ${tlsDir}/client-ca.pem ''; }; # The device node is the host's, so its ownership is set here — the # container can only be given a group that already matches. # # ⚠️ If the host also sets `security.tpm2.tssGroup`, both rules match # `tpmrm*` and the later one wins. That is survivable (either group # reaches the device) but worth knowing before debugging a 0660 node the # store still cannot open. users.groups.${tpmGroup} = lib.mkIf (baoDeploy.seal == "pkcs11") { gid = baoDeploy.tpmGid; }; services.udev.extraRules = lib.mkIf (baoDeploy.seal == "pkcs11") '' KERNEL=="tpmrm[0-9]*", MODE="0660", GROUP="${tpmGroup}" ''; # The token store is a HOST path (see the bind mount below), so the # directory has to exist before the container starts — `bindMounts` only # adds a `RequiresMountsFor`, and nothing in nixos-containers creates a # `hostPath`. Everything else about it is the container's: the identity # that must own the contents is allocated there. systemd.services.swarm-bao-token-dir = lib.mkIf (baoDeploy.seal == "pkcs11") { description = "create the swarm secret store's PKCS11 token directory"; before = [ "container@${cfg.machine}.service" ]; requiredBy = [ "container@${cfg.machine}.service" ]; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; }; # Create-only on purpose. A plain `install -d` re-imposes 0700 root on # every boot, which locks the store's own service back out of the # directory the moment the host reboots. script = '' test -d ${tokenStoreDir} || install -d -m 0700 ${tokenStoreDir} ''; }; containers.${cfg.machine} = { autoStart = true; ephemeral = false; # Journal files on the host, not inside the container: nixpkgs hardcodes # --link-journal=try-guest, and EXTRA_NSPAWN_FLAGS expands after it. extraFlags = [ "--link-journal=host" ]; # Shared host netns, like every sibling swarm container. Unlike them the # gateway is NOT the client here (see the no-vhost note at the top), so # sharing the netns is what lets the store bind the host's own addresses # rather than a convenience for nginx. privateNetwork = false; # Only material an operator has to be able to back up crosses the # boundary — the store's identity and its seal. The raft state # deliberately does not: `ephemeral = false` keeps the container's own # /var, systemd owns `${stateDir}` through `StateDirectory=`, and # binding over it is what breaks the unit. bindMounts = { # Read-only: `swarm-bao-certs` on the host is the only writer, and # the store has no reason to modify its own identity. ${tlsDir} = { hostPath = tlsDir; isReadOnly = true; }; } // lib.optionalAttrs (baoDeploy.seal == "pkcs11") { # Writable: the library keeps its sqlite store here, and the seal # reads the token through it on every unseal. ${tokenStoreDir} = { hostPath = tokenStoreDir; isReadOnly = false; }; # The node itself, not just permission to use it: `allowedDevices` # renders `DeviceAllow=`, the cgroup gate and nothing more, while # nspawn builds its own /dev and cannot create device nodes. Whether # the seal's own user may open it is a third question, tracked apart. "/dev/tpmrm0" = { hostPath = "/dev/tpmrm0"; isReadOnly = false; }; }; # The seal talks to the TPM through the kernel's resource manager, so # the device has to cross the container boundary or the store cannot # unseal itself — which is the whole point of pkcs11 over shamir. allowedDevices = lib.optionals (baoDeploy.seal == "pkcs11") [ { node = "/dev/tpmrm0"; modifier = "rw"; } ]; config = { ... }: { imports = [ (import ./swarm-container-resolver.nix { inherit (networkCfg) bridgeIp; dnsConsumers = [ "openbao.service" ]; }) ]; system.stateVersion = "26.05"; # Shares the host netns, so its own firewall.service would rewrite # the HOST ruleset at every boot. The host firewall owns filtering. networking.firewall.enable = false; # The resolver unit imported above owns /etc/resolv.conf; leaving # resolvconf on would let host-tracking regenerate it empty. networking.resolvconf.enable = lib.mkForce false; # `${tokenGroup}` takes whatever gid this container allocates — # nothing outside reads it. `${tpmGroup}` must take the PINNED one, # because the device node it names is the host's and matching is by # number; letting this side auto-allocate is the whole bug. users.groups = lib.mkIf (baoDeploy.seal == "pkcs11") { ${tokenGroup} = { }; ${tpmGroup}.gid = baoDeploy.tpmGid; }; # Provisioned here rather than on the host because the store has to # end up reachable by openbao's `DynamicUser`, and that identity # only exists inside this container — a host unit can create the # same files and has no name to hand them to. The store itself # stays on the host mount: this changes who writes it, not where it # lives. systemd.services.swarm-bao-token = lib.mkIf (baoDeploy.seal == "pkcs11") { description = "provision the swarm secret store's TPM-backed PKCS11 token"; before = [ "openbao.service" ]; requiredBy = [ "openbao.service" ]; path = [ pkgs.openssl pkgs.tpm2-pkcs11 pkgs.tpm2-tools ]; serviceConfig = { Type = "oneshot"; RemainAfterExit = true; }; script = '' set -euo pipefail install -d -m 0770 -g ${tokenGroup} ${tokenStoreDir} # Required whenever the store is not at its default location, or the # library cannot find the token the seal asks for. export TPM2_PKCS11_STORE=${tokenStoreDir} # Absence is the only trigger. `openssl rand` is the same generator # the grafana admin key uses; the value never passes through a nix # expression, which would render it world-readable into the store. for p in so-pin user-pin; do if [ ! -e ${tokenStoreDir}/$p ]; then ( umask 077; openssl rand -hex 16 > ${tokenStoreDir}/$p ) chmod 0400 ${tokenStoreDir}/$p fi done if [ ! -e ${tokenStoreDir}/tpm2_pkcs11.sqlite3 ]; then pid=$(tpm2_ptool init --path ${tokenStoreDir} | sed -n 's/.*id: //p') tpm2_ptool addtoken --path ${tokenStoreDir} --pid="$pid" \ --label=swarm-bao \ --sopin="$(cat ${tokenStoreDir}/so-pin)" \ --userpin="$(cat ${tokenStoreDir}/user-pin)" fi # Keyed on the label, not on the store's existence: a store # provisioned by an earlier version of this module has a token # and an unusable AES key, and has to gain this one without # being destroyed. See `sealKeyLabel` for why it is RSA. # # Substitution rather than a `grep -q` pipeline: `grep -q` # exits at the first match and SIGPIPEs the lister, which # `pipefail` reports as failure — the one wrong answer here # adds a second key under the label and breaks the seal. objects=$(tpm2_ptool listobjects --path ${tokenStoreDir} \ --label=swarm-bao || true) case "$objects" in *${sealKeyLabel}*) ;; *) tpm2_ptool addkey --path ${tokenStoreDir} --label=swarm-bao \ --userpin="$(cat ${tokenStoreDir}/user-pin)" \ --algorithm=rsa2048 --key-label=${sealKeyLabel} ;; esac # Outside the branch above on purpose: a store provisioned by an # earlier version of this module is root-owned and unreadable to # the seal, and only the pins stay private to root. chgrp ${tokenGroup} ${tokenStoreDir}/tpm2_pkcs11.sqlite3 chmod 0660 ${tokenStoreDir}/tpm2_pkcs11.sqlite3 ( umask 077; printf 'BAO_HSM_PIN=%s\n' "$(cat ${tokenStoreDir}/user-pin)" > ${pinEnvFile} ) chmod 0400 ${pinEnvFile} ''; }; services.openbao = { enable = true; package = baoDeploy.package; settings = { listener = listeners; storage.raft.path = stateDir; } // advertise // telemetry // sealSettings; }; # The private key crosses the user boundary here, not on the mount. # `swarm-bao-certs` installs it 0600 root-owned, and the unit runs # as a `DynamicUser`, so the bind-mounted file is unreadable to it — # PID 1 opens the source as root and re-exposes it under # `${serverKeyCredentialPath}`, owned by the service's own account. # # One assignment, not two: `serviceConfig.X = …` beside a # `serviceConfig = …` is a duplicate attribute inside a single # attrset literal and does not parse. Module merging happens across # `config` blocks, not within a literal — so the seal's half joins # with `optionalAttrs`. systemd.services.openbao.serviceConfig = lib.optionalAttrs haveServerTls { LoadCredential = [ "${serverKeyCredential}:${serverKeyPath}" ]; } # The PIN reaches openbao as an environment variable read from a # 0400 file the provisioning unit wrote — never as a value in this # expression, which would render it world-readable into the store. # `TPM2_PKCS11_STORE` is required because the store is not at the # library's default location. // lib.optionalAttrs (baoDeploy.seal == "pkcs11") { EnvironmentFile = pinEnvFile; Environment = [ "TPM2_PKCS11_STORE=${tokenStoreDir}" ]; # `DynamicUser` implies `ProtectSystem=strict`, which leaves the # bind mount read-only to this unit however it is owned, and the # pkcs11 library opens its sqlite store read-write. ReadWritePaths = [ tokenStoreDir ]; # Two groups, two different jobs: the store's sqlite file, and # the TPM device the seal opens through it. `CapabilityBounding # Set=` is empty upstream, so there is no `CAP_DAC_OVERRIDE` to # fall back on — group membership is the only way in. SupplementaryGroups = [ tokenGroup tpmGroup ]; }; # ⚠️ Upstream sets `restartIfChanged = false` on this unit, on # purpose: a restart SEALS the store and disconnects every client. # So a change to the settings above does NOT take effect on # `nixos-rebuild switch` — it lands in the config file and waits. # Restarting is an operator action with an unseal on the far side of # it, which is why nothing here tries to be clever about it. }; }; }) ]; }