Measured against a running server built from this module's own rendered config, not a hand-written one -- which is the fix to the method as much as to the file. Two shapes died in a row: a bare name authenticates anyone who sends it, and a name plus an nkey refuses to start at all (nkey users do not take usernames or passwords). A malformed key is fail-closed too, so the only way to a live server here is a real key whose seed nobody but the responder holds.
287 lines
12 KiB
Nix
287 lines
12 KiB
Nix
{
|
|
pkgs,
|
|
lib,
|
|
config,
|
|
...
|
|
}:
|
|
let
|
|
cfg = config.services.hyperhive.swarm.nats;
|
|
autheliaCfg = config.services.hyperhive.swarm.authelia;
|
|
autheliaUrl = autheliaCfg.url;
|
|
|
|
# The account the callout responder authenticates as, and the account
|
|
# authorized clients are placed in. Two accounts rather than one: an
|
|
# account is NATS' isolation boundary, so a responder that shares an
|
|
# account with its clients can be published to by the things it
|
|
# authorizes.
|
|
calloutAccount = "AUTH";
|
|
clientAccount = "APP";
|
|
in
|
|
{
|
|
# The swarm's message queue: one NATS server, reached by every hive.
|
|
#
|
|
# ⚠️ There is deliberately NO gateway vhost here, and this is the first
|
|
# swarm service where that is true — the next reader will go looking for
|
|
# one. NATS speaks its own TCP protocol rather than HTTP, so nginx
|
|
# cannot front it the way it fronts the forge, matrix and authelia.
|
|
# Cross-hive reach is the wireguard mesh; `gateway.localNames` and the
|
|
# per-service vhost pattern do not apply.
|
|
|
|
options.services.hyperhive.swarm.nats = {
|
|
enable = lib.mkOption {
|
|
type = lib.types.bool;
|
|
default = false;
|
|
description = ''
|
|
Run the swarm's message queue in a `swarm-nats` container on this
|
|
host. A swarm has one queue, so this belongs on the same host as
|
|
the rest of the shared services.
|
|
|
|
Off by default, and off means *absent*: no container is created
|
|
and nothing else in the evaluated config changes.
|
|
'';
|
|
};
|
|
|
|
# ⚠️ Deliberately NO `package` option, unlike this module's siblings.
|
|
# `services.nats` upstream does not expose one — it resolves
|
|
# `pkgs.nats-server` itself — so an option here would either be
|
|
# ignored or need an overlay to mean anything, and an option that
|
|
# does not control what it names is worse than its absence. Pin the
|
|
# build with `nixpkgs.overlays` if you need to.
|
|
|
|
port = lib.mkOption {
|
|
type = lib.types.port;
|
|
default = 4222;
|
|
description = ''
|
|
TCP port the queue listens on. 4222 is upstream's default and
|
|
sits outside hyperhive's claimed ranges (dashboard 7000, forge
|
|
3000, matrix 8008, every agent in 8100..8999 via FNV-1a hash).
|
|
'';
|
|
};
|
|
|
|
clientId = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "swarm-nats";
|
|
description = ''
|
|
OAuth2 client id the queue's authentication path identifies
|
|
itself with. Must match the `id` of the corresponding entry in
|
|
`services.hyperhive.swarm.authelia.oidc.clients` — which this
|
|
module contributes for you when both run on this host.
|
|
'';
|
|
};
|
|
|
|
calloutUserPublicKey = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
example = "UDXU4RCSJNZOIQHZNWXHXORDPRTGNJAHAHFRGZNEEJCPQTT2M7NLCBBQ";
|
|
description = ''
|
|
Public half of the **user** nkey the auth-callout responder
|
|
authenticates as.
|
|
|
|
`auth_callout.auth_users` exempts this identity from needing
|
|
callout approval — it is the one that answers auth requests, so
|
|
it cannot wait for itself. **That exemption is exactly why it
|
|
needs a credential of its own**: without one the escape hatch is
|
|
an open door, and on a container sharing the host netns it is an
|
|
open door reachable from every agent container.
|
|
|
|
An nkey rather than a password for the same reason
|
|
`calloutIssuerPublicKey` is: only the public half appears here,
|
|
and nix renders it into the world-readable store harmlessly. The
|
|
seed reaches the responder and nothing else, so until the
|
|
responder exists **nobody can authenticate as this user at all**
|
|
— which is what makes a hive with no responder genuinely closed
|
|
rather than merely gated.
|
|
|
|
Required when `enable` is set.
|
|
'';
|
|
};
|
|
|
|
calloutIssuerPublicKey = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
example = "ACYR44YO3XZRZBJIYLI5SL6LOPIW37JTD52LNOBHUE34XMH7N5ABMFJH";
|
|
description = ''
|
|
Public half of the account nkey whose signature the server
|
|
accepts on a user JWT minted by the auth-callout responder.
|
|
|
|
**A public key, and therefore a value rather than a path** —
|
|
the deliberate exception to the rule that credentials are
|
|
`*File` options. It is published to every client that connects
|
|
and its whole job is to be widely known; the matching *seed* is
|
|
the secret, is never named here, and reaches only the responder.
|
|
|
|
Required when `enable` is set. Without it the server has no
|
|
issuer to trust and no client can be authorized — which is the
|
|
fail-closed state described below, but arrived at by accident
|
|
rather than on purpose, so it fails at eval instead.
|
|
'';
|
|
};
|
|
};
|
|
|
|
config = lib.mkIf cfg.enable {
|
|
assertions = [
|
|
{
|
|
# Fail at EVAL, not at boot: a queue that comes up unable to
|
|
# authenticate anyone presents as every client hanging, which is
|
|
# several layers from "the operator never set the issuer".
|
|
assertion = cfg.calloutIssuerPublicKey != "";
|
|
message = ''
|
|
services.hyperhive.swarm.nats.enable requires
|
|
nats.calloutIssuerPublicKey — the public half of the account
|
|
nkey that signs user JWTs for this queue.
|
|
|
|
It is public and belongs in config; the matching seed is a
|
|
secret and is delivered to the callout responder instead. See
|
|
docs/swarm/secrets.md for which is which.
|
|
'';
|
|
}
|
|
{
|
|
# Without this the callout-exempt user has no credential, and
|
|
# NATS accepts `CONNECT {"user":"auth"}` from anyone. An eval
|
|
# failure is the only place to catch that: the rendered config is
|
|
# valid, the server starts, and the hole is invisible until
|
|
# somebody connects.
|
|
assertion = cfg.calloutUserPublicKey != "";
|
|
message = ''
|
|
services.hyperhive.swarm.nats.enable requires
|
|
nats.calloutUserPublicKey — the public half of the user nkey
|
|
the auth-callout responder authenticates as.
|
|
|
|
It is exempt from callout approval by design, which is exactly
|
|
why it needs its own credential: a `users` entry with a name
|
|
and no key authenticates anyone who sends that name.
|
|
'';
|
|
}
|
|
{
|
|
assertion = autheliaUrl != null;
|
|
message = ''
|
|
services.hyperhive.swarm.nats.enable requires
|
|
services.hyperhive.swarm.authelia.url — the queue authenticates
|
|
clients by validating tokens that authelia issued.
|
|
|
|
It defaults to this host's own instance only when this host
|
|
runs authelia. A hive that federates with a swarm sets it
|
|
explicitly to wherever that provider lives.
|
|
'';
|
|
}
|
|
];
|
|
|
|
# One declaration, two readers. The queue knows which client id it
|
|
# authenticates under; making the operator restate it in authelia's
|
|
# client list would be a second source of truth for a string whose
|
|
# mismatch is an opaque 401 from the token endpoint.
|
|
#
|
|
# `client_credentials` rather than an authorization-code flow: a
|
|
# queue's clients are daemons with nobody to redirect, so a
|
|
# non-interactive grant is what makes a hive able to authenticate at
|
|
# all. No redirect URI exists or is wanted.
|
|
services.hyperhive.swarm.authelia.oidc.clients = lib.mkIf autheliaCfg.enable [
|
|
{
|
|
id = cfg.clientId;
|
|
description = "HyperHive swarm queue";
|
|
redirectUris = [ ];
|
|
}
|
|
];
|
|
|
|
containers.swarm-nats = {
|
|
autoStart = true;
|
|
ephemeral = false;
|
|
# Shared host netns, like every sibling container.
|
|
#
|
|
# ⚠️ Which is exactly why the server below must refuse everyone
|
|
# until the callout responder exists: on this netns the queue is
|
|
# reachable from every agent container on the hive, so an
|
|
# unauthenticated interim state would be a hole rather than a
|
|
# rough edge.
|
|
privateNetwork = false;
|
|
config =
|
|
{ ... }:
|
|
{
|
|
system.stateVersion = "26.05";
|
|
|
|
# Shared host netns: this container's own firewall.service
|
|
# would rewrite the HOST ruleset at every boot. The host
|
|
# firewall owns all filtering.
|
|
networking.firewall.enable = false;
|
|
# Keep the host-copied /etc/resolv.conf intact — resolvconf's
|
|
# host-tracking regenerates it empty, since the host's copy
|
|
# does not cross the boundary after start.
|
|
networking.resolvconf.enable = lib.mkForce false;
|
|
|
|
services.nats = {
|
|
enable = true;
|
|
|
|
serverName = "swarm-nats";
|
|
port = cfg.port;
|
|
settings = {
|
|
# Two accounts, and the callout user lives in neither of
|
|
# the accounts it authorizes into.
|
|
accounts = {
|
|
# ⚠️ The nkey is not decoration and its absence was a real
|
|
# hole: a `users` entry carrying only a `user` name has no
|
|
# credential, and `CONNECT {"user":"auth"}` is then
|
|
# accepted with no password at all. Since the name is a
|
|
# literal in this public module, that made the
|
|
# callout-exempt identity walk-in-able from every
|
|
# container on the shared netns — the same class of hole
|
|
# this module exists to close, moved rather than fixed.
|
|
# Caught in review on the first version of this file.
|
|
# An nkey and NOTHING else, both halves measured against a
|
|
# running server rather than reasoned about:
|
|
#
|
|
# { user = "auth"; } → `CONNECT {"user":"auth"}`
|
|
# is accepted with no
|
|
# credential at all
|
|
# { user = "auth"; nkey = "U…"; } → refuses to START:
|
|
# "Nkey users do not take
|
|
# usernames or passwords"
|
|
# { nkey = "U…"; } → what this is
|
|
#
|
|
# A malformed key is fail-closed too: the server exits with
|
|
# "Not a valid public nkey for a user" rather than starting
|
|
# with a hole. So the only way to get a live server here is
|
|
# a real key whose seed nobody but the responder holds.
|
|
${calloutAccount}.users = [ { nkey = cfg.calloutUserPublicKey; } ];
|
|
${clientAccount} = { };
|
|
};
|
|
|
|
authorization = {
|
|
timeout = "2s";
|
|
# 🔒 THIS BLOCK IS THE FAIL-CLOSED STATE, and it is the
|
|
# measured one rather than the obvious one.
|
|
#
|
|
# Measured on the pinned nats-server 2.14.1: both
|
|
# `authorization { }` and `authorization { users: [] }`
|
|
# accept an anonymous client and answer PONG — they read
|
|
# like "authorize nobody" and are wide open. An
|
|
# auth_callout block sets `auth_required` and refuses
|
|
# every client whose credential no responder has
|
|
# approved, so a config whose responder does not exist
|
|
# yet denies everyone.
|
|
#
|
|
# `nats-server -t` calls all three valid; it parses, it
|
|
# does not authenticate. Only running them tells the
|
|
# difference.
|
|
#
|
|
# ⇒ this is both the safe interim state and the final
|
|
# shape. Nothing here has to be swapped out when the
|
|
# responder lands beside it — it only starts being able
|
|
# to say yes.
|
|
auth_callout = {
|
|
issuer = cfg.calloutIssuerPublicKey;
|
|
auth_users = [ cfg.calloutUserPublicKey ];
|
|
account = calloutAccount;
|
|
};
|
|
};
|
|
};
|
|
};
|
|
|
|
# The server binary, so an operator with a shell in here can
|
|
# run `nats-server -t` against the generated config. The unit
|
|
# resolves ExecStart through the store path and puts nothing
|
|
# on PATH.
|
|
environment.systemPackages = [ pkgs.nats-server ];
|
|
};
|
|
};
|
|
};
|
|
}
|