hyperhive/nix/host-modules/swarm.nix
atlas f2840612c0 feat(3167): options + certificate name for the swarm UI
New swarm-ui module: enable (derived from swarm.controller.enable - the
UI reads that daemon's state over its socket, so the host that runs the
controller is the host that can serve the UI), domain (defaults to the
swarm apex; an option so a hive can pin it like forge/matrix can), and
package.

Adds the name to swarm.serviceDomains, which is both the services
sub-CA's nameConstraints set and the leaf's SAN set. The apex is a
SIBLING of forge./chat./auth., not a parent, so nothing issues for it
implicitly - left out, the vhost falls back to the hive leaf and the
swarm's front page opens with a name mismatch.

Asserts the UI domain differs from the hive domain: the gateway's
default server already answers for the latter, and two vhosts claiming
one server_name resolve to whichever nginx picks rather than erroring.
2026-08-12 17:29:13 +02:00

305 lines
13 KiB
Nix

# The swarm's directory: one entry per hive, **including this one**,
# identical on every host in the swarm. `services.hyperhive.hiveName`
# says which entry is us, and `peerHives` below derives the rest.
#
# Why a directory rather than a per-host peer list: every field here is
# intrinsic to the hive it describes — none of them says anything about
# the *pair*. A list where every field is intrinsic is a directory each
# host was keeping its own copy of, which is O(n²) duplication that
# deduplicates without loss. It is also a correctness gain: two hosts
# could hold different endpoints for the same third hive and nothing
# detected it. One entry per hive makes that unrepresentable.
#
# Consumed by hive-c0re's environment (HYPERHIVE_PEERS — see
# ./hive-c0re), identity.rs + the dashboard's P33RS tab, and the mesh in
# ./swarm-wireguard.nix. The mesh lives there rather than here because
# bringing up an interface is host networking rather than swarm
# bookkeeping, and a host that runs no hive still needs it.
{
lib,
config,
...
}:
let
cfg = config.services.hyperhive;
swarmCfg = cfg.swarm;
# Public hostnames of the swarm's own services, in declaration order.
# `serviceDomains` below is this set sorted + deduplicated.
#
# ⚠️ These are NOT required to be under `swarm.domain`. An earlier
# revision asserted that, reasoning that the services sub-CA is
# constrained to the swarm's tree — but the sub-CA is constrained to
# the **configured names** (./swarm-ca.nix) and the swarm root carries
# no name constraints at all, so any configured name is issuable. The
# assertion encoded an intended shape, not a property of the code, and
# it rejected the supported migration path: a hive pinning its old
# `forge.<hive domain>` while joining a swarm at a different apex.
serviceDomains' = [
swarmCfg.forge.domain
swarmCfg.matrix.gatewayHost
swarmCfg.authelia.domain
]
# The swarm UI's name is a SIBLING of the other three, not a parent of
# them — the apex is as much a name needing a certificate as
# `forge.<apex>` is, and no CA in the hierarchy issues for it
# implicitly. Left out, its vhost falls back to the hive leaf and the
# swarm's front page opens with a name mismatch.
++ lib.optional swarmCfg.ui.enable swarmCfg.ui.domain;
in
{
options.services.hyperhive.swarm.hives = lib.mkOption {
type = lib.types.attrsOf (
lib.types.submodule (
{ name, ... }:
{
options = {
domain = lib.mkOption {
type = lib.types.str;
# `<name>.<swarm.domain>` is a derivation from two values an
# operator had to state explicitly (both are required), not a
# guess — and it is what makes this directory worth copying:
# a conventional swarm is `{ pr1ma = { }; umbra = { }; }`,
# names only, with a non-conventional hive saying so and only
# that. A shared file is read far more often than written.
#
# Total rather than a throw when `swarm.domain` is unset, for
# the reason ./hive-network.nix:155 gives in full: defaults
# that interpolate the domain are forced *while the assertion
# list evaluates*, so a throw here would replace the message
# naming the missing option with a coercion error naming this
# one. `.invalid` is reserved (RFC 2606) and fails loudly at
# resolution if it ever escaped — which the required-domain
# assertion is there to stop.
default = if swarmCfg.domain == null then "${name}.invalid" else "${name}.${swarmCfg.domain}";
defaultText = lib.literalExpression ''"''${name}.''${services.hyperhive.swarm.domain}"'';
example = "lab.example.com";
description = ''
Public DNS domain this hive occupies used for dashboard
links, peer HTTPS checks and Matrix federation discovery.
Defaults to `<name>.<swarm.domain>`, the convention every
hive in a swarm follows, so a conventional directory is
names only. Set it for a hive that is addressed by
something else.
'';
};
certFingerprint = lib.mkOption {
type = lib.types.nullOr lib.types.str;
default = null;
example = "sha256:b1946ac92492d2347c6235b4d2611184a3f5b6cae6c19d6e3c2f0a8e7d4c9f12";
description = ''
Expected TLS certificate fingerprint for this hive's HTTPS
endpoint. Null = trust the CA bundle which for a hive
inside the swarm CA hierarchy is the normal case, since
every hive under the swarm root already chains to it.
Set it to pin a leaf that no CA in the bundle vouches for.
Format: the literal `sha256:` followed by exactly 64
hex digits (case-insensitive, no colon separators) the
SHA-256 digest of the DER-encoded leaf certificate.
Generate with `openssl x509 -noout -fingerprint -sha256`,
then strip the colons and prepend `sha256:`. A malformed
value is ignored with a warning rather than weakening
trust. See docs/swarm/README.md for the full recipe.
Scopes only to hive-c0re's own peer HTTPS checks it does
NOT help Matrix federation, which validates against the
container's trust bundle. There is no per-hive CA field to
cover that case any more: the swarm root is the trust path
(see ./swarm-ca.nix).
'';
};
wireguardPublicKey = lib.mkOption {
type = lib.types.nullOr lib.types.str;
default = null;
example = "base64pubkey=";
description = ''
WireGuard public key for this hive's host. Required when
`services.hyperhive.swarm.wireguard.enable = true` and
you want this hive reachable over the mesh. Null = TLS-
only peering (public internet, no mesh tunnel).
'';
};
wireguardEndpoint = lib.mkOption {
type = lib.types.nullOr lib.types.str;
default = null;
example = "203.0.113.1:51820";
description = ''
WireGuard endpoint for this hive in `host:port` form.
Null = this hive has no reachable endpoint, so the tunnel
is initiated from the other side.
Reads like a fact about the relationship and is not: it
says whether *this* hive can be dialled, which every other
hive in the swarm needs the same answer to.
'';
};
wireguardAddress = lib.mkOption {
type = lib.types.nullOr lib.types.str;
default = null;
example = "10.100.0.2/32";
description = ''
IP address (with prefix) of this hive's host on the
WireGuard mesh. Used as the `allowedIPs` for its
WireGuard config entry and injected into `HYPERHIVE_PEERS`
so hive-c0re can route intra-swarm traffic to the mesh
address rather than the public domain. Required to include
a hive in the mesh (entries missing this field are
silently excluded from `wg-hive`).
'';
};
};
}
)
);
default = { };
example = {
pr1ma = {
wireguardAddress = "10.100.0.1/32";
wireguardEndpoint = "203.0.113.1:51820";
};
edge = {
domain = "edge.elsewhere.example";
wireguardAddress = "10.100.0.2/32";
};
};
description = ''
Every hive in this swarm, keyed by `hiveName` **including this
host's own hive**. The same attrset is meant to be identical on
every host in the swarm, so it can be written once and shared;
`services.hyperhive.hiveName` is what makes a given host read it
as "me and four others" rather than "five peers".
Each entry needs no fields at all in the conventional case: a
hive's `domain` defaults to `<name>.<swarm.domain>`, so the whole
directory is usually a list of names.
It must contain an entry for `hiveName`, which is asserted this
host's own address is read out of it (it is where
`services.hyperhive.domain` derives from), and a hive that lists
everyone but itself would otherwise derive its own peer set as
*everything* and peer with itself.
'';
};
options.services.hyperhive.swarm.peerHives = lib.mkOption {
type = lib.types.attrsOf (lib.types.attrsOf lib.types.unspecified);
readOnly = true;
internal = true;
description = ''
Read-only: `hives` minus this host's own entry. Derived once here
rather than in each consumer, because "everything that isn't me"
is a filter four different modules were re-implementing and only
one of them has to be wrong for a hive to peer with itself.
'';
};
options.services.hyperhive.swarm.serviceDomains = lib.mkOption {
type = lib.types.listOf lib.types.str;
readOnly = true;
internal = true;
description = ''
Read-only: the public hostnames of the swarm's own services, in a
stable sorted order. Second derived set alongside `peerHives`, and
here for the same reason the CA that name-constrains these and
the leaf that carries them as SANs must agree exactly, and two
modules each assembling the list is how they stop agreeing.
Sorted and deduplicated deliberately: consumers compare this list
against what they issued last time to decide whether to re-issue,
so an unstable order would churn a certificate that other things
are meant to pin.
'';
};
config = {
services.hyperhive.swarm.peerHives = lib.filterAttrs (name: _: name != cfg.hiveName) swarmCfg.hives;
services.hyperhive.swarm.serviceDomains = lib.sort (a: b: a < b) (
lib.unique (lib.filter (d: d != null && d != "") serviceDomains')
);
assertions = [
{
# An EMPTY `hives` fires this too, deliberately: since
# `swarm.domain` became required, every hive is in a swarm — a
# swarm of one is still a swarm — so a directory with no entry
# for this host is missing one either way. It also has to fire
# here, because this host's own domain is now read out of the
# directory: without the entry `services.hyperhive.domain` is
# null and the generic required-domain assertion in
# ./hive-network.nix would fire instead, naming an option the
# operator should no longer be setting.
#
# Guarded on `hiveName != null` so the required-hiveName
# assertion in ./hyperhive.nix is what fires for that case —
# two assertions naming the same missing value is noise.
assertion = cfg.hiveName == null || swarmCfg.hives ? ${cfg.hiveName};
message = ''
services.hyperhive.swarm.hives has no entry for this hive
(services.hyperhive.hiveName = "${toString cfg.hiveName}").
`hives` describes every hive in the swarm including this one,
so that every host can share one identical attrset, and this
hive's own domain is derived from its entry. Add:
services.hyperhive.swarm.hives."${toString cfg.hiveName}" = { };
No fields are needed: `domain` defaults to
`<name>.<swarm.domain>`. Set it in the entry if this hive is
addressed by something else.
Declared hives: ${lib.concatStringsSep ", " (lib.attrNames swarmCfg.hives)}
'';
}
];
};
# `enableRequiredServices` is declared in ./swarm-required-services.nix
# together with the per-service `enable`s it asserts — it is a
# deployment-shape switch rather than swarm bookkeeping, so it lives
# with its consequences instead of here.
options.services.hyperhive.swarm.snapshotStore = {
address = lib.mkOption {
type = lib.types.nullOr lib.types.str;
default = null;
example = "10.100.0.1";
description = ''
Mesh address of the swarm's snapshot store the single
`btrfs receive` endpoint every hive in this swarm pushes agent
snapshots to. Bare IP, no prefix.
There is exactly **one** store per swarm, not one per peer: the
receiver keys destinations by *agent*, so an agent that migrates
between hives keeps a single unbroken incremental chain. Per-hive
stores would split that chain in two, which is the case the store
exists to serve.
Null means this swarm has no store configured, and pushing fails
saying so rather than guessing an address. Set it on every hive
that pushes; the receiving host separately sets
`services.hyperhive.snapshotStore.enable`.
'';
};
port = lib.mkOption {
type = lib.types.port;
default = 51821;
description = ''
TCP port the swarm's snapshot store listens on. Must match the
receiving host's `services.hyperhive.snapshotStore.port`.
Defaulted (unlike `address`) because it is a shared convention
both sides read from the same option docs whereas an address
is deployment-specific and cannot be guessed.
'';
};
};
}