hyperhive/nix/modules/hive-network.nix
atlas eb61660d35 chore(nix): replace tracker tags with prose in nix comments
Part of the tracker-tag cleanup: the hive convention is prose, not
issue-tracker tags, in code. Reword the 21 tags in the nix tree
(flake.nix + the hive-c0re/ci/gateway/network modules) to describe
the thing they pointed at, preserving the context without the tag.

Comment-only — no eval or logic change. Validated with nix fmt
(no reformatting) and nix flake check --no-build (all checks
evaluate clean); the full build check was skipped locally because
the shared remote builder is degraded, so CI will exercise the
build derivations once the runner recovers.
2026-06-09 11:25:56 +02:00

226 lines
8.7 KiB
Nix

{
lib,
config,
...
}:
let
cfg = config.services.hyperhive.network;
in
{
# Hive-internal network — host-side bridge + per-agent DNS resolver.
# Containers stay on shared host netns at v1; this module stands the
# bridge + resolver up so the endpoint is in place before network
# isolation flips containers to private netns. Full design: docs/network.md.
options.services.hyperhive.network = {
enable = lib.mkOption {
type = lib.types.bool;
default = config.services.hyperhive.enable;
defaultText = lib.literalExpression "config.services.hyperhive.enable";
example = false;
description = ''
Stand up the hive-internal bridge + dnsmasq resolver.
Defaults to `config.services.hyperhive.enable` so it comes
on automatically with the rest of hyperhive. Requires
`services.hyperhive.domain` to be set the dnsmasq resolver
is authoritative for `<hive-domain>` and its sub-domains.
When enabled: a bridge interface (`bridgeName`) appears on the
host with `bridgeIp` assigned, and the hive-gateway container
runs a dnsmasq listening on that IP for `<hive-domain>` +
sub-domains. Agent containers still default to shared host
netns the endpoint is up but only used once
`isolateContainers = true` flips containers to private netns
+ veth peers on this bridge.
'';
};
bridgeName = lib.mkOption {
type = lib.types.str;
default = "hive-br0";
example = "h0";
description = ''
Name of the host-side bridge interface the hive uses for
inter-container traffic. Kept short so it survives the
IFNAMSIZ (15-char) cap, and prefixed so it's obviously
hive-managed in `ip link` output.
'';
};
bridgeIp = lib.mkOption {
type = lib.types.str;
default = "10.42.0.1";
example = "172.30.0.1";
description = ''
IPv4 address assigned to the bridge interface on the host
side. Becomes the DNS server address agents point at (and
the upstream the gateway proxies to once netns isolation
lands). Default `10.42.0.1` is in RFC 1918 space and
unlikely to clash with operator's existing setup; override
if a different range is already in use.
'';
};
bridgePrefixLength = lib.mkOption {
type = lib.types.int;
default = 24;
example = 16;
description = ''
Netmask prefix length for the bridge subnet. Default `/24`
gives 254 usable per-agent addresses, enough for any
single-host hive. Operator with a larger swarm or a tighter
addressing scheme overrides.
'';
};
upstreamDns = lib.mkOption {
type = lib.types.listOf lib.types.str;
default = [
"1.1.1.1"
"9.9.9.9"
];
example = [
"192.168.1.1"
"8.8.8.8"
];
description = ''
Upstream DNS servers dnsmasq forwards non-hive queries to.
Defaults to Cloudflare + Quad9. Override for operators on
private networks who need a specific resolver (corporate
DNS, pi-hole, etc.). The hive resolver itself stays
authoritative for `<hive-domain>` and its sub-domains
regardless of upstream choice.
'';
};
isolateContainers = lib.mkOption {
type = lib.types.bool;
default = false;
example = true;
description = ''
Flip agent containers from shared host netns to private netns.
When true, each agent container gets a dedicated veth
pair attached to `bridgeName` and a deterministic IP from
the bridge subnet. The bridge (already up when `enable = true`)
becomes the sole routed path between the host and agent
containers.
The host-side nix effect (this option) is:
- Sets `HIVE_NETWORK_ISOLATION=1` in the c0re service env so
the Rust lifecycle knows to pass `--private-network` +
bridge settings when creating/updating containers.
- Enables IP forwarding + NAT so agents can reach the internet
through the host.
- Adds a firewall rule DROP'ing traffic from the bridge subnet
to the host's loopback addresses defence-in-depth so a
compromised agent can't reach the c0re dashboard (already
bound to 127.0.0.1) or other host-loopback services even if
the routing table somehow leaks.
- Allows HTTP/HTTPS (80/443) traffic from the bridge subnet to
the host so agents can reach the gateway container (shared
host netns, proxies the operator's per-agent UI).
**Prerequisite**: all agents must have
`hyperhive.web.useUnixSocket = true` before enabling isolation.
Agents that still bind TCP on `0.0.0.0:<port>` will be
reachable at their bridge IP from other agents on the same
subnet defeating the isolation goal. The gateway routes via
unix sockets so gateway reach still works regardless.
**Migration**: containers are destroyed and re-created when
the network isolation flag flips. Operator state under
`/agents/<name>/state/` is bind-mounted and survives; the
container rootfs (nix store paths) is recreated cleanly.
**Rust counterpart**: `hive-c0re` reads `HIVE_NETWORK_ISOLATION`
and `HIVE_NETWORK_BRIDGE` from its service env and uses them in
`lifecycle::set_nspawn_flags` to configure `PRIVATE_NETWORK`,
`LOCAL_ADDRESS`, and `HOST_BRIDGE` in each container's
`nixos-containers/<name>.conf`. See `docs/network.md` for the
full design.
'';
};
};
config = lib.mkMerge [
(lib.mkIf cfg.enable {
assertions = [
{
assertion = config.services.hyperhive.domain != null;
message = ''
services.hyperhive.network.enable = true requires
services.hyperhive.domain to be set the resolver needs a
domain to be authoritative for. Either pin a hostname
(`services.hyperhive.domain = "example.com";`) or set
`services.hyperhive.network.enable = false` explicitly.
'';
}
];
# Virtual bridge — veth pairs attach when isolateContainers flips on.
networking.bridges.${cfg.bridgeName}.interfaces = [ ];
# Bridge IP — dnsmasq (in the gateway container) binds here.
networking.interfaces.${cfg.bridgeName}.ipv4.addresses = [
{
address = cfg.bridgeIp;
prefixLength = cfg.bridgePrefixLength;
}
];
# DNS only on the bridge interface — no external amplification surface.
networking.firewall.interfaces.${cfg.bridgeName} = {
allowedUDPPorts = [ 53 ];
allowedTCPPorts = [ 53 ];
};
})
# Guard: fires unconditionally on isolateContainers so the assertion
# is not silently swallowed when enable=false.
(lib.mkIf cfg.isolateContainers {
assertions = [
{
assertion = cfg.enable;
message = ''
services.hyperhive.network.isolateContainers = true requires
services.hyperhive.network.enable = true (the bridge and
resolver must be running before isolation is flipped on).
'';
}
];
})
# Container isolation overlay — see docs/network.md#container-isolation.
(lib.mkIf (cfg.enable && cfg.isolateContainers) {
# Agents route internet traffic via the bridge; NAT masquerades their RFC-1918 IPs.
boot.kernel.sysctl."net.ipv4.ip_forward" = 1;
networking.nat = {
enable = true;
internalInterfaces = [ cfg.bridgeName ];
};
# Defence-in-depth: DROP bridge→loopback so compromised agents can't
# reach host-loopback services even via routing table leaks.
networking.firewall.extraInputRules = ''
ip saddr ${cfg.bridgeIp}/${toString cfg.bridgePrefixLength} ip daddr 127.0.0.0/8 drop
'';
# Allow isolated agents to reach the gateway (nginx on the host, shared
# netns). Port 80 covers `http://forge.<domain>`, per-agent UI proxies,
# and any other HTTP services the gateway fronts. Port 443 for HTTPS.
networking.firewall.interfaces.${cfg.bridgeName}.allowedTCPPorts = [
80
443
];
# Tells hive-c0re to pass PRIVATE_NETWORK + bridge settings to each
# container. HIVE_NETWORK_SUBNET is host-bridge IP/prefix, not canonical
# network address — the Rust side normalises before subnet arithmetic.
systemd.services.hive-c0re.environment = {
HIVE_NETWORK_ISOLATION = "1";
HIVE_NETWORK_BRIDGE = cfg.bridgeName;
HIVE_NETWORK_SUBNET = "${cfg.bridgeIp}/${toString cfg.bridgePrefixLength}";
};
})
];
}