hyperhive/nix/modules/hive-network.nix
atlas 39b4c65922 network: add isolateContainers option for #14 netns isolation
Adds `services.hyperhive.network.isolateContainers` (bool, default
false). When enabled alongside `network.enable`, activates:

- IP forwarding + NAT masquerade so isolated agents reach the internet
- nftables DROP rule blocking bridge-subnet → loopback (defence-in-depth
  against compromised agent reaching the c0re dashboard)
- `HIVE_NETWORK_ISOLATION`, `HIVE_NETWORK_BRIDGE`, `HIVE_NETWORK_SUBNET`
  injected into the hive-c0re service env; the Rust lifecycle reads these
  to set `PRIVATE_NETWORK`, `LOCAL_ADDRESS`, and `HOST_BRIDGE` in each
  agent container's conf

Config block rewritten as `lib.mkMerge [...]` — the prior `lib.mkIf //
lib.mkIf` pattern was invalid nix (mkIf returns a tagged value, not an
attrset; // on it is a type error). See docs/network.md for full design.
2026-06-03 11:19:29 +02:00

264 lines
11 KiB
Nix

{
lib,
config,
...
}:
let
cfg = config.services.hyperhive.network;
in
{
# Hive-internal network — host-side bridge + per-agent DNS resolver.
# Containers stay on shared host netns at v1; this module stands the
# bridge + resolver up so the endpoint is in place before network
# isolation flips containers to private netns. Full design: docs/network.md.
options.services.hyperhive.network = {
enable = lib.mkOption {
type = lib.types.bool;
default = false;
example = true;
description = ''
Stand up the hive-internal bridge + dnsmasq resolver.
Off by default while v1 phases in. When enabled:
a bridge interface (`bridgeName`) appears on the host with
`bridgeIp` assigned, and the hive-gateway container runs a
dnsmasq listening on that IP for `<hive-domain>` +
sub-domains. Agent containers still default to shared host
netns at v1 the endpoint is up but only used once #14
lands and flips containers to a private netns + veth peer
on this bridge.
'';
};
bridgeName = lib.mkOption {
type = lib.types.str;
default = "hive-br0";
example = "h0";
description = ''
Name of the host-side bridge interface the hive uses for
inter-container traffic. Kept short so it survives the
IFNAMSIZ (15-char) cap, and prefixed so it's obviously
hive-managed in `ip link` output.
'';
};
bridgeIp = lib.mkOption {
type = lib.types.str;
default = "10.42.0.1";
example = "172.30.0.1";
description = ''
IPv4 address assigned to the bridge interface on the host
side. Becomes the DNS server address agents point at (and
the upstream the gateway proxies to once netns isolation
lands). Default `10.42.0.1` is in RFC 1918 space and
unlikely to clash with operator's existing setup; override
if a different range is already in use.
'';
};
bridgePrefixLength = lib.mkOption {
type = lib.types.int;
default = 24;
example = 16;
description = ''
Netmask prefix length for the bridge subnet. Default `/24`
gives 254 usable per-agent addresses, enough for any
single-host hive. Operator with a larger swarm or a tighter
addressing scheme overrides.
'';
};
upstreamDns = lib.mkOption {
type = lib.types.listOf lib.types.str;
default = [
"1.1.1.1"
"9.9.9.9"
];
example = [
"192.168.1.1"
"8.8.8.8"
];
description = ''
Upstream DNS servers dnsmasq forwards non-hive queries to.
Defaults to Cloudflare + Quad9. Override for operators on
private networks who need a specific resolver (corporate
DNS, pi-hole, etc.). The hive resolver itself stays
authoritative for `<hive-domain>` and its sub-domains
regardless of upstream choice.
'';
};
isolateContainers = lib.mkOption {
type = lib.types.bool;
default = false;
example = true;
description = ''
Flip agent containers from shared host netns to private netns
(#14). When true, each agent container gets a dedicated veth
pair attached to `bridgeName` and a deterministic IP from
the bridge subnet. The bridge (already up when `enable = true`)
becomes the sole routed path between the host and agent
containers.
The host-side nix effect (this option) is:
- Sets `HIVE_NETWORK_ISOLATION=1` in the c0re service env so
the Rust lifecycle knows to pass `--private-network` +
bridge settings when creating/updating containers.
- Enables IP forwarding + NAT so agents can reach the internet
through the host.
- Adds a firewall rule DROP'ing traffic from the bridge subnet
to the host's loopback addresses defence-in-depth so a
compromised agent can't reach the c0re dashboard (already
bound to 127.0.0.1) or other host-loopback services even if
the routing table somehow leaks.
- Allows HTTP/HTTPS (80/443) traffic from the bridge subnet to
the host so agents can reach the gateway container (shared
host netns, proxies the operator's per-agent UI).
**Prerequisite**: all agents must have
`hyperhive.web.useUnixSocket = true` before enabling isolation.
Agents that still bind TCP on `0.0.0.0:<port>` will be
reachable at their bridge IP from other agents on the same
subnet defeating the isolation goal. The gateway routes via
unix sockets so gateway reach still works regardless.
**Migration**: containers are destroyed and re-created when
the network isolation flag flips. Operator state under
`/agents/<name>/state/` is bind-mounted and survives; the
container rootfs (nix store paths) is recreated cleanly.
**Rust counterpart**: `hive-c0re` reads `HIVE_NETWORK_ISOLATION`
and `HIVE_NETWORK_BRIDGE` from its service env and uses them in
`lifecycle::set_nspawn_flags` to configure `PRIVATE_NETWORK`,
`LOCAL_ADDRESS`, and `HOST_BRIDGE` in each container's
`nixos-containers/<name>.conf`. See `docs/network.md` for the
full design.
'';
};
};
config = lib.mkMerge [
(lib.mkIf cfg.enable {
assertions = [
{
assertion = config.services.hyperhive.domain != null;
message = ''
services.hyperhive.network.enable = true requires
services.hyperhive.domain to be set the resolver needs a
domain to be authoritative for. Either pin a hostname
(`services.hyperhive.domain = "example.com";`) or leave
`network.enable` at its default of false.
'';
}
{
assertion = config.services.hyperhive.gateway.enable;
message = ''
services.hyperhive.network.enable = true requires
services.hyperhive.gateway.enable = true the dnsmasq
resolver runs inside the hive-gateway container (single
front-door for both DNS and HTTP). Enable the gateway or
leave `network.enable` at its default of false.
'';
}
];
# Bridge interface on the host. Empty interfaces list = purely
# virtual bridge (no slave NICs attached). Per-agent veth pairs
# will join this bridge once #14 lands; at v1 it stands alone.
networking.bridges.${cfg.bridgeName}.interfaces = [ ];
# Host-side IP assignment on the bridge. This is what dnsmasq
# (inside the gateway container, shared host netns) binds on.
networking.interfaces.${cfg.bridgeName}.ipv4.addresses = [
{
address = cfg.bridgeIp;
prefixLength = cfg.bridgePrefixLength;
}
];
# Open the resolver port in the host firewall for traffic from
# the bridge subnet only. Other interfaces stay closed —
# external DNS-amplification surface is not exposed.
networking.firewall.interfaces.${cfg.bridgeName} = {
allowedUDPPorts = [ 53 ];
allowedTCPPorts = [ 53 ];
};
})
# Container network isolation (#14 v1). Ships as a separate overlay
# on top of the base bridge config (which stays unconditional) so
# operators can stand the bridge + resolver up first, validate
# everything, then flip isolation on independently.
(lib.mkIf (cfg.enable && cfg.isolateContainers) {
assertions = [
{
# Isolation without the bridge is a no-op: agents would get
# private netns but no reachable gateway. The assertion on
# `cfg.enable` above already gates the bridge, but making the
# dependency explicit here avoids confusing "bridge up, no
# isolation" vs "isolation on, no bridge" states.
assertion = cfg.enable;
message = ''
services.hyperhive.network.isolateContainers = true requires
services.hyperhive.network.enable = true (the bridge and
resolver must be running before isolation is flipped on).
'';
}
];
# IP forwarding — agents need to route through the bridge to reach
# the internet. NixOS firewall's `nat.enable` sets this too, but
# making it explicit here keeps the intent visible alongside the
# NAT rule.
boot.kernel.sysctl."net.ipv4.ip_forward" = 1;
# NAT/masquerade: translate agent bridge IPs → host's outbound
# IP for internet-bound traffic. Without masquerade, packets from
# 10.42.0.X arrive at external servers with an RFC-1918 source
# that can't be routed back.
networking.nat = {
enable = true;
# `internalInterfaces` causes `MASQUERADE` on packets from the
# bridge leaving via any external interface. Only traffic from
# agents crosses the bridge — hive-gateway / forge / matrix
# stay on host netns and don't need NAT.
internalInterfaces = [ cfg.bridgeName ];
};
# DROP traffic from the bridge subnet to host loopback addresses.
# Defence-in-depth: the c0re dashboard already binds 127.0.0.1
# (not 0.0.0.0), so bridge-sourced traffic can't reach it via
# the bridge IP. But a misconfigured service that slips to
# 0.0.0.0 would otherwise be reachable. The DROP rule closes that
# window. Use `extraInputRules` (nftables `input` chain, priority
# 0, same ruleset as `allowedTCPPorts`) so it's processed before
# the accept rules for bridge-side DNS we added above.
#
# The rule fires only when an agent container has a bridge IP
# (i.e. after the Rust side also ships `PRIVATE_NETWORK=1`);
# until then all containers share host netns and no traffic
# originates from 10.42.0.0/24 so this is a dead letter.
networking.firewall.extraInputRules = ''
ip saddr ${cfg.bridgeIp}/${toString cfg.bridgePrefixLength} ip daddr 127.0.0.0/8 drop
'';
# Signal to the c0re Rust side that container isolation is
# enabled. c0re reads `HIVE_NETWORK_ISOLATION` from its service
# environment and uses it in `lifecycle::set_nspawn_flags` to set
# `PRIVATE_NETWORK=1`, `LOCAL_ADDRESS=<hash-ip>`, and
# `HOST_BRIDGE=<bridgeName>` in each agent's container config.
# Also forwards the bridge name + subnet so c0re can wire the
# veth without hardcoding.
#
# `systemd.services.hive-c0re.environment` is an attrset; NixOS
# merges contributions from all modules that set it, so this
# cross-module injection is idiomatic and doesn't require a
# dedicated option in hive-c0re.nix.
systemd.services.hive-c0re.environment = {
HIVE_NETWORK_ISOLATION = "1";
HIVE_NETWORK_BRIDGE = cfg.bridgeName;
HIVE_NETWORK_SUBNET = "${cfg.bridgeIp}/${toString cfg.bridgePrefixLength}";
};
})
];
}