hyperhive/nix/host-modules/swarm-victoriametrics.nix
atlas d3b40da1c8 deploy: give every option an enable, and name the controller one
Two corrections from review, applied forward on this branch rather than
by rewriting it.

`deploy.<service>` was a bare bool, which makes
`deploy.forgejo = { enable; ci; }` unrepresentable -- the nested
CI-runner sub-option this namespace was designed around. Every entry is
now an attrset with an `enable`, so a second per-host deployment
decision becomes an ordinary addition rather than a migration.

`deploy.controller` is now `deploy.swarm-controller`, consistent with
`deploy.swarm-ui`, which was introduced in the same commit.

89 references rewritten across 24 files -- nix, Rust, docs, and the
repo's own CLAUDE.md.

The prefix-anchored sweep missed exactly one, and it was live code:
hive-tls.nix spells it `hyperhiveCfg.deploy.controller` -- the only
`hyperhiveCfg` prefix among 45 references. A suffix grep
(`\.deploy\.<name>`) finds it; a path-anchored one cannot, because the
head of a reference is whatever alias the reading file happens to bind.
2026-08-30 04:23:22 +02:00

192 lines
7.6 KiB
Nix

# The swarm's metrics store: one VictoriaMetrics for the whole swarm, in a
# `swarm-victoriametrics` nixos-container.
#
# A LOCAL time-series database rather than only an external sink, and that is
# the point rather than a convenience: the swarm dashboard has to stay
# readable when the outside world is unreachable. Same failure-domain property
# the hive-status KV has — a view of the system must not depend on the system
# it is viewing being healthy.
#
# It is the collector that feeds this (the gateway OTEL collector), not the
# agents directly: one ingest point per swarm, authenticated there.
{
pkgs,
lib,
config,
...
}:
let
cfg = config.services.hyperhive.swarm.victoriametrics;
deployCfg = config.services.hyperhive.deploy;
networkCfg = config.services.hyperhive.network;
hyperhiveCfg = config.services.hyperhive;
gatewayCfg = hyperhiveCfg.gateway;
swarmDomain = hyperhiveCfg.swarm.domain;
# Total on a null swarm domain for the same reason every sibling module is:
# the required-domain assertion in hive-network.nix should be what an
# operator sees, not a coercion error from here.
domainBase = if swarmDomain == null then "invalid" else swarmDomain;
in
{
# `enable` moved to `services.hyperhive.deploy.victoriametrics.enable` — see
# ./deploy.nix. What stays here is what the store IS: its package,
# domain, retention and wiring.
options.services.hyperhive.swarm.victoriametrics = {
package = lib.mkOption {
type = lib.types.package;
default = pkgs.victoriametrics;
defaultText = lib.literalExpression "pkgs.victoriametrics";
description = "VictoriaMetrics package to run.";
};
machine = lib.mkOption {
type = lib.types.str;
readOnly = true;
default = "swarm-victoriametrics";
description = ''
Container name. Read-only: the name appears in host paths and in
`machinectl`, so it is a fact other modules may read rather than a
knob.
'';
};
domain = lib.mkOption {
type = lib.types.str;
default = "metrics.${domainBase}";
defaultText = lib.literalExpression ''"metrics.''${services.hyperhive.swarm.domain}"'';
description = ''
Name the gateway serves this on. A sibling of the swarm's other
service names, so the swarm-services sub-CA can issue for it see
`hive-tls.nix` for why a service name being a sibling rather than a
child decides which CA may sign it.
'';
};
port = lib.mkOption {
type = lib.types.port;
default = 8428;
description = ''
Port VictoriaMetrics listens on, bound to loopback only (see
below). Upstream's own default, kept so an operator reading
VictoriaMetrics documentation finds what they expect.
'';
};
retentionPeriod = lib.mkOption {
type = lib.types.str;
default = "5y";
example = "90d";
description = ''
How long samples are kept.
Deliberately a high default rather than a required option: the two
failure directions are not symmetric. Too long fills a disk, which
is visible and recoverable by lowering this; too short **destroys
history**, silently and permanently. So the safe default is generous
and an operator lowers it once they have measured how fast this swarm
actually accumulates data.
'';
};
};
config = lib.mkIf (hyperhiveCfg.enable && deployCfg.victoriametrics.enable) {
# The gateway name and the quick-link, both inside `deployCfg.victoriametrics.enable` — that
# guard is the load-bearing part. Every hive in a swarm may know this
# store exists, but only the host that RUNS it may claim the name; a
# client hive declaring the vhost would answer for a service it does not
# have.
services.hyperhive.gateway.localNames = [ cfg.domain ];
# Declared here rather than in the collector, so this store's logs are
# collected because it runs, not because a list elsewhere remembered it.
services.hyperhive.swarm.otel.journaldUnits = [ "victoriametrics" ];
services.hyperhive.swarm.controller.links = [
{
label = "Metrics";
icon = "📈";
url = "https://${cfg.domain}/";
}
];
# This store publishes its own health as prometheus metrics on the same
# listener it serves queries on, so the swarm's collector can scrape it
# with no exporter and no extra port.
#
# Declared here rather than in the collector's module because that is the
# rule the option carries: an entry exists only where the service that
# named it runs, which is what keeps scraper and target on one host by
# construction instead of by luck.
#
# The loopback literal introduces no new assumption — it is the address
# this module already pins the listener to, and the same one the
# collector's `otlphttp/victoriametrics` exporter already writes to. If
# that reach is ever wrong, it is wrong for the write path first.
services.hyperhive.swarm.otel.scrapeTargets.victoriametrics = "127.0.0.1:${toString cfg.port}";
services.nginx.virtualHosts."${cfg.domain}" = (gatewayCfg.lib.tlsFor cfg.domain) // {
listen = gatewayCfg.lib.listen;
extraConfig = gatewayCfg.lib.securityHeaders;
locations."/" = {
proxyPass = "http://127.0.0.1:${toString cfg.port}/";
};
};
containers.${cfg.machine} = {
autoStart = true;
ephemeral = false;
# Shared host netns, like every sibling swarm container: the gateway
# reaches this at 127.0.0.1:<port>.
privateNetwork = false;
config =
{ ... }:
{
imports = [
(import ./swarm-container-resolver.nix {
inherit (networkCfg) bridgeIp;
dnsConsumers = [ "victoriametrics.service" ];
})
];
system.stateVersion = "26.05";
# This container shares the host netns, so its own firewall.service
# would rewrite the HOST ruleset at every boot. The host firewall
# owns all filtering.
networking.firewall.enable = false;
# resolvconf stays off because the resolver unit imported above
# owns /etc/resolv.conf. Leaving it on would let host-tracking
# regenerate the file empty, since the host's copy doesn't cross
# the boundary after start.
networking.resolvconf.enable = lib.mkForce false;
services.victoriametrics = {
enable = true;
package = cfg.package;
retentionPeriod = cfg.retentionPeriod;
# ⚠️ PINNED TO LOOPBACK, and this is a correction rather than a
# preference: upstream's default is `:8428`, i.e. every
# interface. The gateway is the only intended client and it is on
# this host, so binding wider would publish an unauthenticated
# write endpoint (see the OTLP note below) to whatever the host
# is reachable on.
listenAddress = "127.0.0.1:${toString cfg.port}";
};
};
};
};
# 🔑 OTLP ingest needs no flag. Measured against the pinned 1.146.0 rather
# than inferred from the module's option list, which has no OTLP switch and
# so reads as though the feature were absent: the running server answers
# `POST /opentelemetry/api/v1/push` with 200 (a nonexistent path answers
# 400, so that 200 means the route exists). The `-opentelemetry.*` flags
# only tune naming and limits, and are reachable via `extraOptions` if a
# deployment ever needs them.
#
# ⚠️ That endpoint is unauthenticated, which is why `listenAddress` above is
# loopback and why the collector — not agents — is the writer.
}