Same rule as matrix and grafana: which build a service runs is a decision of the host that runs it. Both stores already had a `deploy.<store>` option for retention, so the package joins something rather than opening a namespace. The prose in both modules claimed the package as part of "what the store IS from any hive's point of view" — a client hive needs the domain and the port to reach a store, never the build it runs. deploy.nix's own comment made the same claim about the pair and is corrected with them. Separately, and the reason this commit adds a fixture rather than a line: NEITHER STORE HAD AN OLD-PATH FIXTURE AT ALL. `swarm.victorialogs.` and `swarm.victoriametrics.` had zero hits in module-eval.nix, so the `enable` shims from the first slice and both `retentionPeriod` shims have been uncovered since they landed — the suite would have gone green with any of them deleted. That is precisely what the wireguard fixture's own comment warns about: a missing shim reads as a clean tree and breaks every existing operator config. `storesOldPath` therefore sets all six old paths, not just the two this commit moves. The case reads the package the CONTAINER renders rather than the option, so a shim that resolves but stops reaching the module fails too. Refs #3772.
245 lines
10 KiB
Nix
245 lines
10 KiB
Nix
# The swarm's metrics store: one VictoriaMetrics for the whole swarm, in a
|
|
# `swarm-victoriametrics` nixos-container.
|
|
#
|
|
# A LOCAL time-series database rather than only an external sink, and that is
|
|
# the point rather than a convenience: the swarm dashboard has to stay
|
|
# readable when the outside world is unreachable. Same failure-domain property
|
|
# the hive-status KV has — a view of the system must not depend on the system
|
|
# it is viewing being healthy.
|
|
#
|
|
# It is the collector that feeds this (the gateway OTEL collector), not the
|
|
# agents directly: one ingest point per swarm, authenticated there.
|
|
{
|
|
pkgs,
|
|
lib,
|
|
config,
|
|
...
|
|
}:
|
|
let
|
|
cfg = config.services.hyperhive.swarm.victoriametrics;
|
|
deployCfg = config.services.hyperhive.deploy;
|
|
networkCfg = config.services.hyperhive.network;
|
|
hyperhiveCfg = config.services.hyperhive;
|
|
gatewayCfg = hyperhiveCfg.gateway;
|
|
autheliaCfg = hyperhiveCfg.swarm.authelia;
|
|
swarmDomain = hyperhiveCfg.swarm.domain;
|
|
|
|
# Total on a null swarm domain for the same reason every sibling module is:
|
|
# the required-domain assertion in hive-network.nix should be what an
|
|
# operator sees, not a coercion error from here.
|
|
domainBase = if swarmDomain == null then "invalid" else swarmDomain;
|
|
in
|
|
{
|
|
# What stays here is what the store IS from any hive's point of view: the
|
|
# name it answers on and the port. `enable`, `package` and
|
|
# `retentionPeriod` are decisions of the host that runs it and live under
|
|
# `deploy.*`.
|
|
options.services.hyperhive.swarm.victoriametrics = {
|
|
machine = lib.mkOption {
|
|
type = lib.types.str;
|
|
readOnly = true;
|
|
default = "swarm-victoriametrics";
|
|
description = ''
|
|
Container name. Read-only: the name appears in host paths and in
|
|
`machinectl`, so it is a fact other modules may read rather than a
|
|
knob.
|
|
'';
|
|
};
|
|
|
|
domain = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "metrics.${domainBase}";
|
|
defaultText = lib.literalExpression ''"metrics.''${services.hyperhive.swarm.domain}"'';
|
|
description = ''
|
|
Name the gateway serves this on. A sibling of the swarm's other
|
|
service names, so the swarm-services sub-CA can issue for it — see
|
|
`hive-tls.nix` for why a service name being a sibling rather than a
|
|
child decides which CA may sign it.
|
|
'';
|
|
};
|
|
|
|
port = lib.mkOption {
|
|
type = lib.types.port;
|
|
default = 8428;
|
|
description = ''
|
|
Port VictoriaMetrics listens on, bound to loopback only (see
|
|
below). Upstream's own default, kept so an operator reading
|
|
VictoriaMetrics documentation finds what they expect.
|
|
'';
|
|
};
|
|
|
|
};
|
|
|
|
# Retention is a property of the store this host runs, not something the
|
|
# swarm has to agree on: it is read only where the container is defined,
|
|
# and a hive that is a *client* of the metrics store never consults it.
|
|
options.services.hyperhive.deploy.victoriametrics.package = lib.mkOption {
|
|
type = lib.types.package;
|
|
default = pkgs.victoriametrics;
|
|
defaultText = lib.literalExpression "pkgs.victoriametrics";
|
|
description = "VictoriaMetrics package to run.";
|
|
};
|
|
|
|
options.services.hyperhive.deploy.victoriametrics.retentionPeriod = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "5y";
|
|
example = "90d";
|
|
description = ''
|
|
How long samples are kept.
|
|
|
|
Deliberately a high default rather than a required option: the two
|
|
failure directions are not symmetric. Too long fills a disk, which
|
|
is visible and recoverable by lowering this; too short **destroys
|
|
history**, silently and permanently. So the safe default is generous
|
|
and an operator lowers it once they have measured how fast this swarm
|
|
actually accumulates data.
|
|
'';
|
|
};
|
|
|
|
config = lib.mkIf (hyperhiveCfg.enable && deployCfg.victoriametrics.enable) {
|
|
# The gateway name and the quick-link, both inside `deployCfg.victoriametrics.enable` — that
|
|
# guard is the load-bearing part. Every hive in a swarm may know this
|
|
# store exists, but only the host that RUNS it may claim the name; a
|
|
# client hive declaring the vhost would answer for a service it does not
|
|
# have.
|
|
services.hyperhive.gateway.localNames = [ cfg.domain ];
|
|
|
|
# Declared here rather than in the collector, so this store's logs are
|
|
# collected because it runs, not because a list elsewhere remembered it.
|
|
services.hyperhive.swarm.otel.journaldUnits = [ "victoriametrics" ];
|
|
|
|
services.hyperhive.swarm.controller.links = [
|
|
{
|
|
label = "Metrics";
|
|
icon = "📈";
|
|
url = "https://${cfg.domain}/";
|
|
}
|
|
];
|
|
|
|
# This store publishes its own health as prometheus metrics on the same
|
|
# listener it serves queries on, so the swarm's collector can scrape it
|
|
# with no exporter and no extra port.
|
|
#
|
|
# Declared here rather than in the collector's module because that is the
|
|
# rule the option carries: an entry exists only where the service that
|
|
# named it runs, which is what keeps scraper and target on one host by
|
|
# construction instead of by luck.
|
|
#
|
|
# The loopback literal is safe for this option and NOT for the write path,
|
|
# which is the distinction that matters now that the two differ. A scrape
|
|
# target is only ever read by a collector on this host, so loopback states
|
|
# a fact. The collector's push goes to the swarm name through the gateway,
|
|
# because the collector need not be here at all.
|
|
services.hyperhive.swarm.otel.scrapeTargets.victoriametrics = "127.0.0.1:${toString cfg.port}";
|
|
|
|
services.nginx.virtualHosts."${cfg.domain}" = (gatewayCfg.lib.tlsFor cfg.domain) // {
|
|
listen = gatewayCfg.lib.listen;
|
|
extraConfig = gatewayCfg.lib.securityHeaders;
|
|
locations = {
|
|
"/" = {
|
|
proxyPass = "http://127.0.0.1:${toString cfg.port}/";
|
|
};
|
|
|
|
# The swarm collector's ingest route, and the only authenticated thing
|
|
# on this vhost. `=` so it outranks the `/` prefix above, which would
|
|
# otherwise carry these writes with no check at all.
|
|
#
|
|
# ⚠️ No `error_page 401 =302` here, and its absence is the point: a
|
|
# redirect is right for a browser and wrong for a pusher, which would
|
|
# follow it and POST its batch at a login page that answers 200 —
|
|
# ingest reporting healthy while storing nothing.
|
|
"= /opentelemetry/api/v1/push" = {
|
|
proxyPass = "http://127.0.0.1:${toString cfg.port}/opentelemetry/api/v1/push";
|
|
extraConfig = ''
|
|
auth_request /__metrics_push_authz;
|
|
'';
|
|
};
|
|
|
|
# The subrequest. Same target and header set as the sibling log
|
|
# store's, which took them from `swarm-ui.nix` — `X-Original-URL` and
|
|
# `X-Original-Method` are what authelia's auth-request implementation
|
|
# reads, and the address it compares the token's audience against.
|
|
"= /__metrics_push_authz" = {
|
|
proxyPass = "https://${autheliaCfg.domain}/api/authz/auth-request";
|
|
# nixpkgs appends its OWN `Host $host` after extraConfig, which
|
|
# would override verifiedProxyTo's — see the comment on
|
|
# verifiedProxyTo in hive-gateway/vhost-lib.nix.
|
|
recommendedProxySettings = false;
|
|
extraConfig = ''
|
|
internal;
|
|
${gatewayCfg.lib.verifiedProxyTo autheliaCfg.domain}
|
|
proxy_pass_request_body off;
|
|
proxy_set_header Content-Length "";
|
|
proxy_set_header X-Original-Method $request_method;
|
|
proxy_set_header X-Original-URL $scheme://$http_host$request_uri;
|
|
proxy_set_header X-Forwarded-Proto $scheme;
|
|
proxy_set_header X-Forwarded-Host $http_host;
|
|
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
|
'';
|
|
};
|
|
};
|
|
};
|
|
|
|
containers.${cfg.machine} = {
|
|
autoStart = true;
|
|
ephemeral = false;
|
|
# Journal files on the host, not inside the container: nixpkgs hardcodes
|
|
# --link-journal=try-guest, and EXTRA_NSPAWN_FLAGS expands after it.
|
|
extraFlags = [ "--link-journal=host" ];
|
|
# Shared host netns, like every sibling swarm container: the gateway
|
|
# reaches this at 127.0.0.1:<port>.
|
|
privateNetwork = false;
|
|
|
|
config =
|
|
{ ... }:
|
|
{
|
|
imports = [
|
|
(import ./swarm-container-resolver.nix {
|
|
inherit (networkCfg) bridgeIp;
|
|
dnsConsumers = [ "victoriametrics.service" ];
|
|
})
|
|
];
|
|
|
|
system.stateVersion = "26.05";
|
|
|
|
# This container shares the host netns, so its own firewall.service
|
|
# would rewrite the HOST ruleset at every boot. The host firewall
|
|
# owns all filtering.
|
|
networking.firewall.enable = false;
|
|
# resolvconf stays off because the resolver unit imported above
|
|
# owns /etc/resolv.conf. Leaving it on would let host-tracking
|
|
# regenerate the file empty, since the host's copy doesn't cross
|
|
# the boundary after start.
|
|
networking.resolvconf.enable = lib.mkForce false;
|
|
|
|
services.victoriametrics = {
|
|
enable = true;
|
|
package = deployCfg.victoriametrics.package;
|
|
retentionPeriod = deployCfg.victoriametrics.retentionPeriod;
|
|
|
|
# ⚠️ PINNED TO LOOPBACK, and this is a correction rather than a
|
|
# preference: upstream's default is `:8428`, i.e. every
|
|
# interface. The gateway is the only intended client and it is on
|
|
# this host, so binding wider would publish an unauthenticated
|
|
# write endpoint (see the OTLP note below) to whatever the host
|
|
# is reachable on.
|
|
listenAddress = "127.0.0.1:${toString cfg.port}";
|
|
};
|
|
};
|
|
};
|
|
};
|
|
|
|
# 🔑 OTLP ingest needs no flag. Measured against the pinned 1.146.0 rather
|
|
# than inferred from the module's option list, which has no OTLP switch and
|
|
# so reads as though the feature were absent: the running server answers
|
|
# `POST /opentelemetry/api/v1/push` with 200 (a nonexistent path answers
|
|
# 400, so that 200 means the route exists). The `-opentelemetry.*` flags
|
|
# only tune naming and limits, and are reachable via `extraOptions` if a
|
|
# deployment ever needs them.
|
|
#
|
|
# ⚠️ That endpoint is unauthenticated, which is why `listenAddress` above is
|
|
# loopback: the only route to it from off-host is the vhost's ingest
|
|
# location, which authenticates. The collector is still the only writer, but
|
|
# it now arrives by the swarm name rather than over loopback, because it need
|
|
# not share a host with this store.
|
|
}
|