From e5c44f835acb20b7a70182f339a7c0fce4a85d9d Mon Sep 17 00:00:00 2001 From: atlas Date: Mon, 24 Aug 2026 14:35:41 +0200 Subject: [PATCH] feat(#3518): expose NATS broker metrics via prometheus-nats-exporter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit NATS has no Prometheus format of its own. It serves a JSON monitoring endpoint, and prometheus-nats-exporter translates that — so this is two changes in order, not one: without the monitoring endpoint the exporter starts cleanly and scrapes nothing, which is the inert-config shape the scrape work exists to avoid. Both listeners are loopback and the exporter is the monitoring endpoint's only intended reader: it is unauthenticated and /connz names every connected client, so the address it binds is the whole access control. The scrape target is declared here rather than in the collector's module, gated on a collector existing to read it — an entry exists only where the service that named it runs. --- nix/host-modules/swarm-nats.nix | 93 +++++++++++++++++++++++++++++++-- 1 file changed, 89 insertions(+), 4 deletions(-) diff --git a/nix/host-modules/swarm-nats.nix b/nix/host-modules/swarm-nats.nix index a29121e5..303ddadb 100644 --- a/nix/host-modules/swarm-nats.nix +++ b/nix/host-modules/swarm-nats.nix @@ -239,6 +239,45 @@ in ''; }; + monitorPort = lib.mkOption { + type = lib.types.port; + default = 8222; + description = '' + Port NATS serves its **monitoring** endpoint on, bound to + loopback. + + Not a metrics endpoint: the server has no Prometheus format of its + own. This serves `/varz`, `/connz`, `/routez` as JSON, and the + exporter below is what translates it — which is why enabling the + exporter without this produces a process that starts cleanly and + scrapes nothing. + + ⚠️ Loopback, and the exporter is the only intended reader. The + endpoint is unauthenticated and `/connz` names every connected + client, so the address it binds is the whole access control. Do + not widen it, and do not put it behind a gateway vhost expecting + that to add one. + ''; + }; + + metricsPort = lib.mkOption { + type = lib.types.port; + default = 7777; + description = '' + Port the Prometheus exporter serves NATS's metrics on, bound to + loopback for the swarm collector to scrape. + + ⚠️ This and {option}`monitorPort` are two more claims on a port + space every swarm container shares — they run in this container but + `privateNetwork = false`, so a collision with any other hyperhive + service is a runtime coin toss over which process gets the port, + with nothing in any log saying so. Both defaults are upstream's + own (`nats-server` 8222, `prometheus-nats-exporter` 7777) and + neither is claimed elsewhere in this repo, checked when they were + added. + ''; + }; + clientId = lib.mkOption { type = lib.types.str; default = "swarm-nats"; @@ -443,6 +482,26 @@ in } ]; + # Declared here rather than in the collector's module, which is the + # rule the option carries: an entry exists only where the service that + # named it runs, so the scraper and the target are on one host by + # construction rather than by luck. + # + # Gated on the collector, because a target nobody reads is a config + # asserting a collection that is not happening. This does NOT make the + # queue scrapeable from another host — that would need the endpoint + # published under a name with a cert and an audience, and nothing here + # should be: both listeners above are deliberately loopback. + # + # ⚠️ `swarm.otel`, not `hyperhive.otel` — two collectors one word + # apart, and only this one reads `scrapeTargets`. Written in full so + # the gate and the option it gates are visibly the same path; gating + # the wrong one is not a build error, it is a target that is always + # declared or never is. + services.hyperhive.swarm.otel.scrapeTargets = lib.mkIf config.services.hyperhive.swarm.otel.enable { + nats = "127.0.0.1:${toString cfg.metricsPort}"; + }; + containers.swarm-nats = { autoStart = true; ephemeral = false; @@ -509,10 +568,36 @@ in # includes other files. The check moves to server start. validateConfig = !cfg.autoGenerateCallout; - settings = calloutBlocks { - userKey = cfg.calloutUserPublicKey; - issuerKey = cfg.calloutIssuerPublicKey; - }; + settings = + calloutBlocks { + userKey = cfg.calloutUserPublicKey; + issuerKey = cfg.calloutIssuerPublicKey; + } + // { + # The monitoring endpoint, which is what the exporter below + # reads. `//` adds a key `calloutBlocks` does not produce + # (`accounts`, `authorization`) — checked, because a shallow + # merge that collided here would drop the auth config while + # rendering a config the server starts on. + # + # ⚠️ Loopback: unauthenticated, and `/connz` lists every + # connected client. + http = "127.0.0.1:${toString cfg.monitorPort}"; + }; + }; + + # The translation layer. NATS has no Prometheus format of its + # own, so this reads the JSON monitoring endpoint above and + # re-serves it in the format the collector scrapes. + # + # Upstream's exporter module rather than a hand-rolled unit, for + # the same reason `services.nats`'s own `settings` is kept: the + # reviewed reasoning about flags and hardening lives there. + services.prometheus.exporters.nats = { + enable = true; + listenAddress = "127.0.0.1"; + port = cfg.metricsPort; + url = "http://127.0.0.1:${toString cfg.monitorPort}"; }; # A wrapper that includes upstream's rendered settings verbatim