From e5c44f835acb20b7a70182f339a7c0fce4a85d9d Mon Sep 17 00:00:00 2001 From: atlas Date: Mon, 24 Aug 2026 14:35:41 +0200 Subject: [PATCH 1/2] feat(#3518): expose NATS broker metrics via prometheus-nats-exporter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit NATS has no Prometheus format of its own. It serves a JSON monitoring endpoint, and prometheus-nats-exporter translates that — so this is two changes in order, not one: without the monitoring endpoint the exporter starts cleanly and scrapes nothing, which is the inert-config shape the scrape work exists to avoid. Both listeners are loopback and the exporter is the monitoring endpoint's only intended reader: it is unauthenticated and /connz names every connected client, so the address it binds is the whole access control. The scrape target is declared here rather than in the collector's module, gated on a collector existing to read it — an entry exists only where the service that named it runs. --- nix/host-modules/swarm-nats.nix | 93 +++++++++++++++++++++++++++++++-- 1 file changed, 89 insertions(+), 4 deletions(-) diff --git a/nix/host-modules/swarm-nats.nix b/nix/host-modules/swarm-nats.nix index a29121e5..303ddadb 100644 --- a/nix/host-modules/swarm-nats.nix +++ b/nix/host-modules/swarm-nats.nix @@ -239,6 +239,45 @@ in ''; }; + monitorPort = lib.mkOption { + type = lib.types.port; + default = 8222; + description = '' + Port NATS serves its **monitoring** endpoint on, bound to + loopback. + + Not a metrics endpoint: the server has no Prometheus format of its + own. This serves `/varz`, `/connz`, `/routez` as JSON, and the + exporter below is what translates it — which is why enabling the + exporter without this produces a process that starts cleanly and + scrapes nothing. + + ⚠️ Loopback, and the exporter is the only intended reader. The + endpoint is unauthenticated and `/connz` names every connected + client, so the address it binds is the whole access control. Do + not widen it, and do not put it behind a gateway vhost expecting + that to add one. + ''; + }; + + metricsPort = lib.mkOption { + type = lib.types.port; + default = 7777; + description = '' + Port the Prometheus exporter serves NATS's metrics on, bound to + loopback for the swarm collector to scrape. + + ⚠️ This and {option}`monitorPort` are two more claims on a port + space every swarm container shares — they run in this container but + `privateNetwork = false`, so a collision with any other hyperhive + service is a runtime coin toss over which process gets the port, + with nothing in any log saying so. Both defaults are upstream's + own (`nats-server` 8222, `prometheus-nats-exporter` 7777) and + neither is claimed elsewhere in this repo, checked when they were + added. + ''; + }; + clientId = lib.mkOption { type = lib.types.str; default = "swarm-nats"; @@ -443,6 +482,26 @@ in } ]; + # Declared here rather than in the collector's module, which is the + # rule the option carries: an entry exists only where the service that + # named it runs, so the scraper and the target are on one host by + # construction rather than by luck. + # + # Gated on the collector, because a target nobody reads is a config + # asserting a collection that is not happening. This does NOT make the + # queue scrapeable from another host — that would need the endpoint + # published under a name with a cert and an audience, and nothing here + # should be: both listeners above are deliberately loopback. + # + # ⚠️ `swarm.otel`, not `hyperhive.otel` — two collectors one word + # apart, and only this one reads `scrapeTargets`. Written in full so + # the gate and the option it gates are visibly the same path; gating + # the wrong one is not a build error, it is a target that is always + # declared or never is. + services.hyperhive.swarm.otel.scrapeTargets = lib.mkIf config.services.hyperhive.swarm.otel.enable { + nats = "127.0.0.1:${toString cfg.metricsPort}"; + }; + containers.swarm-nats = { autoStart = true; ephemeral = false; @@ -509,10 +568,36 @@ in # includes other files. The check moves to server start. validateConfig = !cfg.autoGenerateCallout; - settings = calloutBlocks { - userKey = cfg.calloutUserPublicKey; - issuerKey = cfg.calloutIssuerPublicKey; - }; + settings = + calloutBlocks { + userKey = cfg.calloutUserPublicKey; + issuerKey = cfg.calloutIssuerPublicKey; + } + // { + # The monitoring endpoint, which is what the exporter below + # reads. `//` adds a key `calloutBlocks` does not produce + # (`accounts`, `authorization`) — checked, because a shallow + # merge that collided here would drop the auth config while + # rendering a config the server starts on. + # + # ⚠️ Loopback: unauthenticated, and `/connz` lists every + # connected client. + http = "127.0.0.1:${toString cfg.monitorPort}"; + }; + }; + + # The translation layer. NATS has no Prometheus format of its + # own, so this reads the JSON monitoring endpoint above and + # re-serves it in the format the collector scrapes. + # + # Upstream's exporter module rather than a hand-rolled unit, for + # the same reason `services.nats`'s own `settings` is kept: the + # reviewed reasoning about flags and hardening lives there. + services.prometheus.exporters.nats = { + enable = true; + listenAddress = "127.0.0.1"; + port = cfg.metricsPort; + url = "http://127.0.0.1:${toString cfg.monitorPort}"; }; # A wrapper that includes upstream's rendered settings verbatim From d2de3e8a2a8deaeedcf7c19542692fb2b9ea2d4e Mon Sep 17 00:00:00 2001 From: atlas Date: Mon, 24 Aug 2026 14:48:38 +0200 Subject: [PATCH 2/2] docs(swarm-otel): scrapeTargets constrains the target, not the scraper MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The option's description claimed declaring an entry from the service's own module put 'the scraper and the target on the same host by construction rather than by luck'. It does not. It constrains where the target is; nothing in it places the collector, and the two enable flags are co-located by a shared lib.mkDefault rather than by construction. Split across hosts, a target is silently never scraped — the service's host declares an entry no local collector reads, the collector's host never enabled the service. No error surfaces, and no assertion can catch it: separate hosts are separate evaluations with no shared context, so the doc telling the truth is the only mechanism there is. The same paragraph already warned co-location was not a guarantee, four lines below the sentence claiming it was; a reader arriving for permission stopped at the permission. This one did. swarm-nats carries the concrete caveat for its own contribution. --- nix/host-modules/swarm-nats.nix | 33 +++++++++++++++++++-------------- nix/host-modules/swarm-otel.nix | 26 ++++++++++++++++++++------ 2 files changed, 39 insertions(+), 20 deletions(-) diff --git a/nix/host-modules/swarm-nats.nix b/nix/host-modules/swarm-nats.nix index 303ddadb..671c2d8e 100644 --- a/nix/host-modules/swarm-nats.nix +++ b/nix/host-modules/swarm-nats.nix @@ -482,22 +482,27 @@ in } ]; - # Declared here rather than in the collector's module, which is the - # rule the option carries: an entry exists only where the service that - # named it runs, so the scraper and the target are on one host by - # construction rather than by luck. + # Declared here rather than in the collector's module: an entry then + # exists only where the service that named it runs. Gated on a + # collector, because a target nobody reads asserts a collection that is + # not happening. # - # Gated on the collector, because a target nobody reads is a config - # asserting a collection that is not happening. This does NOT make the - # queue scrapeable from another host — that would need the endpoint - # published under a name with a cert and an audience, and nothing here - # should be: both listeners above are deliberately loopback. + # ⚠️ KNOWN LIMITATION, and it is SILENT. That rule constrains the + # target, not the scraper — nothing places the collector on this host. + # `swarm-required-services.nix` derives `nats.enable` and `otel.enable` + # from one `lib.mkDefault`, so they are co-located by *default*, and an + # operator may split them. Split, NATS is never scraped: this host + # declares an entry no local collector reads, the collector's host never + # enabled this module. No error, no warning — a healthy exporter and an + # empty dashboard. # - # ⚠️ `swarm.otel`, not `hyperhive.otel` — two collectors one word - # apart, and only this one reads `scrapeTargets`. Written in full so - # the gate and the option it gates are visibly the same path; gating - # the wrong one is not a build error, it is a target that is always - # declared or never is. + # Not guardable: separate hosts are separate evaluations with no shared + # context, so this one cannot see what that one runs. Saying so is the + # only mechanism there is. + # + # ⚠️ `swarm.otel`, not `hyperhive.otel` — two collectors one word apart, + # and only this one reads `scrapeTargets`. Written in full so the gate + # and the option it gates are visibly the same path. services.hyperhive.swarm.otel.scrapeTargets = lib.mkIf config.services.hyperhive.swarm.otel.enable { nats = "127.0.0.1:${toString cfg.metricsPort}"; }; diff --git a/nix/host-modules/swarm-otel.nix b/nix/host-modules/swarm-otel.nix index 06e9ec9d..4c8cc955 100644 --- a/nix/host-modules/swarm-otel.nix +++ b/nix/host-modules/swarm-otel.nix @@ -243,12 +243,26 @@ in ` = ":"`. **A service declares its own entry, from its own module, under its - own `enable`.** That is what puts the scraper and the target on the - same host by construction rather than by luck: an entry exists only - where the service that named it runs. Do not assemble the list here. - Every swarm service being co-located is a property of the all-local - deployment, not a guarantee — and that is precisely the case where - the difference is invisible until a swarm splits across hosts. + own `enable`.** Do not assemble the list here: an entry then exists + only where the service that named it runs, so a target is never + declared on a host that does not serve it. + + ⚠️ **That constrains the TARGET, not the SCRAPER, and the difference + is a silent gap.** Nothing here puts the collector on the same host. + Services are co-located by *default* — `swarm-required-services.nix` + derives their `enable` flags from one `lib.mkDefault` — not by + construction, and an operator may split them. + + When they are split the target is simply never scraped: the + service's host declares an entry no local collector reads, and the + collector's host never enabled that service so has no entry at all. + No error, no eval failure, no warning. + + **No assertion can catch this.** Two hosts are separate NixOS + evaluations with no shared context, so neither can see what the + other runs. A service that would be seriously wrong to lose should + say so in its own contribution, because saying it is the only + mechanism available. Samples land in a swarm-level pipeline that stamps `swarm` and **never** `hive`: a swarm service does not belong to a hive, and