From 3a0a7346b13586cc059dbc102272e7d378aa4dc8 Mon Sep 17 00:00:00 2001 From: atlas Date: Thu, 1 Oct 2026 09:52:49 +0200 Subject: [PATCH] nix: split swarm-otel into service and deploy-mode files `swarm.otel` (what the swarm collector is to every hive: its domain, receiver ports, producer names, client id and the audiences that client may present) moves to nix/host-modules/swarm-otel-service.nix, together with the `journaldUnits` removal module and the helpers its defaults read. Everything else -- the `deploy.swarm-otel` options, the whole `config` block including `containers.swarm-otel`, and the helpers only they read -- stays in nix/host-modules/swarm-otel.nix, which default.nix now imports after the new file. The service file needs `domainBase` (for `domain`), `baoCfg` (for `storeProducerName`) and `cfg` plus `pushAudiences` (for `audience`). `swarmDomain` and `domainBase` move. `cfg`, `hyperhiveCfg`, `baoCfg`, `vmCfg`, `vlCfg`, `metricsPushUrl`, `logsPushUrl` and `pushAudiences` are read on both sides and none is an option, so each is bound in both files; the comments on the push URLs say the two copies must agree. `swarmCfg` was bound and never read, so neither file carries it. A pure move: option paths, option definitions and config are unchanged. Comments changed: the transition comment above `deploy.swarm-otel` now names the file `swarm.otel` lives in, and the push-URL and `pushAudiences` comments now describe the per-file readers and the duplicate. Three otel fixtures evaluate to the same host and container toplevel derivations, `swarm.otel.*` and `deploy.swarm-otel.*` values before and after. Refs #3742 --- nix/host-modules/default.nix | 1 + nix/host-modules/swarm-otel-service.nix | 335 ++++++++++++++++++++++++ nix/host-modules/swarm-otel.nix | 325 +---------------------- 3 files changed, 345 insertions(+), 316 deletions(-) create mode 100644 nix/host-modules/swarm-otel-service.nix diff --git a/nix/host-modules/default.nix b/nix/host-modules/default.nix index e469350d..0f6e8522 100644 --- a/nix/host-modules/default.nix +++ b/nix/host-modules/default.nix @@ -52,6 +52,7 @@ ./swarm-controller.nix ./swarm-grafana-service.nix ./swarm-grafana.nix + ./swarm-otel-service.nix ./swarm-otel.nix ./swarm-snapshot-store.nix ./swarm-ui.nix diff --git a/nix/host-modules/swarm-otel-service.nix b/nix/host-modules/swarm-otel-service.nix new file mode 100644 index 00000000..d7e6bfa3 --- /dev/null +++ b/nix/host-modules/swarm-otel-service.nix @@ -0,0 +1,335 @@ +# The swarm's telemetry collector as every hive sees it: the name it is +# reached on, its receiver ports, the client it is registered as and the +# audiences that client may present, identical on every host. What the host +# running it decides, and the container itself, are in ./swarm-otel.nix. +{ + lib, + config, + ... +}: +let + cfg = config.services.hyperhive.swarm.otel; + vmCfg = config.services.hyperhive.swarm.victoriametrics; + vlCfg = config.services.hyperhive.swarm.victorialogs; + hyperhiveCfg = config.services.hyperhive; + baoCfg = hyperhiveCfg.swarm.bao; + swarmDomain = hyperhiveCfg.swarm.domain; + + # Total on a null swarm domain for the same reason every sibling module is: + # the required-domain assertion in hive-network.nix should be what an + # operator sees, not a coercion error from here. + domainBase = if swarmDomain == null then "invalid" else swarmDomain; + + # Each store's OTLP route, by domain: the audiences `audience` below registers + # the client for. ⚠️ ./swarm-otel.nix binds the same strings for the + # exporters that request those tokens; the two copies must agree. + metricsPushUrl = "https://${vmCfg.domain}/opentelemetry/api/v1/push"; + logsPushUrl = "https://${vlCfg.domain}/insert/opentelemetry/v1/logs"; + + pushAudiences = { + victoriametrics = metricsPushUrl; + victorialogs = logsPushUrl; + }; +in +{ + imports = [ + (lib.mkRemovedOptionModule [ "services" "hyperhive" "swarm" "otel" "journaldUnits" ] '' + The swarm collector no longer takes a unit allowlist: it ships the + journal of every system unit on the host and drops only user + sessions (records under a `user-.slice`). Delete the setting; + nothing replaces it. + '') + ]; + + # `enable` moved to `services.hyperhive.deploy.swarm-otel.enable` — see ./deploy.nix. + # ⚠️ That is the SWARM collector. The per-hive one keeps its own + # `services.hyperhive.otel.enable` (./otel.nix) and is a different + # option entirely — every hive runs that one. + options.services.hyperhive.swarm.otel = { + + machine = lib.mkOption { + type = lib.types.str; + readOnly = true; + default = "swarm-otel"; + description = '' + Name of the nixos-container this collector runs in — also the + `machinectl` name, so other modules may read it rather than + repeating the literal. + ''; + }; + + port = lib.mkOption { + type = lib.types.port; + default = 4319; + description = '' + First port of this collector's receiver range. Every hive in + {option}`services.hyperhive.swarm.hives` gets its **own** + authenticated receiver — that is what makes the `hive` label + unforgeable — so the range is one port per hive, starting here, in + sorted-name order. + + ⚠️ Internal. No client is ever told a port: a hive reaches its own + receiver as `https://''${domain}/`, and the gateway routes on + that path. So adding a hive, which renumbers the ones after it, is + harmless — nginx is rendered from this same evaluation and moves + with it. + + ⚠️ **Deliberately not 4318**, the OTLP/HTTP default, because the + hive tier already uses it (`services.hyperhive.otel.collector.port`) + and every swarm container shares the host's network namespace. Two + listeners claiming one port on one host is not a build failure — + it is a runtime coin toss over which one gets it, with nothing in + any log saying so. The same collision cost a release when grafana + and the forge both defaulted to 3000. The assertions below check + the whole derived range against every port this module and the hive + tier declare, which is as far as a module can see. + ''; + }; + + telemetryPort = lib.mkOption { + type = lib.types.port; + default = 8889; + description = '' + Port this collector serves its **own** metrics on — queue depth, + refused and dropped samples, exporter failures. How you find out + that telemetry is being lost, so it is worth keeping rather than + switching off. + + ⚠️ **Deliberately not 8889's neighbour 8888**, which is the + collector's built-in default and therefore what the hive tier + already binds. Two collectors share a network namespace whenever + they are co-located, and unlike the OTLP port this one appears + nowhere in either config — it is a default inside the binary, so + nothing that compares configured ports can see the clash. The + second collector to start simply dies with + `bind: address already in use`. + ''; + }; + + producerPort = lib.mkOption { + type = lib.types.port; + default = 4390; + description = '' + Port the swarm tier's own OTLP/HTTP receiver for a + **swarm-level** producer (`swarm-controller`'s vcs/jobq counters + today) listens on, at `127.0.0.1`. Reached through the gateway at + `https://''${domain}/''${producerName}/`, same shape as a hive's + own receiver — a swarm-level producer may not share a host with + this collector (mara, on the swarm-controller topology: "the + swarm services dont have to run on the same host as the swarm + controller"), so loopback-only reachability is not a supported + shape here any more than it is for a hive. + + Authenticated by `oidc/''${producerName}`, checking for the + audience `services.hyperhive.swarm.controller.queueClientId` + requests — not the per-hive `hiveClientPrefix` namespace, because + a swarm-level producer is explicitly not a hive (see that + option's own doc comment). The `swarm` label this receiver's + samples get is still the one the receiver's own *existence* + establishes, same as before — there is no hive identity being + forged or attributed here, just a caller proving it is the one + principal allowed to push into this pipeline. + + ⚠️ Deliberately NOT derived from `port + (number of hives)`: that + range grows every time a hive is added, and a fixed offset from + it would silently start colliding once the hive count caught up. + Kept as its own reserved value instead, and the assertion below + still catches a real collision (including one hive growth + eventually causes) rather than starting a collector that quietly + drops one receiver's samples. + ''; + }; + + storeProducerPort = lib.mkOption { + type = lib.types.port; + default = 4391; + description = '' + Port the receiver for the **secret store's own forwarder** listens + on, at `127.0.0.1`. Reached through the gateway at + `https://''${domain}/''${storeProducerName}/`, the same + path-per-producer shape a hive and the swarm-tier producer above + both use. + + Its own receiver, and not a second sender into + {option}`services.hyperhive.swarm.otel.producerPort`, because an + `oidc` authenticator checks exactly ONE audience: that receiver + admits the audience `swarm-controller` registers for itself, and + the store's forwarder is a different principal with a client of its + own. Sharing the route would mean sharing that client — which is + the thing the store exists to make unnecessary. + + ⚠️ A reserved value rather than an offset from + {option}`services.hyperhive.swarm.otel.port`, for the reason stated + on `producerPort`: that range grows with the hive count and would + eventually walk into any fixed offset from it. The collision + assertion below covers this port too. + ''; + }; + + storeProducerName = lib.mkOption { + type = lib.types.str; + readOnly = true; + default = baoCfg.machine; + defaultText = lib.literalExpression "services.hyperhive.swarm.bao.machine"; + description = '' + Path and component name of the store forwarder's own receiver + (`otlp/''${storeProducerName}`, `oidc/''${storeProducerName}`). + Published so `swarm-bao.nix` addresses the receiver by name rather + than repeating the string, exactly as `producerName` above is + published for `swarm-controller.nix`. + + Read-only and derived from the store's container name: two + spellings of a name both ends have to agree on is a 404 on a + request that authenticated perfectly. + ''; + }; + + producerName = lib.mkOption { + type = lib.types.str; + readOnly = true; + default = "swarm"; + description = '' + The swarm-tier producer's own component/path/authenticator name + — the same reserved literal this module's collector components + (`otlp/swarm`, `oidc/swarm`, `resource/swarm`, `metrics/swarm`) + are already named after internally. Published as an option so a + sibling module (`swarm-controller.nix`) can address the receiver + by name — `https://''${domain}/''${producerName}/` — instead of + repeating the string. Read-only for the same reason + `clientId`/`machine`/`domain` are: two spellings of a name that + has to match on both ends is a mismatch waiting to happen. + ''; + }; + + domain = lib.mkOption { + type = lib.types.str; + default = "otel.${domainBase}"; + defaultText = lib.literalExpression ''"otel.''${services.hyperhive.swarm.domain}"''; + description = '' + Name the gateway serves this on. A sibling of the swarm's other + service names, so the swarm-services sub-CA can issue for it — see + `hive-tls.nix` for why a service name being a sibling rather than a + child decides which CA may sign it. + + This is what the **hive** tier's exporter reaches — the hive + collector is a plain producer against this name exactly like every + other client of a swarm service, resolved locally by dnsmasq on a + co-located host and over the real network otherwise. There is no + separate loopback-vs-remote knob to get wrong: everything in this + swarm addresses its siblings by name. + ''; + }; + + scrapeTargets = lib.mkOption { + type = lib.types.attrsOf lib.types.str; + default = { }; + example = lib.literalExpression ''{ forgejo = "127.0.0.1:3000"; }''; + description = '' + Prometheus exposition endpoints this collector scrapes, as + ` = ":"`. + + A path and query may follow the port — + `"127.0.0.1:8202/v1/sys/metrics?format=prometheus"` — for an + exporter that does not serve `/metrics`. Both are optional and + omitted when absent, so a bare `host:port` is scraped exactly as + before. + + **A service declares its own entry, from its own module, under its + own `enable`.** Do not assemble the list here: an entry then exists + only where the service that named it runs, so a target is never + declared on a host that does not serve it. + + ⚠️ **That constrains the TARGET, not the SCRAPER, and the difference + is a silent gap.** Nothing here puts the collector on the same host. + Services are co-located by *default* — `swarm-required-services.nix` + derives their `enable` flags from one `lib.mkDefault` — not by + construction, and an operator may split them. + + When they are split the target is simply never scraped: the + service's host declares an entry no local collector reads, and the + collector's host never enabled that service so has no entry at all. + No error, no eval failure, no warning. + + **No assertion can catch this.** Two hosts are separate NixOS + evaluations with no shared context, so neither can see what the + other runs. A service that would be seriously wrong to lose should + say so in its own contribution, because saying it is the only + mechanism available. + + Samples land in a swarm-level pipeline that stamps `swarm` and + **never** `hive`: a swarm service does not belong to a hive, and + `hive` stays a property of which authenticated receiver accepted a + push, not something a scrape can acquire. + + Empty by default, in which case no scrape receiver, processor or + pipeline is emitted at all — an enabled scraper with nothing to + scrape is the inert configuration this option exists to avoid. + ''; + }; + + publishedScrapeTargets = lib.mkOption { + type = lib.types.attrsOf lib.types.str; + default = { }; + example = lib.literalExpression ''{ forge = "https://forge.example.com/metrics"; }''; + description = '' + Prometheus endpoints this collector scrapes **by name, with a + credential**, as ` = ""`. + + A service module declares its own entry, from its own module, the + same way it does for `scrapeTargets` — and the collector's OAuth2 + client derives its permitted audiences from these URLs, so a + target and the authorisation to reach it are **one declaration**. + Two lists that must agree would be a drift to maintain, and its + failure mode is the bad one: a target whose audience was forgotten + authenticates against nothing and reads as a scrape failure rather + than a config mistake. + + ⚠️ A **full URL**, not `host:port`. Authelia validates a bearer + token against the address being requested, so the string here is + also the audience the token is minted for; a near miss (a trailing + slash, `http` for `https`) presents as a valid token rejected at + the target, several layers from its cause. + + ⚠️ Deliberately separate from `scrapeTargets`. An entry there is + trusted because the scraper and the target share a host — that + option is loopback-and-unauthenticated by contract. An entry here + is trusted because it presents a credential. One shape for both + would leave a reader unable to tell which of those a given target + relies on. + ''; + }; + + clientId = lib.mkOption { + type = lib.types.str; + readOnly = true; + default = "swarm-collector"; + description = '' + OAuth2 client id this collector authenticates as. Published so + authelia's `access_control` rules can name it without carrying a + second copy of the string, exactly as + `services.hyperhive.swarm.authelia.hiveClientPrefix` is published + for the queue's responder. + + Two spellings drifting apart is not a build failure: authelia + refuses a rule naming an unregistered client in its startup + validator, so the swarm's SSO service fails to *restart* — long + after the change that caused it evaluated cleanly. + ''; + }; + + audience = lib.mkOption { + type = lib.types.listOf lib.types.str; + readOnly = true; + default = lib.attrValues cfg.publishedScrapeTargets ++ lib.attrValues pushAudiences; + description = '' + Every audience this collector's OAuth2 client is permitted to + present a token for — the scrape targets it reads with a credential, + plus the stores it pushes to. Published read-only so + ./glue-swarm-otel-oidc-client.nix can register the client wherever + authelia runs without restating the derivation: this option and + that registration are the same fact seen from two hosts, and a + second formula for it would be free to drift from this one. + ''; + }; + }; +} diff --git a/nix/host-modules/swarm-otel.nix b/nix/host-modules/swarm-otel.nix index 366e4d3b..c0d6ffad 100644 --- a/nix/host-modules/swarm-otel.nix +++ b/nix/host-modules/swarm-otel.nix @@ -17,7 +17,6 @@ let cfg = config.services.hyperhive.swarm.otel; deployCfg = config.services.hyperhive.deploy; - swarmCfg = config.services.hyperhive.swarm; otelCfg = config.services.hyperhive.otel; vmCfg = config.services.hyperhive.swarm.victoriametrics; vlCfg = config.services.hyperhive.swarm.victorialogs; @@ -25,12 +24,6 @@ let gatewayCfg = hyperhiveCfg.gateway; baoCfg = hyperhiveCfg.swarm.bao; baoDeploy = deployCfg.bao; - swarmDomain = hyperhiveCfg.swarm.domain; - - # Total on a null swarm domain for the same reason every sibling module is: - # the required-domain assertion in hive-network.nix should be what an - # operator sees, not a coercion error from here. - domainBase = if swarmDomain == null then "invalid" else swarmDomain; # The collector names its components `/`, where `` is a # hive name for the per-hive pipelines and this literal for the swarm tier's @@ -225,13 +218,14 @@ let # Each store's OTLP route, by domain. One binding because the same string is # both the address requested and the audience the token is minted for — two # spellings present as a valid token refused at the store. + # ⚠️ ./swarm-otel-service.nix binds both again for `swarm.otel.audience`, + # the audiences the client is registered for; the two copies must agree. metricsPushUrl = "https://${vmCfg.domain}/opentelemetry/api/v1/push"; logsPushUrl = "https://${vlCfg.domain}/insert/opentelemetry/v1/logs"; - # Read by the extensions block, `service.extensions`, each exporter's - # authenticator and the OIDC client's audience list. Derived rather than - # repeated: an authenticator naming an unlisted extension starts clean and - # authenticates nothing. + # Read by the extensions block, `service.extensions` and each exporter's + # authenticator. Derived rather than repeated: an authenticator naming an + # unlisted extension starts clean and authenticates nothing. pushAudiences = { victoriametrics = metricsPushUrl; victorialogs = logsPushUrl; @@ -314,311 +308,10 @@ let storeRetry = import ./lib/store-retry.nix { }; in { - imports = [ - (lib.mkRemovedOptionModule [ "services" "hyperhive" "swarm" "otel" "journaldUnits" ] '' - The swarm collector no longer takes a unit allowlist: it ships the - journal of every system unit on the host and drops only user - sessions (records under a `user-.slice`). Delete the setting; - nothing replaces it. - '') - ]; - - # `enable` moved to `services.hyperhive.deploy.swarm-otel.enable` — see ./deploy.nix. - # ⚠️ That is the SWARM collector. The per-hive one keeps its own - # `services.hyperhive.otel.enable` (./otel.nix) and is a different - # option entirely — every hive runs that one. - options.services.hyperhive.swarm.otel = { - - machine = lib.mkOption { - type = lib.types.str; - readOnly = true; - default = "swarm-otel"; - description = '' - Name of the nixos-container this collector runs in — also the - `machinectl` name, so other modules may read it rather than - repeating the literal. - ''; - }; - - port = lib.mkOption { - type = lib.types.port; - default = 4319; - description = '' - First port of this collector's receiver range. Every hive in - {option}`services.hyperhive.swarm.hives` gets its **own** - authenticated receiver — that is what makes the `hive` label - unforgeable — so the range is one port per hive, starting here, in - sorted-name order. - - ⚠️ Internal. No client is ever told a port: a hive reaches its own - receiver as `https://''${domain}/`, and the gateway routes on - that path. So adding a hive, which renumbers the ones after it, is - harmless — nginx is rendered from this same evaluation and moves - with it. - - ⚠️ **Deliberately not 4318**, the OTLP/HTTP default, because the - hive tier already uses it (`services.hyperhive.otel.collector.port`) - and every swarm container shares the host's network namespace. Two - listeners claiming one port on one host is not a build failure — - it is a runtime coin toss over which one gets it, with nothing in - any log saying so. The same collision cost a release when grafana - and the forge both defaulted to 3000. The assertions below check - the whole derived range against every port this module and the hive - tier declare, which is as far as a module can see. - ''; - }; - - telemetryPort = lib.mkOption { - type = lib.types.port; - default = 8889; - description = '' - Port this collector serves its **own** metrics on — queue depth, - refused and dropped samples, exporter failures. How you find out - that telemetry is being lost, so it is worth keeping rather than - switching off. - - ⚠️ **Deliberately not 8889's neighbour 8888**, which is the - collector's built-in default and therefore what the hive tier - already binds. Two collectors share a network namespace whenever - they are co-located, and unlike the OTLP port this one appears - nowhere in either config — it is a default inside the binary, so - nothing that compares configured ports can see the clash. The - second collector to start simply dies with - `bind: address already in use`. - ''; - }; - - producerPort = lib.mkOption { - type = lib.types.port; - default = 4390; - description = '' - Port the swarm tier's own OTLP/HTTP receiver for a - **swarm-level** producer (`swarm-controller`'s vcs/jobq counters - today) listens on, at `127.0.0.1`. Reached through the gateway at - `https://''${domain}/''${producerName}/`, same shape as a hive's - own receiver — a swarm-level producer may not share a host with - this collector (mara, on the swarm-controller topology: "the - swarm services dont have to run on the same host as the swarm - controller"), so loopback-only reachability is not a supported - shape here any more than it is for a hive. - - Authenticated by `oidc/''${producerName}`, checking for the - audience `services.hyperhive.swarm.controller.queueClientId` - requests — not the per-hive `hiveClientPrefix` namespace, because - a swarm-level producer is explicitly not a hive (see that - option's own doc comment). The `swarm` label this receiver's - samples get is still the one the receiver's own *existence* - establishes, same as before — there is no hive identity being - forged or attributed here, just a caller proving it is the one - principal allowed to push into this pipeline. - - ⚠️ Deliberately NOT derived from `port + (number of hives)`: that - range grows every time a hive is added, and a fixed offset from - it would silently start colliding once the hive count caught up. - Kept as its own reserved value instead, and the assertion below - still catches a real collision (including one hive growth - eventually causes) rather than starting a collector that quietly - drops one receiver's samples. - ''; - }; - - storeProducerPort = lib.mkOption { - type = lib.types.port; - default = 4391; - description = '' - Port the receiver for the **secret store's own forwarder** listens - on, at `127.0.0.1`. Reached through the gateway at - `https://''${domain}/''${storeProducerName}/`, the same - path-per-producer shape a hive and the swarm-tier producer above - both use. - - Its own receiver, and not a second sender into - {option}`services.hyperhive.swarm.otel.producerPort`, because an - `oidc` authenticator checks exactly ONE audience: that receiver - admits the audience `swarm-controller` registers for itself, and - the store's forwarder is a different principal with a client of its - own. Sharing the route would mean sharing that client — which is - the thing the store exists to make unnecessary. - - ⚠️ A reserved value rather than an offset from - {option}`services.hyperhive.swarm.otel.port`, for the reason stated - on `producerPort`: that range grows with the hive count and would - eventually walk into any fixed offset from it. The collision - assertion below covers this port too. - ''; - }; - - storeProducerName = lib.mkOption { - type = lib.types.str; - readOnly = true; - default = baoCfg.machine; - defaultText = lib.literalExpression "services.hyperhive.swarm.bao.machine"; - description = '' - Path and component name of the store forwarder's own receiver - (`otlp/''${storeProducerName}`, `oidc/''${storeProducerName}`). - Published so `swarm-bao.nix` addresses the receiver by name rather - than repeating the string, exactly as `producerName` above is - published for `swarm-controller.nix`. - - Read-only and derived from the store's container name: two - spellings of a name both ends have to agree on is a 404 on a - request that authenticated perfectly. - ''; - }; - - producerName = lib.mkOption { - type = lib.types.str; - readOnly = true; - default = "swarm"; - description = '' - The swarm-tier producer's own component/path/authenticator name - — the same reserved literal this module's collector components - (`otlp/swarm`, `oidc/swarm`, `resource/swarm`, `metrics/swarm`) - are already named after internally. Published as an option so a - sibling module (`swarm-controller.nix`) can address the receiver - by name — `https://''${domain}/''${producerName}/` — instead of - repeating the string. Read-only for the same reason - `clientId`/`machine`/`domain` are: two spellings of a name that - has to match on both ends is a mismatch waiting to happen. - ''; - }; - - domain = lib.mkOption { - type = lib.types.str; - default = "otel.${domainBase}"; - defaultText = lib.literalExpression ''"otel.''${services.hyperhive.swarm.domain}"''; - description = '' - Name the gateway serves this on. A sibling of the swarm's other - service names, so the swarm-services sub-CA can issue for it — see - `hive-tls.nix` for why a service name being a sibling rather than a - child decides which CA may sign it. - - This is what the **hive** tier's exporter reaches — the hive - collector is a plain producer against this name exactly like every - other client of a swarm service, resolved locally by dnsmasq on a - co-located host and over the real network otherwise. There is no - separate loopback-vs-remote knob to get wrong: everything in this - swarm addresses its siblings by name. - ''; - }; - - scrapeTargets = lib.mkOption { - type = lib.types.attrsOf lib.types.str; - default = { }; - example = lib.literalExpression ''{ forgejo = "127.0.0.1:3000"; }''; - description = '' - Prometheus exposition endpoints this collector scrapes, as - ` = ":"`. - - A path and query may follow the port — - `"127.0.0.1:8202/v1/sys/metrics?format=prometheus"` — for an - exporter that does not serve `/metrics`. Both are optional and - omitted when absent, so a bare `host:port` is scraped exactly as - before. - - **A service declares its own entry, from its own module, under its - own `enable`.** Do not assemble the list here: an entry then exists - only where the service that named it runs, so a target is never - declared on a host that does not serve it. - - ⚠️ **That constrains the TARGET, not the SCRAPER, and the difference - is a silent gap.** Nothing here puts the collector on the same host. - Services are co-located by *default* — `swarm-required-services.nix` - derives their `enable` flags from one `lib.mkDefault` — not by - construction, and an operator may split them. - - When they are split the target is simply never scraped: the - service's host declares an entry no local collector reads, and the - collector's host never enabled that service so has no entry at all. - No error, no eval failure, no warning. - - **No assertion can catch this.** Two hosts are separate NixOS - evaluations with no shared context, so neither can see what the - other runs. A service that would be seriously wrong to lose should - say so in its own contribution, because saying it is the only - mechanism available. - - Samples land in a swarm-level pipeline that stamps `swarm` and - **never** `hive`: a swarm service does not belong to a hive, and - `hive` stays a property of which authenticated receiver accepted a - push, not something a scrape can acquire. - - Empty by default, in which case no scrape receiver, processor or - pipeline is emitted at all — an enabled scraper with nothing to - scrape is the inert configuration this option exists to avoid. - ''; - }; - - publishedScrapeTargets = lib.mkOption { - type = lib.types.attrsOf lib.types.str; - default = { }; - example = lib.literalExpression ''{ forge = "https://forge.example.com/metrics"; }''; - description = '' - Prometheus endpoints this collector scrapes **by name, with a - credential**, as ` = ""`. - - A service module declares its own entry, from its own module, the - same way it does for `scrapeTargets` — and the collector's OAuth2 - client derives its permitted audiences from these URLs, so a - target and the authorisation to reach it are **one declaration**. - Two lists that must agree would be a drift to maintain, and its - failure mode is the bad one: a target whose audience was forgotten - authenticates against nothing and reads as a scrape failure rather - than a config mistake. - - ⚠️ A **full URL**, not `host:port`. Authelia validates a bearer - token against the address being requested, so the string here is - also the audience the token is minted for; a near miss (a trailing - slash, `http` for `https`) presents as a valid token rejected at - the target, several layers from its cause. - - ⚠️ Deliberately separate from `scrapeTargets`. An entry there is - trusted because the scraper and the target share a host — that - option is loopback-and-unauthenticated by contract. An entry here - is trusted because it presents a credential. One shape for both - would leave a reader unable to tell which of those a given target - relies on. - ''; - }; - - clientId = lib.mkOption { - type = lib.types.str; - readOnly = true; - default = "swarm-collector"; - description = '' - OAuth2 client id this collector authenticates as. Published so - authelia's `access_control` rules can name it without carrying a - second copy of the string, exactly as - `services.hyperhive.swarm.authelia.hiveClientPrefix` is published - for the queue's responder. - - Two spellings drifting apart is not a build failure: authelia - refuses a rule naming an unregistered client in its startup - validator, so the swarm's SSO service fails to *restart* — long - after the change that caused it evaluated cleanly. - ''; - }; - - audience = lib.mkOption { - type = lib.types.listOf lib.types.str; - readOnly = true; - default = lib.attrValues cfg.publishedScrapeTargets ++ lib.attrValues pushAudiences; - description = '' - Every audience this collector's OAuth2 client is permitted to - present a token for — the scrape targets it reads with a credential, - plus the stores it pushes to. Published read-only so - ./glue-swarm-otel-oidc-client.nix can register the client wherever - authelia runs without restating the derivation: this option and - that registration are the same fact seen from two hosts, and a - second formula for it would be free to drift from this one. - ''; - }; - }; - - # What stays above is what the collector IS to the swarm — the client it is - # registered as, where it exports. The secret is a path on the machine that - # runs it, so it hangs off the deployment. `enable` already lives in - # ./deploy.nix, which also carries the rename. + # What the collector IS to the swarm — the client it is registered as, where + # it exports — is `swarm.otel` in ./swarm-otel-service.nix. The secret is a + # path on the machine that runs it, so it hangs off the deployment. `enable` + # already lives in ./deploy.nix, which also carries the rename. options.services.hyperhive.deploy.swarm-otel = { clientSecretFile = lib.mkOption { type = lib.types.nullOr lib.types.str;