diff --git a/nix/host-modules/hive-gateway/default.nix b/nix/host-modules/hive-gateway/default.nix index f3ff3863..a492ca48 100644 --- a/nix/host-modules/hive-gateway/default.nix +++ b/nix/host-modules/hive-gateway/default.nix @@ -129,8 +129,13 @@ in # Every request to every hyperhive service passes through here, so this # is the one unit that can say a service was unreachable rather than # merely quiet. Named even on hives that run no swarm collector: the - # option is inert unless one is collecting on this host. - services.hyperhive.swarm.otel.journaldUnits = [ "nginx" ]; + # option is inert unless one is collecting on this host. nginx + # `Requires=` the self-signed copy, so its journal is the other half of + # why nginx did not start. + services.hyperhive.swarm.otel.journaldUnits = [ + "nginx" + ] + ++ lib.optional useSelfSigned "hive-gateway-self-signed-cert"; assertions = [ { diff --git a/nix/host-modules/hive-tls.nix b/nix/host-modules/hive-tls.nix index cc71e5fe..7771b705 100644 --- a/nix/host-modules/hive-tls.nix +++ b/nix/host-modules/hive-tls.nix @@ -367,6 +367,13 @@ in # rebuild of every hive that is not the CA host, about a fallback # that no longer happens. + # Both run on the deploy and sit on the gateway's start path, so a + # failure in either takes TLS down for every service behind it. + services.hyperhive.swarm.otel.journaldUnits = [ + "hive-tls-ca" + "swarm-services-cert" + ]; + # Generate (and rotate) the hive CA + gateway leaf before anything # serves it. Idempotent: the CA is created once and reused; the leaf # is re-signed on expiry under the same CA so the anchor is stable. diff --git a/nix/host-modules/swarm-bao.nix b/nix/host-modules/swarm-bao.nix index 49dd41fc..da734167 100644 --- a/nix/host-modules/swarm-bao.nix +++ b/nix/host-modules/swarm-bao.nix @@ -1500,7 +1500,7 @@ in # addresses on every command. environment.systemPackages = [ baoCli ]; - # The in-container unit plus the two host-side ones this module defines. + # The in-container unit plus the host-side ones this module defines. # `swarm-bao-pki` and `swarm-bao-matrix-token` are declared by the glue # modules that create them, per the option's own rule — and a name # nothing defines is silently ignored, so naming them from here would @@ -1510,6 +1510,13 @@ in "swarm-bao-certs" "swarm-bao-token" "swarm-bao-forwarder-oidc" + "swarm-bao-controller-policy" + "swarm-bao-secret-publisher-policy" + "swarm-bao-matrix-ctl-policy" + "swarm-bao-matrix-token-policy" + "swarm-bao-queue-agent-policy" + "swarm-bao-grafana-oidc-policy" + "swarm-bao-otel-oidc-policy" "swarm-bao-services-issuer-policy" ]; diff --git a/nix/host-modules/swarm-otel.nix b/nix/host-modules/swarm-otel.nix index 9ee4ee0b..0b5cc904 100644 --- a/nix/host-modules/swarm-otel.nix +++ b/nix/host-modules/swarm-otel.nix @@ -660,9 +660,17 @@ in # delivering says so in the store it stopped delivering to. That is # less circular than it sounds: the failure that matters here is a # single hive's receiver refusing pushes, not the process dying. + # + # The container's own host unit is what says the collector never started + # — a failed bind source or a `Requires=` that did not come up — which + # the process inside cannot report. The activation has no module to name + # it: nixos-rebuild runs switch-to-configuration as this transient unit, + # and its syslog lines are the deploy's start, finish or failure status. services.hyperhive.swarm.otel.journaldUnits = [ "opentelemetry-collector" "swarm-bao-otel-oidc" + "container@${cfg.machine}" + "nixos-rebuild-switch-to-configuration" ]; # The metrics counterpart to the journal line above, closing the same diff --git a/nix/module-eval/swarm-otel-core.nix b/nix/module-eval/swarm-otel-core.nix index 0bbfa10e..6192fe94 100644 --- a/nix/module-eval/swarm-otel-core.nix +++ b/nix/module-eval/swarm-otel-core.nix @@ -61,7 +61,54 @@ let deploy.bao.otelOidcClientCertFile = "/etc/pki/bao-otel-oidc.pem"; deploy.bao.otelOidcClientKeyFile = "/etc/pki/bao-otel-oidc-key.pem"; }; + + # The collector beside the store holding a bootstrap token: the one shape in + # which every unit an apply can leave failed renders on the same host. + otelApplyPath = hive { + deploy.swarm-otel.enable = true; + deploy.authelia.enable = true; + deploy.bao.enable = true; + deploy.bao.bootstrapTokenFile = "/run/secrets/bao-bootstrap.token"; + }; + + # Units on the path a deploy takes to TLS, the store's grants and the + # collector itself. A failure among them silences ingest, so without their + # journals the store can show that ingest stopped but not which unit + # stopped it. + applyPathUnits = [ + "container@swarm-otel" + "hive-tls-ca" + "swarm-services-cert" + "hive-gateway-self-signed-cert" + "swarm-bao-controller-policy" + "swarm-bao-secret-publisher-policy" + "swarm-bao-matrix-ctl-policy" + "swarm-bao-matrix-token-policy" + "swarm-bao-queue-agent-policy" + "swarm-bao-grafana-oidc-policy" + "swarm-bao-otel-oidc-policy" + "swarm-bao-services-issuer-policy" + ]; cases = [ + { + # Listed AND defined, because a name that matches nothing is not an + # error anywhere: a unit renamed out from under its entry would pass a + # membership check and still never reach the store. + name = "every unit on the apply path is defined and ships its journal"; + ok = + let + m = otelApplyPath; + in + lib.all ( + u: builtins.elem u m.services.hyperhive.swarm.otel.journaldUnits && m.systemd.services ? ${u} + ) applyPathUnits; + } + { + # Transient, so nothing here defines it and only membership can be + # pinned: nixos-rebuild names the unit it runs the activation in. + name = "the activation's journal ships beside the units it starts"; + ok = builtins.elem "nixos-rebuild-switch-to-configuration" otelApplyPath.services.hyperhive.swarm.otel.journaldUnits; + } { # 🩸 The arm that guards the ruling this slice landed under, the # collector's half of ./swarm-grafana.nix's own. There is ONE delivery