Watch
0
0
Fork
You've already forked hyperhive
0
hyperhive/nix/module-eval/hive-tls.nix
atlas 6b1e825c0a swarm-otel: ship the whole host journal, drop user sessions after it
The swarm collector's journald receiver read only the units listed in
`services.hyperhive.swarm.otel.journaldUnits`. A unit nobody listed
never reached the store, and a misspelt entry shipped nothing without
an error. The list existed to keep an operator's desktop session out of
a store every swarm operator can read, but the receiver can only match
positively, so the only way to express "not user sessions" was to name
every service instead.

The receiver now reads the whole host journal, and a new
`filter/exclude-user-sessions` processor in the `logs/<swarm>` pipeline
drops records whose `_SYSTEMD_SLICE` is `user-<uid>.slice` (session
scopes and `user@<uid>.service`). The per-hive `logs/<hive>` pipelines
carry agent-container journals only and get no filter.

`journaldUnits` is removed with `mkRemovedOptionModule`, together with
its non-empty assertion and the entry each host module added. The four
module-eval membership checks go with it, replaced by one structural
case in swarm-otel-core.

Closes #3646
2026-09-30 23:01:49 +02:00

122 lines
4.7 KiB
Nix

# `checks.module-eval-hive-tls` — see ./lib.nix for the shared rationale (why
# this suite exists, naming convention, "evaluates not executes").
#
# The swarm-services leaf's renewal: a timer that really re-runs the
# issuance, at a threshold read off the store role's lifetime.
{
pkgs,
lib,
self,
nixosSystem,
}:
let
inherit
(import ./lib.nix {
inherit
pkgs
lib
self
nixosSystem
;
})
hive
runGroup
;
# Every service on one host, so the store's granting unit renders the role
# this leaf is issued through.
allLocal = hive {
deploy.singleHostSwarm = true;
};
# The same host with the role's lifetime moved, so a threshold that is a
# number of its own shows up as one that did not move with it.
shortTtl = hive {
deploy.singleHostSwarm = true;
deploy.bao.servicesPkiLeafTtlHours = 48;
};
units = allLocal.systemd.services;
boot = units.swarm-services-cert;
timer = allLocal.systemd.timers.swarm-services-cert-renew;
# What the timer starts, read off the timer rather than assumed from its
# name: a timer's `Unit=` defaults to its own name, and pointing it at the
# boot unit is exactly the regression the cases below exist for.
triggeredName = lib.removeSuffix ".service" (
timer.timerConfig.Unit or "swarm-services-cert-renew.service"
);
triggered = units.${triggeredName};
remainsAfterExit = u: u.serviceConfig.RemainAfterExit or false;
renewAt = hours: "renewat=$(( ${toString hours} * 3600 / 2 ))";
cases = [
{
# Control: the boot issuance is the unit a timer cannot re-run. Were it
# not `RemainAfterExit`, the case after next would pass vacuously.
name = "the boot issuance renders, and stays active after it exits";
ok = lib.hasInfix "pki/issue/swarm-services" boot.script && remainsAfterExit boot;
}
{
name = "a timer for the services leaf is enabled and fires daily";
ok =
lib.elem "timers.target" timer.wantedBy
&& timer.timerConfig.OnCalendar == "daily"
&& timer.timerConfig.Persistent;
}
{
# Starting an active unit is a no-op, so a timer aimed at a
# `RemainAfterExit` unit fires on schedule and renews nothing.
name = "the timer starts a unit that re-runs the issuance and does not stay active";
ok =
units ? ${triggeredName}
&& !(remainsAfterExit triggered)
&& triggered.script == boot.script
# The whole set, `PATH` included: an environment copied off the
# boot unit's merged options renders a second `PATH` and fails.
&& triggered.environment == boot.environment;
}
{
# A restart of anything the gateway's cert import `Requires=` takes
# nginx down with it, so the renewal must be something nothing else
# depends on, and must run behind the boot issuance.
name = "nothing requires or waits on the renewal, and it runs after the boot issuance";
ok =
(triggered.requiredBy or [ ]) == [ ]
&& (triggered.wantedBy or [ ]) == [ ]
&& (triggered.before or [ ]) == [ ]
&& lib.elem "swarm-services-cert.service" triggered.after
&& !(lib.any (u: lib.elem "${triggeredName}.service" ((u.requires or [ ]) ++ (u.after or [ ]))) (
lib.attrValues (removeAttrs units [ triggeredName ])
));
}
{
# Moved with the role, in both places: the threshold is half of what the
# store grants, and the store grants what the option says.
name = "the leaf is renewed at half the role's lifetime, read from the option";
ok =
lib.hasInfix (renewAt 720) boot.script
&& lib.hasInfix (renewAt 48) shortTtl.systemd.services.swarm-services-cert.script
&& lib.hasInfix ''-in "$svcleaf" -noout -checkend "$renewat"'' boot.script
&& lib.hasInfix "max_ttl=48h" shortTtl.systemd.services.swarm-bao-controller-policy.script;
}
{
# The store replaces its root once it has less than one leaf lifetime
# left, and the hive asks for a fresh leaf, with the new root, at that
# same threshold. A number of its own here keeps a leaf whose root the
# store has already replaced.
name = "the services root is re-checked at the store's own replacement threshold";
ok =
let
shortBoot = shortTtl.systemd.services.swarm-services-cert.script;
shortStore = shortTtl.systemd.services.swarm-bao-controller-policy.script;
in
lib.hasInfix ''-in "$svcroot" -noout -checkend ${toString (720 * 3600)} '' boot.script
&& lib.hasInfix ''-in "$svcroot" -noout -checkend ${toString (48 * 3600)} '' shortBoot
&& lib.hasInfix "-checkend ${toString (48 * 3600)} <<<\"$root_ca\"" shortStore;
}
];
in
runGroup "hive-tls" cases