Every unit that writes a bao policy or cert-auth role ran only while the operator-placed bootstrap token existed, and skipped silently otherwise. The token lives 24h, so on any real swarm a PR adding or changing a grant deployed with its unit skipped, and each one needed a manual token refresh (plus a root `bao policy write` when it added a path). A `bao-granter` principal now writes them. Its leaf is minted by swarm-bao-pki on the store host (0600 root, never copied off it), and its policy covers `swarm-*` policies, `swarm-*` cert-auth roles and `pki/roles/swarm-*` by glob, plus the mount and services-root paths the controller's unit already used. All ten granting units (controller, secret-publisher, matrix-ctl, matrix-token, queue-agent, grafana-oidc, otel-oidc, forwarder-oidc, services-issuer, nats-tls) log in with it instead of reading the token. They keep the 2880 x 30s retry, now require swarm-bao-pki, and when the store refuses the granter they fail and print the one-time step instead of skipping. swarm-bao-granter-role is the one unit left on the token. It enables the auth mounts (moved out of the controller's unit) and writes the granter's own policy and role. The bootstrap policy is renamed `bao-bootstrap` and shrinks to those five stanzas; it is shipped at /etc/hyperhive/bao-bootstrap-policy.hcl. The old name `swarm-bootstrap` matched the granter's own `swarm-*` glob. The granter's CN joins certAuthCns, so no hive can be named into its role. An assertion keeps both pki role names under `swarm-`. With no client CA the granting units no longer render, and a warning says so. module-eval pins the granter's policy stanza by stanza, what it cannot reach, that every call a granting unit makes is granted, and that only swarm-bao-granter-role reads the token. Refs #4704
126 lines
4.9 KiB
Nix
126 lines
4.9 KiB
Nix
# `checks.module-eval-hive-tls` — see ./lib.nix for the shared rationale (why
|
|
# this suite exists, naming convention, "evaluates not executes").
|
|
#
|
|
# The swarm-services leaf's renewal: a timer that really re-runs the
|
|
# issuance, at a threshold read off the store role's lifetime.
|
|
{
|
|
pkgs,
|
|
lib,
|
|
self,
|
|
nixosSystem,
|
|
}:
|
|
let
|
|
inherit
|
|
(import ./lib.nix {
|
|
inherit
|
|
pkgs
|
|
lib
|
|
self
|
|
nixosSystem
|
|
;
|
|
})
|
|
hive
|
|
runGroup
|
|
;
|
|
|
|
# Every service on one host, so the store's granting unit renders the role
|
|
# this leaf is issued through.
|
|
allLocal = hive {
|
|
deploy.singleHostSwarm = true;
|
|
};
|
|
|
|
# The same host with the role's lifetime moved, so a threshold that is a
|
|
# number of its own shows up as one that did not move with it.
|
|
shortTtl = hive {
|
|
deploy.singleHostSwarm = true;
|
|
deploy.bao.servicesPkiLeafTtlHours = 48;
|
|
};
|
|
|
|
units = allLocal.systemd.services;
|
|
boot = units.swarm-services-cert;
|
|
timer = allLocal.systemd.timers.swarm-services-cert-renew;
|
|
|
|
# What the timer starts, read off the timer rather than assumed from its
|
|
# name: a timer's `Unit=` defaults to its own name, and pointing it at the
|
|
# boot unit is exactly the regression the cases below exist for.
|
|
triggeredName = lib.removeSuffix ".service" (
|
|
timer.timerConfig.Unit or "swarm-services-cert-renew.service"
|
|
);
|
|
triggered = units.${triggeredName};
|
|
|
|
remainsAfterExit = u: u.serviceConfig.RemainAfterExit or false;
|
|
|
|
renewAt = hours: "renewat=$(( ${toString hours} * 3600 / 2 ))";
|
|
|
|
cases = [
|
|
{
|
|
# Control: the boot issuance is the unit a timer cannot re-run. Were it
|
|
# not `RemainAfterExit`, the case after next would pass vacuously.
|
|
name = "the boot issuance renders, and stays active after it exits";
|
|
ok = lib.hasInfix "pki/issue/swarm-services" boot.script && remainsAfterExit boot;
|
|
}
|
|
{
|
|
name = "a timer for the services leaf is enabled and fires daily";
|
|
ok =
|
|
lib.elem "timers.target" timer.wantedBy
|
|
&& timer.timerConfig.OnCalendar == "daily"
|
|
&& timer.timerConfig.Persistent;
|
|
}
|
|
{
|
|
# Starting an active unit is a no-op, so a timer aimed at a
|
|
# `RemainAfterExit` unit fires on schedule and renews nothing.
|
|
name = "the timer starts a unit that re-runs the issuance and does not stay active";
|
|
ok =
|
|
units ? ${triggeredName}
|
|
&& !(remainsAfterExit triggered)
|
|
&& triggered.script == boot.script
|
|
# The whole set, `PATH` included: an environment copied off the
|
|
# boot unit's merged options renders a second `PATH` and fails.
|
|
&& triggered.environment == boot.environment;
|
|
}
|
|
{
|
|
# A restart of anything the gateway's cert import `Requires=` takes
|
|
# nginx down with it, so the renewal must be something nothing else
|
|
# depends on, and must run behind the boot issuance.
|
|
name = "nothing requires or waits on the renewal, and it runs after the boot issuance";
|
|
ok =
|
|
(triggered.requiredBy or [ ]) == [ ]
|
|
&& (triggered.wantedBy or [ ]) == [ ]
|
|
&& (triggered.before or [ ]) == [ ]
|
|
&& lib.elem "swarm-services-cert.service" triggered.after
|
|
&& !(lib.any (u: lib.elem "${triggeredName}.service" ((u.requires or [ ]) ++ (u.after or [ ]))) (
|
|
lib.attrValues (removeAttrs units [ triggeredName ])
|
|
));
|
|
}
|
|
{
|
|
name = "the renewal's journal ships beside the boot issuance's";
|
|
ok = lib.elem triggeredName allLocal.services.hyperhive.swarm.otel.journaldUnits;
|
|
}
|
|
{
|
|
# Moved with the role, in both places: the threshold is half of what the
|
|
# store grants, and the store grants what the option says.
|
|
name = "the leaf is renewed at half the role's lifetime, read from the option";
|
|
ok =
|
|
lib.hasInfix (renewAt 720) boot.script
|
|
&& lib.hasInfix (renewAt 48) shortTtl.systemd.services.swarm-services-cert.script
|
|
&& lib.hasInfix ''-in "$svcleaf" -noout -checkend "$renewat"'' boot.script
|
|
&& lib.hasInfix "max_ttl=48h" shortTtl.systemd.services.swarm-bao-controller-policy.script;
|
|
}
|
|
{
|
|
# The store replaces its root once it has less than one leaf lifetime
|
|
# left, and the hive asks for a fresh leaf, with the new root, at that
|
|
# same threshold. A number of its own here keeps a leaf whose root the
|
|
# store has already replaced.
|
|
name = "the services root is re-checked at the store's own replacement threshold";
|
|
ok =
|
|
let
|
|
shortBoot = shortTtl.systemd.services.swarm-services-cert.script;
|
|
shortStore = shortTtl.systemd.services.swarm-bao-controller-policy.script;
|
|
in
|
|
lib.hasInfix ''-in "$svcroot" -noout -checkend ${toString (720 * 3600)} '' boot.script
|
|
&& lib.hasInfix ''-in "$svcroot" -noout -checkend ${toString (48 * 3600)} '' shortBoot
|
|
&& lib.hasInfix "-checkend ${toString (48 * 3600)} <<<\"$root_ca\"" shortStore;
|
|
}
|
|
];
|
|
in
|
|
runGroup "hive-tls" cases
|