credential units: 24h retry shape; start a failed nginx when the cert lands
Six credential-fetch units retried 4 times at 15s, so an apply during
which the store or gateway was down for more than about a minute left
them in start-limit-hit, and nothing started them again once the store
came back. The swarm-services leaf could also land after nginx had
already given up on it, and the hook that propagates a new leaf only
reloaded a running nginx, so a stopped one stayed down until a second
apply.
- nix/host-modules/lib/store-retry.nix: the 2880 x 30s / 25h window
shape swarm-services-cert already had, as one attrset.
- swarm-services-cert, swarm-bao-otel-oidc, swarm-bao-forwarder-oidc,
swarm-bao-matrix-token, swarm-bao-queue-agent, swarm-bao-grafana-oidc,
hive-agent-bao-identity and hive-agent-forge-token use it.
queue-identity.nix no longer has a fetch unit (ccb5bd3b), and
forge-token.nix is a fetch unit with the same short budget that was
added after the census in #4662.
- The swarm-services-cert propagation hook now reset-fails and starts
(--no-block) a loaded nginx that is not active; an active nginx keeps
the re-import + reload.
- module-eval-bao-grants: one case pinning the shape on every host-side
fetch unit, swarm-services-cert included.
Refs #4662
This commit is contained in:
parent
3db1233da0
commit
b3b42d3279
10 changed files with 137 additions and 125 deletions
|
|
@ -61,6 +61,8 @@ let
|
||||||
# its default would decide whether the hive-side courier delivers into a
|
# its default would decide whether the hive-side courier delivers into a
|
||||||
# container that reads what it is given or into one that never looks.
|
# container that reads what it is given or into one that never looks.
|
||||||
configured = cfg.addr != null;
|
configured = cfg.addr != null;
|
||||||
|
|
||||||
|
storeRetry = import ../host-modules/lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
options.services.hyperhive.agent.bao = {
|
options.services.hyperhive.agent.bao = {
|
||||||
|
|
@ -96,21 +98,14 @@ in
|
||||||
pkgs.openbao
|
pkgs.openbao
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for a store that comes up around the same time this container
|
# ../host-modules/lib/store-retry.nix. From an agent the store is reached
|
||||||
# does, not for one that is sealed: a few short attempts cover the race,
|
# through the gateway's stream passthrough, so either being down fails
|
||||||
# and a longer window would only delay the report of a real failure.
|
# the login below.
|
||||||
#
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# `StartLimit*` are `[Unit]` settings, so they go here and not in
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# `serviceConfig` — systemd ignores them under `[Service]`. The window
|
|
||||||
# has to exceed `RestartSec` times the burst.
|
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
User = agentName;
|
User = agentName;
|
||||||
Group = agentName;
|
Group = agentName;
|
||||||
# Bare ids, no paths: the terse `LoadCredential=` form that inherits a
|
# Bare ids, no paths: the terse `LoadCredential=` form that inherits a
|
||||||
|
|
|
||||||
|
|
@ -49,6 +49,8 @@ let
|
||||||
# The store's address is the whole switch, as in ./bao.nix and
|
# The store's address is the whole switch, as in ./bao.nix and
|
||||||
# ./queue-identity.nix.
|
# ./queue-identity.nix.
|
||||||
configured = cfg.addr != null;
|
configured = cfg.addr != null;
|
||||||
|
|
||||||
|
storeRetry = import ../host-modules/lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
options.services.hyperhive.agent.forge.tokenFile = lib.mkOption {
|
options.services.hyperhive.agent.forge.tokenFile = lib.mkOption {
|
||||||
|
|
@ -90,9 +92,9 @@ in
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
pkgs.diffutils
|
pkgs.diffutils
|
||||||
];
|
];
|
||||||
startLimitBurst = 4;
|
# ../host-modules/lib/store-retry.nix.
|
||||||
startLimitIntervalSec = 300;
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
serviceConfig = {
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
# Not `RemainAfterExit`: the timer below has to be able to start this
|
# Not `RemainAfterExit`: the timer below has to be able to start this
|
||||||
# unit again, and an active unit cannot be started.
|
# unit again, and an active unit cannot be started.
|
||||||
|
|
@ -100,8 +102,6 @@ in
|
||||||
# in it, alive between runs instead.
|
# in it, alive between runs instead.
|
||||||
RemainAfterExit = false;
|
RemainAfterExit = false;
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
User = agentName;
|
User = agentName;
|
||||||
Group = agentName;
|
Group = agentName;
|
||||||
RuntimeDirectory = runtimeDir;
|
RuntimeDirectory = runtimeDir;
|
||||||
|
|
|
||||||
|
|
@ -82,6 +82,7 @@ let
|
||||||
matrixMachine = "hive-matrix";
|
matrixMachine = "hive-matrix";
|
||||||
|
|
||||||
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
|
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
config = lib.mkMerge [
|
config = lib.mkMerge [
|
||||||
|
|
@ -151,30 +152,18 @@ in
|
||||||
deployCfg.bao.package
|
deployCfg.bao.package
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for the race this loses, not for an unseal. `swarm-bao` comes up
|
# ./lib/store-retry.nix. ⚠️ This unit is `Before=` the homeserver's
|
||||||
# seconds before this unit asks, and the cert-auth role it logs in
|
# container, and an auto-restarting unit is still `activating`, so the
|
||||||
# against is written seconds after — so a few short attempts cover it.
|
# homeserver waits for as long as the login below keeps failing — up to
|
||||||
# ⚠️ `swarm-bao-controller-policy`'s 2880 × 30s is NOT the model to copy.
|
# the full 24h on a store that stays sealed.
|
||||||
# That unit blocks nothing; this one is `Before=` the homeserver's
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# container, and whether that ordering waits across an auto-restart is
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# unverified — so an hours-long window would be a bet on an unknown,
|
|
||||||
# where a minute is not. A store still sealed after it keeps the degrade
|
|
||||||
# below, as today.
|
|
||||||
#
|
|
||||||
# `StartLimit*` are `[Unit]` settings, so they go here and not in
|
|
||||||
# `serviceConfig` — systemd ignores them under `[Service]`. The window
|
|
||||||
# has to exceed `RestartSec × burst`.
|
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
# What actually bounds the read below. Stated here rather than
|
# What actually bounds each attempt below. Stated here rather than
|
||||||
# left to systemd's default, so the number a boot waits on is in
|
# left to systemd's default, so the number a boot waits on is in
|
||||||
# the file that waits.
|
# the file that waits.
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
};
|
};
|
||||||
environment = {
|
environment = {
|
||||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||||
|
|
@ -193,9 +182,11 @@ in
|
||||||
|
|
||||||
# A sealed or uninitialised store answers on the port and never
|
# A sealed or uninitialised store answers on the port and never
|
||||||
# answers the read, so "the store is up" is not the same as "the
|
# answers the read, so "the store is up" is not the same as "the
|
||||||
# store can answer". `TimeoutStartSec` above is the bound; the
|
# store can answer". `TimeoutStartSec` above bounds each attempt and
|
||||||
# homeserver only `Wants=` this unit, so hitting it degrades to
|
# a timed-out attempt is retried like any other failure; the
|
||||||
# keeping the local token rather than holding up the container.
|
# homeserver only `Wants=` this unit, so a retry budget spent
|
||||||
|
# degrades to keeping the local token rather than failing the
|
||||||
|
# container.
|
||||||
# `bao`'s own message is the only thing separating a missing value
|
# `bao`'s own message is the only thing separating a missing value
|
||||||
# from a refused identity from an unreachable host. This unit's
|
# from a refused identity from an unreachable host. This unit's
|
||||||
# degraded mode is correct for all three, so it reports which one
|
# degraded mode is correct for all three, so it reports which one
|
||||||
|
|
|
||||||
|
|
@ -79,6 +79,7 @@ let
|
||||||
credentialPath = "secret/swarm/hives/${hyperhiveCfg.hiveName}/queue/agent";
|
credentialPath = "secret/swarm/hives/${hyperhiveCfg.hiveName}/queue/agent";
|
||||||
|
|
||||||
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
|
atomicWriteSecret = import ./lib/atomic-write-secret.nix { };
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
options.services.hyperhive.deploy.hive-controller.queue = {
|
options.services.hyperhive.deploy.hive-controller.queue = {
|
||||||
|
|
@ -172,10 +173,10 @@ in
|
||||||
requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ];
|
requires = lib.optionals baoDeploy.enable [ "swarm-bao-pki.service" ];
|
||||||
# Ordered before hive-c0re, so no agent container renders ahead of an
|
# Ordered before hive-c0re, so no agent container renders ahead of an
|
||||||
# attempt at its credential. `Wants=`, not `Requires=`: a store this
|
# attempt at its credential. `Wants=`, not `Requires=`: a store this
|
||||||
# unit can't reach delays hive-c0re's start by its own start-limit
|
# unit can't reach delays hive-c0re's start for as long as this unit
|
||||||
# window (`TimeoutStartSec`, retried up to `startLimitBurst` times
|
# retries (up to the 24h of ./lib/store-retry.nix) rather than failing
|
||||||
# below) rather than failing it — hive-c0re starts once that window
|
# it — hive-c0re starts once the retries end, whatever credential is
|
||||||
# elapses, whatever credential is or isn't on disk by then.
|
# or isn't on disk by then.
|
||||||
before = [ "hive-c0re.service" ];
|
before = [ "hive-c0re.service" ];
|
||||||
wantedBy = [
|
wantedBy = [
|
||||||
"multi-user.target"
|
"multi-user.target"
|
||||||
|
|
@ -185,26 +186,15 @@ in
|
||||||
baoDeploy.package
|
baoDeploy.package
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for the race this loses, not for an unseal: `swarm-bao` comes up
|
# ./lib/store-retry.nix.
|
||||||
# seconds before this unit asks, and the cert-auth role it logs in
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# against is written seconds after, so a few short attempts cover it.
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# An hours-long window would be a bet on a store that is sealed, and the
|
|
||||||
# degrade below is already correct for that.
|
|
||||||
#
|
|
||||||
# `StartLimit*` are `[Unit]` settings, so they go here and not in
|
|
||||||
# `serviceConfig` — systemd ignores them under `[Service]`. The window
|
|
||||||
# has to exceed `RestartSec × burst`.
|
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
# What actually bounds the reads below. Stated here rather than left to
|
# What actually bounds each attempt at the reads below. Stated here
|
||||||
# systemd's default, so the number a boot waits on is in the file that
|
# rather than left to systemd's default, so the number a boot waits on
|
||||||
# waits.
|
# is in the file that waits.
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
};
|
};
|
||||||
environment = {
|
environment = {
|
||||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||||
|
|
|
||||||
|
|
@ -299,6 +299,7 @@ let
|
||||||
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
|
// lib.optionalAttrs (baoDeploy.serverCaFile != null) {
|
||||||
BAO_CACERT = baoDeploy.serverCaFile;
|
BAO_CACERT = baoDeploy.serverCaFile;
|
||||||
};
|
};
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
# Host-side TLS trust root for the self-signed gateway mode.
|
# Host-side TLS trust root for the self-signed gateway mode.
|
||||||
|
|
@ -751,21 +752,14 @@ in
|
||||||
];
|
];
|
||||||
wants = [ "container@${baoCfg.machine}.service" ];
|
wants = [ "container@${baoCfg.machine}.service" ];
|
||||||
path = servicesCertPath;
|
path = servicesCertPath;
|
||||||
# Sized like the store's own granting units, and for the same
|
# ./lib/store-retry.nix: the login below fails for as long as the
|
||||||
# reason: under `seal = "shamir"` an operator unseals BY HAND, and
|
# store is sealed or down.
|
||||||
# the login below fails for as long as that takes. 2880 × 30s is
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# 24h inside a 25h window — `StartLimit*` are `[Unit]` settings, so
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# the window must exceed `RestartSec × burst` or it closes between
|
|
||||||
# attempts and the burst is never reached.
|
|
||||||
startLimitBurst = 2880;
|
|
||||||
startLimitIntervalSec = 90000;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
UMask = "0077";
|
UMask = "0077";
|
||||||
SyslogIdentifier = "swarm-services-cert";
|
SyslogIdentifier = "swarm-services-cert";
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 30;
|
|
||||||
};
|
};
|
||||||
environment = servicesCertEnvironment;
|
environment = servicesCertEnvironment;
|
||||||
script = ''
|
script = ''
|
||||||
|
|
@ -926,18 +920,18 @@ in
|
||||||
mv -f "$svcroot.new" "$svcroot"
|
mv -f "$svcroot.new" "$svcroot"
|
||||||
chmod 0644 "$svcroot"
|
chmod 0644 "$svcroot"
|
||||||
|
|
||||||
# Propagation, for the retry and renewal paths only. Ordered before
|
# Propagation. nginx serves a *copy* of the leaf, so re-issuing the
|
||||||
# both of these, so on a normal boot they have not run yet,
|
# source changes nothing until the copy is remade. A running
|
||||||
# `is-active` is false, and ordering alone does the work. What this
|
# gateway (`swarm-services-cert-renew` rotating the leaf) gets the
|
||||||
# covers is the store coming up hours after the gateway did, and
|
# copy remade and a reload. A gateway that is not running gets
|
||||||
# `swarm-services-cert-renew` rotating the leaf under a running
|
# started: a start job still waiting on this unit absorbs the
|
||||||
# gateway: nginx serves a *copy* of the leaf, so re-issuing the
|
# request, and a start that already ended on this unit's
|
||||||
# source changes nothing until the copy is remade.
|
# `requiredBy` has nothing else that would start it again.
|
||||||
#
|
#
|
||||||
# ⚠️ `--no-block`, and it is not a preference. This unit declares
|
# ⚠️ `--no-block`, and it is not a preference. This unit declares
|
||||||
# `Before=` both of these, so a blocking `systemctl restart`
|
# `Before=` both of these, so a blocking `systemctl restart` or
|
||||||
# enqueues a job that systemd will not start until this unit is
|
# `start` enqueues a job that systemd will not start until this
|
||||||
# active — and this unit is not active until its ExecStart
|
# unit is active — and this unit is not active until its ExecStart
|
||||||
# returns, which is waiting on that job. A deadlock, held until
|
# returns, which is waiting on that job. A deadlock, held until
|
||||||
# the 24h retry window's `TimeoutStartSec` fires. Queueing the
|
# the 24h retry window's `TimeoutStartSec` fires. Queueing the
|
||||||
# job and letting it run once we exit is the only ordering that
|
# job and letting it run once we exit is the only ordering that
|
||||||
|
|
@ -952,6 +946,10 @@ in
|
||||||
echo "swarm-services leaf rotated — re-importing and reloading nginx"
|
echo "swarm-services leaf rotated — re-importing and reloading nginx"
|
||||||
systemctl restart --no-block hive-gateway-self-signed-cert.service
|
systemctl restart --no-block hive-gateway-self-signed-cert.service
|
||||||
systemctl reload --no-block nginx.service
|
systemctl reload --no-block nginx.service
|
||||||
|
elif [ "$(systemctl show -P LoadState nginx.service)" = loaded ]; then
|
||||||
|
echo "swarm-services leaf landed with nginx not running — starting it"
|
||||||
|
systemctl reset-failed nginx.service
|
||||||
|
systemctl start --no-block nginx.service
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# The bundle is assembled by `hive-tls-ca`, which runs BEFORE this
|
# The bundle is assembled by `hive-tls-ca`, which runs BEFORE this
|
||||||
|
|
|
||||||
32
nix/host-modules/lib/store-retry.nix
Normal file
32
nix/host-modules/lib/store-retry.nix
Normal file
|
|
@ -0,0 +1,32 @@
|
||||||
|
# Retry shape for a oneshot that fetches a credential or certificate from
|
||||||
|
# the secret store: every 30s for 24h. Under `seal = "shamir"` an operator
|
||||||
|
# unseals BY HAND, and a fetch fails for as long as that takes, or for as
|
||||||
|
# long as the store or the gateway in front of it is down. `start-limit-hit`
|
||||||
|
# does not clear when the store comes back, so a budget shorter than the
|
||||||
|
# outage leaves the unit failed until something starts it again.
|
||||||
|
#
|
||||||
|
# `StartLimit*` are `[Unit]` settings — systemd ignores them under
|
||||||
|
# `[Service]` — and the window must exceed `RestartSec × burst` or it closes
|
||||||
|
# between attempts and the burst is never reached: 2880 × 30s is 24h inside
|
||||||
|
# a 25h window.
|
||||||
|
#
|
||||||
|
# ⚠️ A unit in auto-restart is still `activating`, so its start job stays
|
||||||
|
# queued across attempts: anything ordered `After=` it waits for as long as
|
||||||
|
# it retries, up to the full 24h.
|
||||||
|
#
|
||||||
|
# Pure attrset — NOT a NixOS module. Use from a unit definition:
|
||||||
|
#
|
||||||
|
# storeRetry = import ./lib/store-retry.nix { };
|
||||||
|
# systemd.services.foo = {
|
||||||
|
# inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
|
# serviceConfig = storeRetry.serviceConfig // { Type = "oneshot"; };
|
||||||
|
# };
|
||||||
|
{ }:
|
||||||
|
{
|
||||||
|
startLimitBurst = 2880;
|
||||||
|
startLimitIntervalSec = 90000;
|
||||||
|
serviceConfig = {
|
||||||
|
Restart = "on-failure";
|
||||||
|
RestartSec = 30;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
@ -1188,6 +1188,7 @@ let
|
||||||
# sharing the netns is what lets the store bind the host's own addresses
|
# sharing the netns is what lets the store bind the host's own addresses
|
||||||
# rather than a convenience for nginx.
|
# rather than a convenience for nginx.
|
||||||
privateNetwork = false;
|
privateNetwork = false;
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
imports = [
|
imports = [
|
||||||
|
|
@ -2379,20 +2380,15 @@ in
|
||||||
baoDeploy.package
|
baoDeploy.package
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for the race this loses: the secret is minted by authelia and
|
# ./lib/store-retry.nix: the secret is minted by authelia and copied in
|
||||||
# copied in by the publisher, both of which may be a host away and
|
# by the publisher, both of which may be a host away and neither of
|
||||||
# neither of which this boot waits on. `StartLimit*` are `[Unit]`
|
# which this boot waits on.
|
||||||
# settings — systemd ignores them under `[Service]` — and the window
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# has to exceed `RestartSec × burst`.
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
SyslogIdentifier = "swarm-bao-forwarder-oidc";
|
SyslogIdentifier = "swarm-bao-forwarder-oidc";
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
};
|
};
|
||||||
environment = {
|
environment = {
|
||||||
BAO_ADDR = "https://${cfg.domain}:${toString cfg.port}";
|
BAO_ADDR = "https://${cfg.domain}:${toString cfg.port}";
|
||||||
|
|
|
||||||
|
|
@ -243,6 +243,7 @@ let
|
||||||
# Shared host netns, like every sibling swarm container: the gateway
|
# Shared host netns, like every sibling swarm container: the gateway
|
||||||
# reaches this at 127.0.0.1:<port>.
|
# reaches this at 127.0.0.1:<port>.
|
||||||
privateNetwork = false;
|
privateNetwork = false;
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
# `enable` moved to `services.hyperhive.deploy.grafana.enable` — see
|
# `enable` moved to `services.hyperhive.deploy.grafana.enable` — see
|
||||||
|
|
@ -673,27 +674,17 @@ in
|
||||||
baoDeploy.package
|
baoDeploy.package
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for the race this loses, not for an unseal: `swarm-bao` comes up
|
# ./lib/store-retry.nix. Grafana's container is ordered after this unit,
|
||||||
# seconds before this unit asks, and the cert-auth role it logs in
|
# so it waits while the login below keeps failing.
|
||||||
# against is written seconds after, so a few short attempts cover it. An
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# hours-long window would be a bet on a sealed store, and the degrade
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# below is already correct for that.
|
|
||||||
#
|
|
||||||
# `StartLimit*` are `[Unit]` settings, so they go here and not in
|
|
||||||
# `serviceConfig` — systemd ignores them under `[Service]`. The window
|
|
||||||
# has to exceed `RestartSec × burst`.
|
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
SyslogIdentifier = "swarm-bao-grafana-oidc";
|
SyslogIdentifier = "swarm-bao-grafana-oidc";
|
||||||
# What actually bounds the read below. Stated here rather than left to
|
# What actually bounds each attempt below. Stated here rather than left
|
||||||
# systemd's default, so the number a boot waits on is in the file that
|
# to systemd's default, so the number a boot waits on is in the file
|
||||||
# waits.
|
# that waits.
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
};
|
};
|
||||||
environment = {
|
environment = {
|
||||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||||
|
|
|
||||||
|
|
@ -310,6 +310,7 @@ let
|
||||||
# store, without either crossing a network boundary that would need
|
# store, without either crossing a network boundary that would need
|
||||||
# its own trust material.
|
# its own trust material.
|
||||||
privateNetwork = false;
|
privateNetwork = false;
|
||||||
|
storeRetry = import ./lib/store-retry.nix { };
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
# `enable` moved to `services.hyperhive.deploy.swarm-otel.enable` — see ./deploy.nix.
|
# `enable` moved to `services.hyperhive.deploy.swarm-otel.enable` — see ./deploy.nix.
|
||||||
|
|
@ -790,27 +791,17 @@ in
|
||||||
baoDeploy.package
|
baoDeploy.package
|
||||||
pkgs.coreutils
|
pkgs.coreutils
|
||||||
];
|
];
|
||||||
# Sized for the race this loses, not for an unseal: `swarm-bao` comes up
|
# ./lib/store-retry.nix. The collector's container is ordered after
|
||||||
# seconds before this unit asks, and the cert-auth role it logs in
|
# this unit, so it waits while the login below keeps failing.
|
||||||
# against is written seconds after, so a few short attempts cover it. An
|
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||||||
# hours-long window would be a bet on a sealed store, and the degrade
|
serviceConfig = storeRetry.serviceConfig // {
|
||||||
# below is already correct for that.
|
|
||||||
#
|
|
||||||
# `StartLimit*` are `[Unit]` settings, so they go here and not in
|
|
||||||
# `serviceConfig` — systemd ignores them under `[Service]`. The window
|
|
||||||
# has to exceed `RestartSec × burst`.
|
|
||||||
startLimitBurst = 4;
|
|
||||||
startLimitIntervalSec = 300;
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
Type = "oneshot";
|
||||||
RemainAfterExit = true;
|
RemainAfterExit = true;
|
||||||
SyslogIdentifier = "swarm-bao-otel-oidc";
|
SyslogIdentifier = "swarm-bao-otel-oidc";
|
||||||
# What actually bounds the read below. Stated here rather than left to
|
# What actually bounds each attempt below. Stated here rather than left
|
||||||
# systemd's default, so the number a boot waits on is in the file that
|
# to systemd's default, so the number a boot waits on is in the file
|
||||||
# waits.
|
# that waits.
|
||||||
TimeoutStartSec = 30;
|
TimeoutStartSec = 30;
|
||||||
Restart = "on-failure";
|
|
||||||
RestartSec = 15;
|
|
||||||
};
|
};
|
||||||
environment = {
|
environment = {
|
||||||
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
BAO_ADDR = "https://${baoCfg.domain}:${toString baoCfg.port}";
|
||||||
|
|
|
||||||
|
|
@ -866,6 +866,34 @@ let
|
||||||
&& u.serviceConfig.Restart == "on-failure"
|
&& u.serviceConfig.Restart == "on-failure"
|
||||||
) grantingUnitNames;
|
) grantingUnitNames;
|
||||||
}
|
}
|
||||||
|
{
|
||||||
|
# A fetch from the store outlives a store or gateway that is down or
|
||||||
|
# sealed for hours: `start-limit-hit` does not clear when the store comes
|
||||||
|
# back, so a shorter budget leaves the credential missing until someone
|
||||||
|
# starts the unit by hand.
|
||||||
|
name = "every unit fetching a credential or certificate from the store retries 2880 times at 30s";
|
||||||
|
ok =
|
||||||
|
lib.all
|
||||||
|
(
|
||||||
|
unit:
|
||||||
|
let
|
||||||
|
s = baoGrantWithConsumers.systemd.services;
|
||||||
|
u = s.${unit};
|
||||||
|
in
|
||||||
|
s ? ${unit}
|
||||||
|
&& toString u.unitConfig.StartLimitBurst == "2880"
|
||||||
|
&& toString u.unitConfig.StartLimitIntervalSec == "90000"
|
||||||
|
&& toString u.serviceConfig.RestartSec == "30"
|
||||||
|
&& u.serviceConfig.Restart == "on-failure"
|
||||||
|
)
|
||||||
|
(
|
||||||
|
[
|
||||||
|
"swarm-services-cert"
|
||||||
|
"swarm-bao-forwarder-oidc"
|
||||||
|
]
|
||||||
|
++ policyReaders
|
||||||
|
);
|
||||||
|
}
|
||||||
{
|
{
|
||||||
# The only unit left acting with the token, so the only one that may
|
# The only unit left acting with the token, so the only one that may
|
||||||
# skip on it.
|
# skip on it.
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue