Six credential-fetch units retried 4 times at 15s, so an apply during
which the store or gateway was down for more than about a minute left
them in start-limit-hit, and nothing started them again once the store
came back. The swarm-services leaf could also land after nginx had
already given up on it, and the hook that propagates a new leaf only
reloaded a running nginx, so a stopped one stayed down until a second
apply.
- nix/host-modules/lib/store-retry.nix: the 2880 x 30s / 25h window
shape swarm-services-cert already had, as one attrset.
- swarm-services-cert, swarm-bao-otel-oidc, swarm-bao-forwarder-oidc,
swarm-bao-matrix-token, swarm-bao-queue-agent, swarm-bao-grafana-oidc,
hive-agent-bao-identity and hive-agent-forge-token use it.
queue-identity.nix no longer has a fetch unit (ccb5bd3b), and
forge-token.nix is a fetch unit with the same short budget that was
added after the census in #4662.
- The swarm-services-cert propagation hook now reset-fails and starts
(--no-block) a loaded nginx that is not active; an active nginx keeps
the re-import + reload.
- module-eval-bao-grants: one case pinning the shape on every host-side
fetch unit, swarm-services-cert included.
Refs #4662
208 lines
8.6 KiB
Nix
208 lines
8.6 KiB
Nix
# This agent's own forge access token, fetched from the swarm secret store by
|
||
# the agent itself.
|
||
#
|
||
# `swarm-controller` mints the token with the forge's admin API and writes it
|
||
# to `swarm/agents/<agent>/forge-token`
|
||
# (`swarm_secret_client::forge::agent_token_path`); no hive is in that chain.
|
||
# This unit logs in to the store with the certificate ./bao.nix already proves
|
||
# it can log in with, and reads its own path. A hive handing the token over
|
||
# would be the hive reading a secret on the agent's behalf.
|
||
#
|
||
# The token rotates: the controller replaces it when the forge's copy stops
|
||
# matching the stored one. So the unit is re-run by a timer, and it swaps the
|
||
# file in by rename, only when the value changed, so a reader never sees half
|
||
# a token and a watcher on the file (forge-avatar-sync.path) fires only on a
|
||
# real change.
|
||
#
|
||
# Consumers read `services.hyperhive.agent.forge.tokenFile` first and fall back
|
||
# to `<state>/forge-token`, the file the hive wrote before this existed.
|
||
{
|
||
pkgs,
|
||
lib,
|
||
config,
|
||
...
|
||
}:
|
||
let
|
||
cfg = config.services.hyperhive.agent.bao;
|
||
|
||
# The name `swarm-controller` minted the token under — see ./queue-identity.nix.
|
||
agentName = config.services.hyperhive.agent.user.name;
|
||
|
||
# The same three ids ./bao.nix and ./queue-identity.nix load.
|
||
certCredential = "hive-agent-bao-cert";
|
||
keyCredential = "hive-agent-bao-key";
|
||
serverCaCredential = "hive-agent-bao-server-ca";
|
||
|
||
unitName = "hive-agent-forge-token";
|
||
|
||
# The nix half of `swarm_secret_client::forge::agent_token_path` plus
|
||
# `path::MOUNT`.
|
||
tokenPath = "secret/swarm/agents/${agentName}/forge-token";
|
||
|
||
runtimeDir = unitName;
|
||
tokenFile = "/run/${runtimeDir}/token";
|
||
# Beside the token, so the rename that replaces it stays in one directory.
|
||
stagingFile = "/run/${runtimeDir}/token.new";
|
||
# bao's stderr, in the unit's own `0700` directory rather than `/tmp`.
|
||
errFile = "/run/${runtimeDir}/bao.err";
|
||
|
||
# The store's address is the whole switch, as in ./bao.nix and
|
||
# ./queue-identity.nix.
|
||
configured = cfg.addr != null;
|
||
|
||
storeRetry = import ../host-modules/lib/store-retry.nix { };
|
||
in
|
||
{
|
||
options.services.hyperhive.agent.forge.tokenFile = lib.mkOption {
|
||
type = lib.types.str;
|
||
readOnly = true;
|
||
default = tokenFile;
|
||
description = ''
|
||
Path this agent's own forge token is fetched to. Read-only: it is a
|
||
fact about where `${unitName}.service` writes, not a knob.
|
||
|
||
🩸 A PATH and never a value. The file is `0400` to the agent user.
|
||
|
||
The file exists only once the swarm has minted a token for this agent
|
||
and the agent has a store identity to fetch it with. Until then every
|
||
consumer falls back to `$HYPERHIVE_STATE_DIR/forge-token`.
|
||
'';
|
||
};
|
||
|
||
config = lib.mkIf configured {
|
||
# Every unit and the bash-task runner (where `hive-forge` and `git` run)
|
||
# resolve the token through this, the same way they find
|
||
# `$HYPERHIVE_STATE_DIR`. A path, never the value.
|
||
systemd.globalEnvironment.HIVE_FORGE_TOKEN_FILE = tokenFile;
|
||
environment.variables.HIVE_FORGE_TOKEN_FILE = tokenFile;
|
||
|
||
systemd.services.${unitName} = {
|
||
description = "fetch this agent's own forge token from the secret store";
|
||
after = [
|
||
"network.target"
|
||
# Ordering only: that unit reports a store this agent cannot reach as
|
||
# itself, and should get to say so before this one reports a path it
|
||
# could not read.
|
||
"hive-agent-bao-identity.service"
|
||
];
|
||
before = [ "hive-forge-notify.service" ];
|
||
wantedBy = [ "multi-user.target" ];
|
||
path = [
|
||
pkgs.openbao
|
||
pkgs.coreutils
|
||
pkgs.diffutils
|
||
];
|
||
# ../host-modules/lib/store-retry.nix.
|
||
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
|
||
serviceConfig = storeRetry.serviceConfig // {
|
||
Type = "oneshot";
|
||
# Not `RemainAfterExit`: the timer below has to be able to start this
|
||
# unit again, and an active unit cannot be started.
|
||
# `RuntimeDirectoryPreserve` is what keeps the directory, and the token
|
||
# in it, alive between runs instead.
|
||
RemainAfterExit = false;
|
||
TimeoutStartSec = 30;
|
||
User = agentName;
|
||
Group = agentName;
|
||
RuntimeDirectory = runtimeDir;
|
||
RuntimeDirectoryMode = "0700";
|
||
RuntimeDirectoryPreserve = "yes";
|
||
UMask = "0377";
|
||
LoadCredential = [
|
||
certCredential
|
||
keyCredential
|
||
serverCaCredential
|
||
];
|
||
};
|
||
environment = {
|
||
BAO_ADDR = cfg.addr;
|
||
BAO_CLIENT_CERT = "%d/${certCredential}";
|
||
BAO_CLIENT_KEY = "%d/${keyCredential}";
|
||
};
|
||
script = ''
|
||
set -euo pipefail
|
||
|
||
# No identity delivered: ./bao.nix's check reports that; saying it
|
||
# twice adds nothing.
|
||
for id in ${lib.escapeShellArg certCredential} ${lib.escapeShellArg keyCredential}; do
|
||
if [ ! -s "$CREDENTIALS_DIRECTORY/$id" ]; then
|
||
echo "this agent has no store identity, so it cannot fetch its own forge token." >&2
|
||
exit 0
|
||
fi
|
||
if [ ! -r "$CREDENTIALS_DIRECTORY/$id" ]; then
|
||
echo "cannot read $CREDENTIALS_DIRECTORY/$id, so this agent cannot present its store identity." >&2
|
||
exit 1
|
||
fi
|
||
done
|
||
|
||
if [ -s "$CREDENTIALS_DIRECTORY/${serverCaCredential}" ]; then
|
||
export BAO_CACERT="$CREDENTIALS_DIRECTORY/${serverCaCredential}"
|
||
fi
|
||
|
||
# `UMask=0377` makes every file this script creates `0400`, so only
|
||
# the redirect that creates a file can write to it: `$err` is removed
|
||
# before each redirect into it.
|
||
err=${lib.escapeShellArg errFile}
|
||
trap 'rm -f "$err" ${lib.escapeShellArg stagingFile}' EXIT
|
||
|
||
rm -f "$err"
|
||
if ! BAO_TOKEN="$(bao login -method=cert -token-only 2>"$err")"; then
|
||
# The redirect creates `$err` before bao starts, so no file means
|
||
# bao never ran. bao prints `Code: <status>` only for an HTTP
|
||
# answer, and `remote error: tls:` only for an alert the store sent.
|
||
re='Code: ([0-9]{3})'
|
||
if [ ! -e "$err" ]; then
|
||
echo "could not create $err, so bao never ran and the store at $BAO_ADDR was not asked." >&2
|
||
elif [[ "$(<"$err")" =~ $re ]]; then
|
||
case "''${BASH_REMATCH[1]}" in
|
||
4*) echo "the swarm secret store at $BAO_ADDR refused this agent's certificate login with HTTP ''${BASH_REMATCH[1]}:" >&2 ;;
|
||
*) echo "the swarm secret store at $BAO_ADDR failed this agent's certificate login with HTTP ''${BASH_REMATCH[1]}:" >&2 ;;
|
||
esac
|
||
elif [[ "$(<"$err")" == *"remote error: tls:"* ]]; then
|
||
echo "the swarm secret store at $BAO_ADDR refused this agent's certificate in the TLS handshake:" >&2
|
||
else
|
||
echo "bao got no answer from the swarm secret store at $BAO_ADDR (network, DNS, or TLS on this side):" >&2
|
||
fi
|
||
if [ -s "$err" ]; then cat "$err" >&2; fi
|
||
exit 1
|
||
fi
|
||
export BAO_TOKEN
|
||
|
||
# 🩸 Degrades where ./bao.nix's check fails, because one policy stanza
|
||
# governs both reads: the one that let the login read `bao-mtls` covers
|
||
# this path too, so a refusal here is a token not minted yet. The
|
||
# file already in place, if any, is kept: a store that is briefly
|
||
# unreachable must not take a working token away.
|
||
#
|
||
# ⚠️ Written by redirect, never echoed: the field is the secret. The
|
||
# staging file is removed first so the redirect creates it — a file
|
||
# `UMask=0377` left behind is `0400` and could not be reopened for
|
||
# writing.
|
||
rm -f "$err" ${lib.escapeShellArg stagingFile}
|
||
if ! bao kv get -field=value ${lib.escapeShellArg tokenPath} > ${lib.escapeShellArg stagingFile} 2>"$err"; then
|
||
echo "no forge token at ${tokenPath} yet; consumers keep using the state-dir token if there is one." >&2
|
||
if [ -s "$err" ]; then cat "$err" >&2; fi
|
||
exit 0
|
||
fi
|
||
|
||
if cmp -s ${lib.escapeShellArg stagingFile} ${lib.escapeShellArg tokenFile}; then
|
||
echo "this agent's forge token at ${tokenPath} is unchanged."
|
||
exit 0
|
||
fi
|
||
mv -f ${lib.escapeShellArg stagingFile} ${lib.escapeShellArg tokenFile}
|
||
echo "fetched this agent's forge token from ${tokenPath}."
|
||
'';
|
||
};
|
||
|
||
# The controller re-checks every agent's token every five minutes; this
|
||
# picks up a rotation within about ten more.
|
||
systemd.timers.${unitName} = {
|
||
description = "re-fetch this agent's forge token from the secret store";
|
||
wantedBy = [ "timers.target" ];
|
||
timerConfig = {
|
||
OnUnitInactiveSec = "10min";
|
||
RandomizedDelaySec = "1min";
|
||
};
|
||
};
|
||
};
|
||
}
|