Watch
0
0
Fork
You've already forked hyperhive
0
hyperhive/nix/agent-modules/forge-token.nix
atlas b3b42d3279 credential units: 24h retry shape; start a failed nginx when the cert lands
Six credential-fetch units retried 4 times at 15s, so an apply during
which the store or gateway was down for more than about a minute left
them in start-limit-hit, and nothing started them again once the store
came back. The swarm-services leaf could also land after nginx had
already given up on it, and the hook that propagates a new leaf only
reloaded a running nginx, so a stopped one stayed down until a second
apply.

- nix/host-modules/lib/store-retry.nix: the 2880 x 30s / 25h window
  shape swarm-services-cert already had, as one attrset.
- swarm-services-cert, swarm-bao-otel-oidc, swarm-bao-forwarder-oidc,
  swarm-bao-matrix-token, swarm-bao-queue-agent, swarm-bao-grafana-oidc,
  hive-agent-bao-identity and hive-agent-forge-token use it.
  queue-identity.nix no longer has a fetch unit (ccb5bd3b), and
  forge-token.nix is a fetch unit with the same short budget that was
  added after the census in #4662.
- The swarm-services-cert propagation hook now reset-fails and starts
  (--no-block) a loaded nginx that is not active; an active nginx keeps
  the re-import + reload.
- module-eval-bao-grants: one case pinning the shape on every host-side
  fetch unit, swarm-services-cert included.

Refs #4662
2026-09-30 07:45:47 +02:00

208 lines
8.6 KiB
Nix
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# This agent's own forge access token, fetched from the swarm secret store by
# the agent itself.
#
# `swarm-controller` mints the token with the forge's admin API and writes it
# to `swarm/agents/<agent>/forge-token`
# (`swarm_secret_client::forge::agent_token_path`); no hive is in that chain.
# This unit logs in to the store with the certificate ./bao.nix already proves
# it can log in with, and reads its own path. A hive handing the token over
# would be the hive reading a secret on the agent's behalf.
#
# The token rotates: the controller replaces it when the forge's copy stops
# matching the stored one. So the unit is re-run by a timer, and it swaps the
# file in by rename, only when the value changed, so a reader never sees half
# a token and a watcher on the file (forge-avatar-sync.path) fires only on a
# real change.
#
# Consumers read `services.hyperhive.agent.forge.tokenFile` first and fall back
# to `<state>/forge-token`, the file the hive wrote before this existed.
{
pkgs,
lib,
config,
...
}:
let
cfg = config.services.hyperhive.agent.bao;
# The name `swarm-controller` minted the token under — see ./queue-identity.nix.
agentName = config.services.hyperhive.agent.user.name;
# The same three ids ./bao.nix and ./queue-identity.nix load.
certCredential = "hive-agent-bao-cert";
keyCredential = "hive-agent-bao-key";
serverCaCredential = "hive-agent-bao-server-ca";
unitName = "hive-agent-forge-token";
# The nix half of `swarm_secret_client::forge::agent_token_path` plus
# `path::MOUNT`.
tokenPath = "secret/swarm/agents/${agentName}/forge-token";
runtimeDir = unitName;
tokenFile = "/run/${runtimeDir}/token";
# Beside the token, so the rename that replaces it stays in one directory.
stagingFile = "/run/${runtimeDir}/token.new";
# bao's stderr, in the unit's own `0700` directory rather than `/tmp`.
errFile = "/run/${runtimeDir}/bao.err";
# The store's address is the whole switch, as in ./bao.nix and
# ./queue-identity.nix.
configured = cfg.addr != null;
storeRetry = import ../host-modules/lib/store-retry.nix { };
in
{
options.services.hyperhive.agent.forge.tokenFile = lib.mkOption {
type = lib.types.str;
readOnly = true;
default = tokenFile;
description = ''
Path this agent's own forge token is fetched to. Read-only: it is a
fact about where `${unitName}.service` writes, not a knob.
🩸 A PATH and never a value. The file is `0400` to the agent user.
The file exists only once the swarm has minted a token for this agent
and the agent has a store identity to fetch it with. Until then every
consumer falls back to `$HYPERHIVE_STATE_DIR/forge-token`.
'';
};
config = lib.mkIf configured {
# Every unit and the bash-task runner (where `hive-forge` and `git` run)
# resolve the token through this, the same way they find
# `$HYPERHIVE_STATE_DIR`. A path, never the value.
systemd.globalEnvironment.HIVE_FORGE_TOKEN_FILE = tokenFile;
environment.variables.HIVE_FORGE_TOKEN_FILE = tokenFile;
systemd.services.${unitName} = {
description = "fetch this agent's own forge token from the secret store";
after = [
"network.target"
# Ordering only: that unit reports a store this agent cannot reach as
# itself, and should get to say so before this one reports a path it
# could not read.
"hive-agent-bao-identity.service"
];
before = [ "hive-forge-notify.service" ];
wantedBy = [ "multi-user.target" ];
path = [
pkgs.openbao
pkgs.coreutils
pkgs.diffutils
];
# ../host-modules/lib/store-retry.nix.
inherit (storeRetry) startLimitBurst startLimitIntervalSec;
serviceConfig = storeRetry.serviceConfig // {
Type = "oneshot";
# Not `RemainAfterExit`: the timer below has to be able to start this
# unit again, and an active unit cannot be started.
# `RuntimeDirectoryPreserve` is what keeps the directory, and the token
# in it, alive between runs instead.
RemainAfterExit = false;
TimeoutStartSec = 30;
User = agentName;
Group = agentName;
RuntimeDirectory = runtimeDir;
RuntimeDirectoryMode = "0700";
RuntimeDirectoryPreserve = "yes";
UMask = "0377";
LoadCredential = [
certCredential
keyCredential
serverCaCredential
];
};
environment = {
BAO_ADDR = cfg.addr;
BAO_CLIENT_CERT = "%d/${certCredential}";
BAO_CLIENT_KEY = "%d/${keyCredential}";
};
script = ''
set -euo pipefail
# No identity delivered: ./bao.nix's check reports that; saying it
# twice adds nothing.
for id in ${lib.escapeShellArg certCredential} ${lib.escapeShellArg keyCredential}; do
if [ ! -s "$CREDENTIALS_DIRECTORY/$id" ]; then
echo "this agent has no store identity, so it cannot fetch its own forge token." >&2
exit 0
fi
if [ ! -r "$CREDENTIALS_DIRECTORY/$id" ]; then
echo "cannot read $CREDENTIALS_DIRECTORY/$id, so this agent cannot present its store identity." >&2
exit 1
fi
done
if [ -s "$CREDENTIALS_DIRECTORY/${serverCaCredential}" ]; then
export BAO_CACERT="$CREDENTIALS_DIRECTORY/${serverCaCredential}"
fi
# `UMask=0377` makes every file this script creates `0400`, so only
# the redirect that creates a file can write to it: `$err` is removed
# before each redirect into it.
err=${lib.escapeShellArg errFile}
trap 'rm -f "$err" ${lib.escapeShellArg stagingFile}' EXIT
rm -f "$err"
if ! BAO_TOKEN="$(bao login -method=cert -token-only 2>"$err")"; then
# The redirect creates `$err` before bao starts, so no file means
# bao never ran. bao prints `Code: <status>` only for an HTTP
# answer, and `remote error: tls:` only for an alert the store sent.
re='Code: ([0-9]{3})'
if [ ! -e "$err" ]; then
echo "could not create $err, so bao never ran and the store at $BAO_ADDR was not asked." >&2
elif [[ "$(<"$err")" =~ $re ]]; then
case "''${BASH_REMATCH[1]}" in
4*) echo "the swarm secret store at $BAO_ADDR refused this agent's certificate login with HTTP ''${BASH_REMATCH[1]}:" >&2 ;;
*) echo "the swarm secret store at $BAO_ADDR failed this agent's certificate login with HTTP ''${BASH_REMATCH[1]}:" >&2 ;;
esac
elif [[ "$(<"$err")" == *"remote error: tls:"* ]]; then
echo "the swarm secret store at $BAO_ADDR refused this agent's certificate in the TLS handshake:" >&2
else
echo "bao got no answer from the swarm secret store at $BAO_ADDR (network, DNS, or TLS on this side):" >&2
fi
if [ -s "$err" ]; then cat "$err" >&2; fi
exit 1
fi
export BAO_TOKEN
# 🩸 Degrades where ./bao.nix's check fails, because one policy stanza
# governs both reads: the one that let the login read `bao-mtls` covers
# this path too, so a refusal here is a token not minted yet. The
# file already in place, if any, is kept: a store that is briefly
# unreachable must not take a working token away.
#
# ⚠️ Written by redirect, never echoed: the field is the secret. The
# staging file is removed first so the redirect creates it — a file
# `UMask=0377` left behind is `0400` and could not be reopened for
# writing.
rm -f "$err" ${lib.escapeShellArg stagingFile}
if ! bao kv get -field=value ${lib.escapeShellArg tokenPath} > ${lib.escapeShellArg stagingFile} 2>"$err"; then
echo "no forge token at ${tokenPath} yet; consumers keep using the state-dir token if there is one." >&2
if [ -s "$err" ]; then cat "$err" >&2; fi
exit 0
fi
if cmp -s ${lib.escapeShellArg stagingFile} ${lib.escapeShellArg tokenFile}; then
echo "this agent's forge token at ${tokenPath} is unchanged."
exit 0
fi
mv -f ${lib.escapeShellArg stagingFile} ${lib.escapeShellArg tokenFile}
echo "fetched this agent's forge token from ${tokenPath}."
'';
};
# The controller re-checks every agent's token every five minutes; this
# picks up a rotation within about ten more.
systemd.timers.${unitName} = {
description = "re-fetch this agent's forge token from the secret store";
wantedBy = [ "timers.target" ];
timerConfig = {
OnUnitInactiveSec = "10min";
RandomizedDelaySec = "1min";
};
};
};
}