The journald receiver runs with --lines=0, so every collector start only ships what's written after it starts, and a restart silently loses whatever landed while it was down. Point it at a file_storage extension so the read cursor survives a restart; start_at stays at its 'end' default since the cursor now covers everything after the first run. Refs #3818 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
297 lines
13 KiB
Nix
297 lines
13 KiB
Nix
# OpenTelemetry wiring for an agent container: the per-agent options the
|
|
# meta flake injects (host-driven from `services.hyperhive.otel.*`), the
|
|
# environment every OTLP producer inside the container reads, and the
|
|
# collector that forwards this container's journal to the hive.
|
|
#
|
|
# Two directions, one endpoint. Metrics are PUSHED by processes in here
|
|
# that speak OTLP themselves; logs are not pushed by anyone, because
|
|
# nothing in a journal has an SDK. The forwarder is what turns the second
|
|
# into the first, and it has to run inside the container — a hive-level
|
|
# reader cannot see this journal at all (the host-side per-container
|
|
# directory is an id-mapped bind mount journald never writes into).
|
|
#
|
|
# Claude Code is one producer here, not the owner. `hive-metric` — and
|
|
# any future in-container exporter — reads the same endpoint, protocol
|
|
# and resource labels, so they are container environment rather than
|
|
# claude settings. Claude's own switches (its telemetry master flag,
|
|
# which signals it emits) stay in `claude-settings.nix`, which reads the
|
|
# options declared below.
|
|
{
|
|
lib,
|
|
pkgs,
|
|
config,
|
|
...
|
|
}:
|
|
let
|
|
cfg = config.hyperhive.otel;
|
|
userName = config.hyperhive.user.name;
|
|
# Hive/swarm display names, read from the per-agent options meta.rs
|
|
# renders (NOT from `environment.variables` — those carry the same
|
|
# names at *runtime* only, so reading them here silently yielded
|
|
# "unknown" on every agent while the process env held the right
|
|
# answer). `null` means the hive did not name itself; "unknown" is then
|
|
# an honest label rather than a guess.
|
|
hiveDisplayName =
|
|
if config.hyperhive.hiveName == null then "unknown" else config.hyperhive.hiveName;
|
|
swarmDisplayName =
|
|
if config.hyperhive.swarmName == null then "unknown" else config.hyperhive.swarmName;
|
|
|
|
# Resource labels every producer in this container stamps on what it
|
|
# emits. `service.name` names the container's role, not one binary
|
|
# inside it: claude's samples and `hive-metric`'s both come from this
|
|
# agent, and their metric names already tell them apart.
|
|
resourceAttributes =
|
|
"service.name=hyperhive-agent,agent=${userName},hive=${hiveDisplayName},swarm=${swarmDisplayName}"
|
|
+ lib.optionalString (cfg.extraResourceAttributes != "") ",${cfg.extraResourceAttributes}";
|
|
|
|
# The variables an OTEL SDK reads on its own, in any language, in any
|
|
# process here.
|
|
#
|
|
# There is no auth header among them, and no mechanism to add one. An
|
|
# agent exports to the hive's own collector, which is the only thing
|
|
# holding a credential for anything upstream; nothing an agent can read
|
|
# is a secret to the swarm. An earlier revision forwarded the
|
|
# operator's upstream token into this container and merged it into the
|
|
# agent's own `~/.claude/settings.json` — which handed every agent the
|
|
# hive's credential, and was removed with the direct-export path it
|
|
# served.
|
|
producerEnv = {
|
|
OTEL_EXPORTER_OTLP_ENDPOINT = cfg.endpoint;
|
|
OTEL_EXPORTER_OTLP_PROTOCOL = cfg.protocol;
|
|
# Force CUMULATIVE temporality — Claude Code defaults to DELTA,
|
|
# which Prometheus/Mimir-family backends (incl. grafana-lgtm)
|
|
# silently drop without a deltatocumulative processor.
|
|
OTEL_EXPORTER_OTLP_METRICS_TEMPORALITY_PREFERENCE = "cumulative";
|
|
OTEL_RESOURCE_ATTRIBUTES = resourceAttributes;
|
|
}
|
|
// lib.optionalAttrs (cfg.metricIntervalMs != null) {
|
|
OTEL_METRIC_EXPORT_INTERVAL = toString cfg.metricIntervalMs;
|
|
};
|
|
|
|
# The exporter's NAME picks the wire protocol, so `protocol` has to be
|
|
# read here as well as handed to the SDKs above — same expression as the
|
|
# swarm tier's upstream exporter in nix/host-modules/swarm-otel.nix.
|
|
logExporter = if cfg.protocol == "grpc" then "otlp" else "otlphttp";
|
|
in
|
|
{
|
|
# OTEL stats export is configured ONCE at host level via
|
|
# `services.hyperhive.otel.*` (see nix/host-modules/otel.nix) and
|
|
# injected into every agent's build by the meta-flake renderer
|
|
# (`hive-c0re/src/meta.rs::otel_config`). These per-agent options are
|
|
# the build-time implementation surface that injection writes into;
|
|
# they are not meant to be set directly in an agent.nix. Marked
|
|
# `internal` so the host option is the only documented operator knob.
|
|
options.hyperhive.otel = {
|
|
enable = lib.mkOption {
|
|
type = lib.types.bool;
|
|
default = false;
|
|
internal = true;
|
|
description = ''
|
|
Export this agent's telemetry — Claude Code's stats (token
|
|
usage, cost, tool calls) and anything pushed with
|
|
`hive-metric` — to an OTLP endpoint. Each agent exports
|
|
directly to the hive's collector, so it keeps working even when
|
|
hive-c0re is down. Host-driven: set
|
|
`services.hyperhive.otel.enable` instead.
|
|
'';
|
|
};
|
|
|
|
endpoint = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
internal = true;
|
|
description = ''
|
|
OTLP collector endpoint, set as `OTEL_EXPORTER_OTLP_ENDPOINT`.
|
|
Host-driven via `services.hyperhive.otel.endpoint`.
|
|
'';
|
|
};
|
|
|
|
protocol = lib.mkOption {
|
|
type = lib.types.enum [
|
|
"http/protobuf"
|
|
"http/json"
|
|
"grpc"
|
|
];
|
|
default = "http/protobuf";
|
|
internal = true;
|
|
description = ''
|
|
OTLP wire protocol, set as `OTEL_EXPORTER_OTLP_PROTOCOL`.
|
|
Host-driven via `services.hyperhive.otel.protocol`.
|
|
'';
|
|
};
|
|
|
|
extraResourceAttributes = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
internal = true;
|
|
description = ''
|
|
Extra comma-separated entries appended to
|
|
`OTEL_RESOURCE_ATTRIBUTES` after the built-in
|
|
`service.name` / `agent` / `hive` / `swarm` labels.
|
|
Host-driven via `services.hyperhive.otel.extraResourceAttributes`.
|
|
'';
|
|
};
|
|
|
|
metricIntervalMs = lib.mkOption {
|
|
type = lib.types.nullOr lib.types.ints.positive;
|
|
default = null;
|
|
internal = true;
|
|
description = ''
|
|
Metric export interval in milliseconds, set as
|
|
`OTEL_METRIC_EXPORT_INTERVAL`. Null leaves the SDK default (60s
|
|
for Claude Code). Host-driven via
|
|
`services.hyperhive.otel.metricIntervalMs`.
|
|
'';
|
|
};
|
|
|
|
debug = lib.mkOption {
|
|
type = lib.types.bool;
|
|
default = false;
|
|
internal = true;
|
|
description = ''
|
|
Emit OTEL SDK diagnostics to stderr (`CLAUDE_CODE_OTEL_DIAG_STDERR=1`).
|
|
Host-driven via `services.hyperhive.otel.debug`.
|
|
'';
|
|
};
|
|
};
|
|
|
|
config = lib.mkIf cfg.enable {
|
|
# Two assignments, because they cover disjoint sets of processes and
|
|
# neither implies the other:
|
|
#
|
|
# - `systemd.globalEnvironment` is merged into the `Environment=`
|
|
# lines of every generated unit, so the harness, the
|
|
# bash/matrix/MCP daemons and everything the bash tool spawns
|
|
# beneath them all get it. This is the assignment that matters:
|
|
# those daemons are claude's *siblings*, not its children, so
|
|
# nothing shipped inside claude's own settings could ever reach
|
|
# them, and `hive-metric` invoked from a tool call exited with
|
|
# "OTEL_EXPORTER_OTLP_ENDPOINT not set".
|
|
# - `environment.variables` lands in `/etc/set-environment`, sourced
|
|
# by `/etc/profile` — login shells, i.e. `hivectl shell` and
|
|
# `hivectl choom`, which systemd does not start.
|
|
#
|
|
# `NIX_REMOTE` in `default.nix` is set both ways for this same
|
|
# reason. The difference is visible on any running agent:
|
|
# `HIVE_ASSETS_DIR` is an `environment.variables` entry and is absent
|
|
# from `systemctl show hive-bash-daemon.service -p Environment`,
|
|
# while `NIX_REMOTE` — the same value, also set globally — is there.
|
|
systemd.globalEnvironment = producerEnv;
|
|
environment.variables = producerEnv;
|
|
|
|
# The journal this container writes is reachable from inside it and
|
|
# from nowhere else, which is why the forwarder runs here rather than
|
|
# the hive reading the container's journal from outside: the host-side
|
|
# per-container directory is an id-mapped bind mount and journald
|
|
# writes nothing into it.
|
|
assertions = [
|
|
{
|
|
# Sibling of the swarm collector's storage assertion in
|
|
# nix/host-modules/swarm-otel.nix, and it exists because the
|
|
# failure is silent at every layer: with a volatile journal the
|
|
# receiver below finds an empty directory, reads nothing, and the
|
|
# collector starts clean and stays healthy forever.
|
|
assertion =
|
|
!(lib.elem config.services.journald.storage [
|
|
"volatile"
|
|
"none"
|
|
]);
|
|
message = ''
|
|
hyperhive.otel.enable is on for agent ${userName}, but
|
|
services.journald.storage is
|
|
"${config.services.journald.storage}" in this container.
|
|
|
|
The log forwarder reads /var/log/journal, which journald only
|
|
writes when it stores persistently: with "volatile" the journal
|
|
lives in /run/log/journal and with "none" there is none at all.
|
|
Either way the forwarder would ship nothing while looking
|
|
healthy.
|
|
'';
|
|
}
|
|
];
|
|
|
|
services.opentelemetry-collector = {
|
|
enable = true;
|
|
# Contrib, and not a preference: `journald` is a contrib receiver.
|
|
# The upstream default build has no way to read a journal at all.
|
|
package = pkgs.opentelemetry-collector-contrib;
|
|
# Runs `otelcol validate` at build time. ⚠️ A parser, not a wiring
|
|
# check — it accepts a pipeline naming a component the build lacks,
|
|
# and the collector then dies at startup. Same caveat as both other
|
|
# tiers; see nix/host-modules/otel.nix for the measurement.
|
|
validateConfigFile = true;
|
|
settings = {
|
|
# `journalctl --follow --lines=0` (what this receiver runs under
|
|
# the hood) ships only what's written *after* it starts — every
|
|
# collector start has a silent gap at the front. `storage:` below
|
|
# persists the read cursor, so a restart resumes from where it
|
|
# left off instead of re-opening that gap; `start_at` stays `end`
|
|
# (its default) because a cursor already covers every run after
|
|
# the first, and `beginning` without one would re-ship the whole
|
|
# journal on every restart. The cursor lives in the unit's own
|
|
# `StateDirectory` (nixpkgs' module already sets one, `%S`, below
|
|
# as `WorkingDirectory` too) — container-lifetime, not the
|
|
# host-persisted state bind mount: it indexes /var/log/journal,
|
|
# itself fresh on every container recreate, so the two must share
|
|
# a lifetime or the cursor outlives the journal it points into.
|
|
extensions.file_storage.directory = "/var/lib/opentelemetry-collector";
|
|
|
|
receivers.journald = {
|
|
# ⚠️ STATED, and it must stay stated: the receiver's own default
|
|
# is the RUNTIME journal (`/run/log/journal`), which in an agent
|
|
# container is empty — journald stores persistently here, so
|
|
# every entry is under /var/log/journal. Dropping this line
|
|
# leaves a collector that validates, starts, reports healthy and
|
|
# forwards nothing.
|
|
directory = "/var/log/journal";
|
|
storage = "file_storage";
|
|
# No `units` allowlist, unlike the swarm tier's receiver. That
|
|
# one needs one because the host's journal also holds an
|
|
# operator's own session; a container's journal is the harness
|
|
# and what the harness spawns, so there is no foreign traffic to
|
|
# filter out and an allowlist would only be a list to forget to
|
|
# update.
|
|
};
|
|
|
|
# Identity the hop above cannot supply. A host-side reader can say
|
|
# which machine a line came from; only a collector running as the
|
|
# agent can say which agent. `hive` and `swarm` are deliberately
|
|
# NOT stamped here even though the env above carries them for
|
|
# in-process producers: the swarm tier upserts `hive` from the
|
|
# receiver that accepted the sample, precisely so the label comes
|
|
# from something the sender cannot write.
|
|
processors.resource.attributes = [
|
|
{
|
|
key = "service.name";
|
|
value = "hyperhive-agent";
|
|
action = "upsert";
|
|
}
|
|
{
|
|
key = "agent";
|
|
value = userName;
|
|
action = "upsert";
|
|
}
|
|
];
|
|
|
|
# The hive's collector, at the same base address every in-process
|
|
# producer already exports to. `endpoint` is a BASE the exporter
|
|
# appends `/v1/logs` to — the path an OTLP/HTTP receiver serves.
|
|
exporters.${logExporter} = {
|
|
endpoint = cfg.endpoint;
|
|
}
|
|
// lib.optionalAttrs (cfg.protocol == "http/json") { encoding = "json"; };
|
|
|
|
service.pipelines.logs = {
|
|
receivers = [ "journald" ];
|
|
processors = [ "resource" ];
|
|
exporters = [ logExporter ];
|
|
};
|
|
|
|
# An extension configured but not listed here is INERT — the
|
|
# journald receiver's `storage: file_storage` above would name a
|
|
# component the collector never starts.
|
|
service.extensions = [ "file_storage" ];
|
|
};
|
|
};
|
|
};
|
|
}
|