The swarm collector's journald receiver read only the units listed in `services.hyperhive.swarm.otel.journaldUnits`. A unit nobody listed never reached the store, and a misspelt entry shipped nothing without an error. The list existed to keep an operator's desktop session out of a store every swarm operator can read, but the receiver can only match positively, so the only way to express "not user sessions" was to name every service instead. The receiver now reads the whole host journal, and a new `filter/exclude-user-sessions` processor in the `logs/<swarm>` pipeline drops records whose `_SYSTEMD_SLICE` is `user-<uid>.slice` (session scopes and `user@<uid>.service`). The per-hive `logs/<hive>` pipelines carry agent-container journals only and get no filter. `journaldUnits` is removed with `mkRemovedOptionModule`, together with its non-empty assertion and the entry each host module added. The four module-eval membership checks go with it, replaced by one structural case in swarm-otel-core. Closes #3646
317 lines
14 KiB
Nix
317 lines
14 KiB
Nix
# OpenTelemetry wiring for an agent container: the per-agent options the
|
|
# meta flake injects (host-driven from `services.hyperhive.otel.*`), the
|
|
# environment every OTLP producer inside the container reads, and the
|
|
# collector that forwards this container's journal to the hive.
|
|
#
|
|
# Two directions, one endpoint. Metrics are PUSHED by processes in here
|
|
# that speak OTLP themselves; logs are not pushed by anyone, because
|
|
# nothing in a journal has an SDK. The forwarder is what turns the second
|
|
# into the first, and it has to run inside the container — a hive-level
|
|
# reader cannot see this journal at all (the host-side per-container
|
|
# directory is an id-mapped bind mount journald never writes into).
|
|
#
|
|
# Claude Code is one producer here, not the owner. `hive-metric` — and
|
|
# any future in-container exporter — reads the same endpoint, protocol
|
|
# and resource labels, so they are container environment rather than
|
|
# claude settings. Claude's own switches (its telemetry master flag,
|
|
# which signals it emits) stay in `claude-settings.nix`, which reads the
|
|
# options declared below.
|
|
{
|
|
lib,
|
|
pkgs,
|
|
config,
|
|
...
|
|
}:
|
|
let
|
|
cfg = config.services.hyperhive.agent.otel;
|
|
userName = config.services.hyperhive.agent.user.name;
|
|
# Hive/swarm display names, read from the per-agent options meta.rs
|
|
# renders (NOT from `environment.variables` — those carry the same
|
|
# names at *runtime* only, so reading them here silently yielded
|
|
# "unknown" on every agent while the process env held the right
|
|
# answer). `null` means the hive did not name itself; "unknown" is then
|
|
# an honest label rather than a guess.
|
|
hiveDisplayName =
|
|
if config.services.hyperhive.agent.hiveName == null then
|
|
"unknown"
|
|
else
|
|
config.services.hyperhive.agent.hiveName;
|
|
swarmDisplayName =
|
|
if config.services.hyperhive.agent.swarmName == null then
|
|
"unknown"
|
|
else
|
|
config.services.hyperhive.agent.swarmName;
|
|
|
|
# Resource labels every producer in this container stamps on what it
|
|
# emits. `service.name` names the container's role, not one binary
|
|
# inside it: claude's samples and `hive-metric`'s both come from this
|
|
# agent, and their metric names already tell them apart.
|
|
resourceAttributes =
|
|
"service.name=hyperhive-agent,agent=${userName},hive=${hiveDisplayName},swarm=${swarmDisplayName}"
|
|
+ lib.optionalString (cfg.extraResourceAttributes != "") ",${cfg.extraResourceAttributes}";
|
|
|
|
# The variables an OTEL SDK reads on its own, in any language, in any
|
|
# process here.
|
|
#
|
|
# There is no auth header among them, and no mechanism to add one. An
|
|
# agent exports to the hive's own collector, which is the only thing
|
|
# holding a credential for anything upstream; nothing an agent can read
|
|
# is a secret to the swarm. An earlier revision forwarded the
|
|
# operator's upstream token into this container and merged it into the
|
|
# agent's own `~/.claude/settings.json` — which handed every agent the
|
|
# hive's credential, and was removed with the direct-export path it
|
|
# served.
|
|
producerEnv = {
|
|
OTEL_EXPORTER_OTLP_ENDPOINT = cfg.endpoint;
|
|
OTEL_EXPORTER_OTLP_PROTOCOL = cfg.protocol;
|
|
# Force CUMULATIVE temporality — Claude Code defaults to DELTA,
|
|
# which Prometheus/Mimir-family backends (incl. grafana-lgtm)
|
|
# silently drop without a deltatocumulative processor.
|
|
OTEL_EXPORTER_OTLP_METRICS_TEMPORALITY_PREFERENCE = "cumulative";
|
|
OTEL_RESOURCE_ATTRIBUTES = resourceAttributes;
|
|
}
|
|
// lib.optionalAttrs (cfg.metricIntervalMs != null) {
|
|
OTEL_METRIC_EXPORT_INTERVAL = toString cfg.metricIntervalMs;
|
|
};
|
|
|
|
# The exporter's NAME picks the wire protocol, so `protocol` has to be
|
|
# read here as well as handed to the SDKs above — same expression as the
|
|
# swarm tier's upstream exporter in nix/host-modules/swarm-otel.nix.
|
|
logExporter = if cfg.protocol == "grpc" then "otlp" else "otlphttp";
|
|
in
|
|
{
|
|
# OTEL stats export is configured ONCE at host level via
|
|
# `services.hyperhive.otel.*` (see nix/host-modules/otel.nix) and
|
|
# injected into every agent's build by the meta-flake renderer
|
|
# (`hive-c0re/src/meta.rs::otel_config`). These per-agent options are
|
|
# the build-time implementation surface that injection writes into;
|
|
# they are not meant to be set directly in an agent.nix. Marked
|
|
# `internal` so the host option is the only documented operator knob.
|
|
options.services.hyperhive.agent.otel = {
|
|
enable = lib.mkOption {
|
|
type = lib.types.bool;
|
|
default = false;
|
|
internal = true;
|
|
description = ''
|
|
Export this agent's telemetry — Claude Code's stats (token
|
|
usage, cost, tool calls) and anything pushed with
|
|
`hive-metric` — to an OTLP endpoint. Each agent exports
|
|
directly to the hive's collector, so it keeps working even when
|
|
hive-c0re is down. Host-driven: set
|
|
`services.hyperhive.otel.enable` instead.
|
|
'';
|
|
};
|
|
|
|
endpoint = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
internal = true;
|
|
description = ''
|
|
OTLP collector endpoint, set as `OTEL_EXPORTER_OTLP_ENDPOINT`.
|
|
Host-driven via `services.hyperhive.otel.endpoint`.
|
|
'';
|
|
};
|
|
|
|
protocol = lib.mkOption {
|
|
type = lib.types.enum [
|
|
"http/protobuf"
|
|
"http/json"
|
|
"grpc"
|
|
];
|
|
default = "http/protobuf";
|
|
internal = true;
|
|
description = ''
|
|
OTLP wire protocol, set as `OTEL_EXPORTER_OTLP_PROTOCOL`.
|
|
Host-driven via `services.hyperhive.otel.protocol`.
|
|
'';
|
|
};
|
|
|
|
extraResourceAttributes = lib.mkOption {
|
|
type = lib.types.str;
|
|
default = "";
|
|
internal = true;
|
|
description = ''
|
|
Extra comma-separated entries appended to
|
|
`OTEL_RESOURCE_ATTRIBUTES` after the built-in
|
|
`service.name` / `agent` / `hive` / `swarm` labels.
|
|
Host-driven via `services.hyperhive.otel.extraResourceAttributes`.
|
|
'';
|
|
};
|
|
|
|
metricIntervalMs = lib.mkOption {
|
|
type = lib.types.nullOr lib.types.ints.positive;
|
|
default = null;
|
|
internal = true;
|
|
description = ''
|
|
Metric export interval in milliseconds, set as
|
|
`OTEL_METRIC_EXPORT_INTERVAL`. Null leaves the SDK default (60s
|
|
for Claude Code). Host-driven via
|
|
`services.hyperhive.otel.metricIntervalMs`.
|
|
'';
|
|
};
|
|
|
|
debug = lib.mkOption {
|
|
type = lib.types.bool;
|
|
default = false;
|
|
internal = true;
|
|
description = ''
|
|
Emit OTEL SDK diagnostics to stderr (`CLAUDE_CODE_OTEL_DIAG_STDERR=1`).
|
|
Host-driven via `services.hyperhive.otel.debug`.
|
|
'';
|
|
};
|
|
};
|
|
|
|
config = lib.mkIf cfg.enable {
|
|
# Two assignments, because they cover disjoint sets of processes and
|
|
# neither implies the other:
|
|
#
|
|
# - `systemd.globalEnvironment` is merged into the `Environment=`
|
|
# lines of every generated unit, so the harness, the
|
|
# bash/matrix/MCP daemons and everything the bash tool spawns
|
|
# beneath them all get it. This is the assignment that matters:
|
|
# those daemons are claude's *siblings*, not its children, so
|
|
# nothing shipped inside claude's own settings could ever reach
|
|
# them, and `hive-metric` invoked from a tool call exited with
|
|
# "OTEL_EXPORTER_OTLP_ENDPOINT not set".
|
|
# - `environment.variables` lands in `/etc/set-environment`, sourced
|
|
# by `/etc/profile` — login shells, i.e. `hivectl shell` and
|
|
# `hivectl choom`, which systemd does not start.
|
|
#
|
|
# `NIX_REMOTE` in `default.nix` is set both ways for this same
|
|
# reason. The difference is visible on any running agent:
|
|
# `HIVE_ASSETS_DIR` is an `environment.variables` entry and is absent
|
|
# from `systemctl show hive-bash-daemon.service -p Environment`,
|
|
# while `NIX_REMOTE` — the same value, also set globally — is there.
|
|
systemd.globalEnvironment = producerEnv;
|
|
environment.variables = producerEnv;
|
|
|
|
# The journal this container writes is reachable from inside it and
|
|
# from nowhere else, which is why the forwarder runs here rather than
|
|
# the hive reading the container's journal from outside: the host-side
|
|
# per-container directory is an id-mapped bind mount and journald
|
|
# writes nothing into it.
|
|
assertions = [
|
|
{
|
|
# Sibling of the swarm collector's storage assertion in
|
|
# nix/host-modules/swarm-otel.nix, and it exists because the
|
|
# failure is silent at every layer: with a volatile journal the
|
|
# receiver below finds an empty directory, reads nothing, and the
|
|
# collector starts clean and stays healthy forever.
|
|
assertion =
|
|
!(lib.elem config.services.journald.storage [
|
|
"volatile"
|
|
"none"
|
|
]);
|
|
message = ''
|
|
services.hyperhive.agent.otel.enable is on for agent ${userName}, but
|
|
services.journald.storage is
|
|
"${config.services.journald.storage}" in this container.
|
|
|
|
The log forwarder reads /var/log/journal, which journald only
|
|
writes when it stores persistently: with "volatile" the journal
|
|
lives in /run/log/journal and with "none" there is none at all.
|
|
Either way the forwarder would ship nothing while looking
|
|
healthy.
|
|
'';
|
|
}
|
|
];
|
|
|
|
services.opentelemetry-collector = {
|
|
enable = true;
|
|
# Contrib, and not a preference: `journald` is a contrib receiver.
|
|
# The upstream default build has no way to read a journal at all.
|
|
package = pkgs.opentelemetry-collector-contrib;
|
|
# Runs `otelcol validate` at build time. ⚠️ A parser, not a wiring
|
|
# check — it accepts a pipeline naming a component the build lacks,
|
|
# and the collector then dies at startup. Same caveat as both other
|
|
# tiers; see nix/host-modules/otel.nix for the measurement.
|
|
validateConfigFile = true;
|
|
settings = {
|
|
# `journalctl --follow --lines=0` (what this receiver runs under
|
|
# the hood) ships only what's written *after* it starts — every
|
|
# collector start has a silent gap at the front. `storage:` below
|
|
# persists the read cursor, so a restart resumes from where it
|
|
# left off instead of re-opening that gap; `start_at` stays `end`
|
|
# (its default) because a cursor already covers every run after
|
|
# the first, and `beginning` without one would re-ship the whole
|
|
# journal on every restart. The cursor lives in the unit's own
|
|
# `StateDirectory` (nixpkgs' module already sets one, `%S`, below
|
|
# as `WorkingDirectory` too) — container-lifetime, not the
|
|
# host-persisted state bind mount: it indexes /var/log/journal,
|
|
# itself fresh on every container recreate, so the two must share
|
|
# a lifetime or the cursor outlives the journal it points into.
|
|
# `create_directory` matters here for a build-time-vs-runtime
|
|
# reason, not a convenience one: `validateConfigFile` above runs
|
|
# `otelcol validate` in the nix build sandbox, before systemd's
|
|
# `StateDirectory=` has ever created this path. Without it the
|
|
# validator refuses with "directory must exist" and the build fails.
|
|
extensions.file_storage = {
|
|
directory = "/var/lib/opentelemetry-collector";
|
|
create_directory = true;
|
|
};
|
|
|
|
receivers.journald = {
|
|
# ⚠️ STATED, and it must stay stated: the receiver's own default
|
|
# is the RUNTIME journal (`/run/log/journal`), which in an agent
|
|
# container is empty — journald stores persistently here, so
|
|
# every entry is under /var/log/journal. Dropping this line
|
|
# leaves a collector that validates, starts, reports healthy and
|
|
# forwards nothing.
|
|
directory = "/var/log/journal";
|
|
storage = "file_storage";
|
|
# No `units` allowlist, and no user-session filter like the swarm
|
|
# tier's: a container's journal is the harness and what the
|
|
# harness spawns, so there is no foreign traffic to filter out and
|
|
# an allowlist would only be a list to forget to update.
|
|
|
|
# journald's PRIORITY carries the level every line already has;
|
|
# without this the record reaches VictoriaLogs with
|
|
# `severity_text: Unspecified` and the store cannot tell an error
|
|
# from a debug line. The table — and the inverted direction that
|
|
# makes it worth a file of its own — lives in
|
|
# ../journald-severity.nix, shared with the swarm tier's receiver.
|
|
operators = import ../journald-severity.nix;
|
|
};
|
|
|
|
# Identity the hop above cannot supply. A host-side reader can say
|
|
# which machine a line came from; only a collector running as the
|
|
# agent can say which agent. `hive` and `swarm` are deliberately
|
|
# NOT stamped here even though the env above carries them for
|
|
# in-process producers: the swarm tier upserts `hive` from the
|
|
# receiver that accepted the sample, precisely so the label comes
|
|
# from something the sender cannot write.
|
|
processors.resource.attributes = [
|
|
{
|
|
key = "service.name";
|
|
value = "hyperhive-agent";
|
|
action = "upsert";
|
|
}
|
|
{
|
|
key = "agent";
|
|
value = userName;
|
|
action = "upsert";
|
|
}
|
|
];
|
|
|
|
# The hive's collector, at the same base address every in-process
|
|
# producer already exports to. `endpoint` is a BASE the exporter
|
|
# appends `/v1/logs` to — the path an OTLP/HTTP receiver serves.
|
|
exporters.${logExporter} = {
|
|
endpoint = cfg.endpoint;
|
|
}
|
|
// lib.optionalAttrs (cfg.protocol == "http/json") { encoding = "json"; };
|
|
|
|
service.pipelines.logs = {
|
|
receivers = [ "journald" ];
|
|
processors = [ "resource" ];
|
|
exporters = [ logExporter ];
|
|
};
|
|
|
|
# An extension configured but not listed here is INERT — the
|
|
# journald receiver's `storage: file_storage` above would name a
|
|
# component the collector never starts.
|
|
service.extensions = [ "file_storage" ];
|
|
};
|
|
};
|
|
};
|
|
}
|