diff --git a/nix/host-modules/swarm-grafana.nix b/nix/host-modules/swarm-grafana.nix index dce69eb4..f5ecd195 100644 --- a/nix/host-modules/swarm-grafana.nix +++ b/nix/host-modules/swarm-grafana.nix @@ -142,6 +142,37 @@ let annotations = { inherit summary; }; }; + # `logCountRule` inverted: fires while the count is below one. `expr` must + # be an ungrouped `stats count()`, which answers 0 rather than no row when + # nothing matched; a grouped count drops the group instead, and a dropped + # instance resolves rather than fires. No data is `Alerting` so a query + # that returns nothing at all still fires. + logAbsenceRule = + args: + let + rule = logCountRule args; + in + rule + // { + data = map ( + query: + if query.refId == "C" then + lib.recursiveUpdate query { + model.conditions = [ + { + evaluator = { + type = "lt"; + params = [ 1 ]; + }; + } + ]; + } + else + query + ) rule.data; + noDataState = "Alerting"; + }; + # The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where # a real deployment needs the uids above. They are substituted here rather # than committed with the literals so the single binding stays single. @@ -1080,7 +1111,64 @@ in for = "5m"; summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue"; }) - ]; + # Logged once per agent process: the connection is made + # once and never retried, so the agent stays unconnected + # until it restarts. The window is how long the rule + # stays firing after that one line. + (logCountRule { + uid = "hh-swarm-queue-connect-failed"; + title = "swarm queue connect failed"; + windowSeconds = 3600; + expr = ''_time:1h _msg:"swarm queue connect failed; this agent publishes nothing upward" | stats by (hive, agent) count() as failures''; + for = "0s"; + summary = "{{ $labels.hive }}/{{ $labels.agent }} could not connect to the swarm queue and publishes nothing upward until it restarts"; + }) + (logCountRule { + uid = "hh-swarm-agent-state-publish-skipped"; + title = "swarm agent state publish skipped"; + windowSeconds = 900; + expr = ''_time:15m _msg:"swarm agent state: publish skipped" | stats by (hive, agent) count() as skipped''; + for = "5m"; + summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping agent state updates: the harness is not connected to the queue"; + }) + # Logged on every connection attempt, so it repeats under + # the reconnect backoff for as long as the path is empty. + (logCountRule { + uid = "hh-swarm-queue-credential-missing"; + title = "swarm queue credential missing"; + windowSeconds = 900; + expr = ''_time:15m _msg:"nothing is stored at this agent's queue credential path; its queue connection retries under backoff" | stats by (hive, agent) count() as missing''; + for = "5m"; + summary = "{{ $labels.hive }}/{{ $labels.agent }} finds nothing at its queue credential path and cannot connect to the queue"; + }) + # Same cadence as the forge rule: one pass every 5m + # (`RECONCILE_INTERVAL` in swarm-controller's + # agent_renewal.rs), one line per failed pass, so one + # failed pass never fires and about three in a row do. + (logCountRule { + uid = "hh-agent-credential-renewal-failed"; + title = "agent credential renewal failing"; + windowSeconds = 600; + expr = ''_time:10m _msg:"agent credential renewal: pass failed; retrying next tick" | stats count() as failures''; + for = "15m"; + summary = "swarm-controller has failed the agent credential renewal pass on consecutive ticks"; + }) + ] + # One rule per declared hive rather than one grouped by + # `hive`, so a hive that sends nothing at all still has + # an instance to fire. `hive` is stamped by the swarm + # collector from the same `swarm.hives` key. + ++ lib.mapAttrsToList ( + name: _: + logAbsenceRule { + uid = "hh-logs-absent-${name}"; + title = "no log records from ${name}"; + windowSeconds = 600; + expr = ''_time:10m hive:="${name}" | stats count() as records''; + for = "5m"; + summary = "hive ${name} has sent no log records to VictoriaLogs for at least 10m"; + } + ) hyperhiveCfg.swarm.hives; } ]; };