Watch
0
0
Fork
You've already forked hyperhive
0

swarm-grafana: alert rules for queue, renewal and missing-log WARNs

Adds rules to the `hyperhive` alerting folder next to the two from #4811:

- swarm queue connect failed (per hive/agent, 1h window, fires at once:
  the line is logged once per agent process and the connection is never
  retried, so the window is how long the rule stays firing)
- swarm agent state publish skipped (per hive/agent, 15m, for 5m, as the
  swarm terminal rule)
- swarm queue credential missing (per hive/agent, 15m, for 5m)
- agent credential renewal failing (10m, for 15m, as the forge reconcile
  rule: same 5m pass cadence, one line per failed pass)
- no log records from <hive>, one per `swarm.hives` entry: fires when
  the hive's ungrouped count over 10m is below 1, and on no data. A new
  `logAbsenceRule` helper wraps `logCountRule` with the inverted
  threshold and noDataState = Alerting.

No contact point and no notification policy: the rules show under
Alerting -> Alert rules and are delivered nowhere.

Refs #4717
Refs #3900
This commit is contained in:
atlas 2026-09-30 15:00:28 +02:00
commit b9667d5975

View file

@ -142,6 +142,37 @@ let
annotations = { inherit summary; }; annotations = { inherit summary; };
}; };
# `logCountRule` inverted: fires while the count is below one. `expr` must
# be an ungrouped `stats count()`, which answers 0 rather than no row when
# nothing matched; a grouped count drops the group instead, and a dropped
# instance resolves rather than fires. No data is `Alerting` so a query
# that returns nothing at all still fires.
logAbsenceRule =
args:
let
rule = logCountRule args;
in
rule
// {
data = map (
query:
if query.refId == "C" then
lib.recursiveUpdate query {
model.conditions = [
{
evaluator = {
type = "lt";
params = [ 1 ];
};
}
];
}
else
query
) rule.data;
noDataState = "Alerting";
};
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where # The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
# a real deployment needs the uids above. They are substituted here rather # a real deployment needs the uids above. They are substituted here rather
# than committed with the literals so the single binding stays single. # than committed with the literals so the single binding stays single.
@ -1080,7 +1111,64 @@ in
for = "5m"; for = "5m";
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue"; summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
}) })
]; # Logged once per agent process: the connection is made
# once and never retried, so the agent stays unconnected
# until it restarts. The window is how long the rule
# stays firing after that one line.
(logCountRule {
uid = "hh-swarm-queue-connect-failed";
title = "swarm queue connect failed";
windowSeconds = 3600;
expr = ''_time:1h _msg:"swarm queue connect failed; this agent publishes nothing upward" | stats by (hive, agent) count() as failures'';
for = "0s";
summary = "{{ $labels.hive }}/{{ $labels.agent }} could not connect to the swarm queue and publishes nothing upward until it restarts";
})
(logCountRule {
uid = "hh-swarm-agent-state-publish-skipped";
title = "swarm agent state publish skipped";
windowSeconds = 900;
expr = ''_time:15m _msg:"swarm agent state: publish skipped" | stats by (hive, agent) count() as skipped'';
for = "5m";
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping agent state updates: the harness is not connected to the queue";
})
# Logged on every connection attempt, so it repeats under
# the reconnect backoff for as long as the path is empty.
(logCountRule {
uid = "hh-swarm-queue-credential-missing";
title = "swarm queue credential missing";
windowSeconds = 900;
expr = ''_time:15m _msg:"nothing is stored at this agent's queue credential path; its queue connection retries under backoff" | stats by (hive, agent) count() as missing'';
for = "5m";
summary = "{{ $labels.hive }}/{{ $labels.agent }} finds nothing at its queue credential path and cannot connect to the queue";
})
# Same cadence as the forge rule: one pass every 5m
# (`RECONCILE_INTERVAL` in swarm-controller's
# agent_renewal.rs), one line per failed pass, so one
# failed pass never fires and about three in a row do.
(logCountRule {
uid = "hh-agent-credential-renewal-failed";
title = "agent credential renewal failing";
windowSeconds = 600;
expr = ''_time:10m _msg:"agent credential renewal: pass failed; retrying next tick" | stats count() as failures'';
for = "15m";
summary = "swarm-controller has failed the agent credential renewal pass on consecutive ticks";
})
]
# One rule per declared hive rather than one grouped by
# `hive`, so a hive that sends nothing at all still has
# an instance to fire. `hive` is stamped by the swarm
# collector from the same `swarm.hives` key.
++ lib.mapAttrsToList (
name: _:
logAbsenceRule {
uid = "hh-logs-absent-${name}";
title = "no log records from ${name}";
windowSeconds = 600;
expr = ''_time:10m hive:="${name}" | stats count() as records'';
for = "5m";
summary = "hive ${name} has sent no log records to VictoriaLogs for at least 10m";
}
) hyperhiveCfg.swarm.hives;
} }
]; ];
}; };