Watch
0
0
Fork
You've already forked hyperhive
0

swarm-grafana: provision alert rules for forge reconcile + swarm terminal publish failures

Two Grafana-managed rules in a "hyperhive" folder, one group evaluated
every 1m, each a LogsQL stats count over the VictoriaLogs datasource:

- forge reconcile failing: `swarm forge objects: write failed; retrying
  next pass` over 10m, for 15m. The pass runs every 5m, so one failed
  pass stays under `for` and a pass failing on every tick fires.
- swarm terminal publish skipped: `swarm terminal: publish skipped` over
  15m, for 5m, one instance per hive/agent.

No contact point or notification policy. Grafana 13 routes to its
built-in `empty` receiver when none is provisioned, so firing rules
are visible under Alerting and sent nowhere, with no send errors.

Refs #4717
This commit is contained in:
atlas 2026-09-29 20:39:51 +02:00 • committed by mara
commit 41d66e2e05

View file

@ -72,6 +72,76 @@ let
(fromSeverityText "TRACE" "trace")
];
# An alert rule that fires while a LogsQL stats query counts more than zero
# lines. `expr` carries its own `_time:` window, so the count covers exactly
# that span; `relativeTimeRange` matches it so the rule editor's preview
# queries the same span.
#
# No lines is the healthy state, so no data is `OK`. A query that fails is
# `Error`, which the rule list shows separately from both.
logCountRule =
{
uid,
title,
windowSeconds,
expr,
for,
summary,
}:
{
inherit uid title for;
condition = "C";
data = [
{
refId = "A";
relativeTimeRange = {
from = windowSeconds;
to = 0;
};
datasourceUid = logsDatasourceUid;
model = {
refId = "A";
datasource = {
type = "victoriametrics-logs-datasource";
uid = logsDatasourceUid;
};
queryType = "stats";
inherit expr;
};
}
{
refId = "B";
datasourceUid = "__expr__";
model = {
refId = "B";
type = "reduce";
expression = "A";
reducer = "last";
};
}
{
refId = "C";
datasourceUid = "__expr__";
model = {
refId = "C";
type = "threshold";
expression = "B";
conditions = [
{
evaluator = {
type = "gt";
params = [ 0 ];
};
}
];
};
}
];
noDataState = "OK";
execErrState = "Error";
annotations = { inherit summary; };
};
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
# a real deployment needs the uids above. They are substituted here rather
# than committed with the literals so the single binding stays single.
@ -969,6 +1039,51 @@ in
}
];
};
# Rules only. With no policy provisioned, Grafana routes every alert
# to its built-in `empty` receiver, which has no integrations: a
# firing rule shows under Alerting → Alert rules and is sent nowhere.
provision.alerting.rules.settings = {
apiVersion = 1;
groups = [
{
orgId = 1;
name = "hyperhive";
folder = "hyperhive";
interval = "1m";
rules = [
# The controller re-runs the forge objects pass every 5m
# (`RECONCILE_INTERVAL` in swarm-controller's
# forge/objects.rs) and warns once per failed write. A
# 10m window always spans the previous tick, so a pass
# failing on every tick keeps the count above zero
# continuously and fires after `for`. One failed pass
# keeps it above zero for only 10m, short of the 15m
# `for`, so it never fires; about three failed passes in
# a row do.
(logCountRule {
uid = "hh-forge-objects-write-failed";
title = "forge reconcile failing";
windowSeconds = 600;
expr = ''_time:10m _msg:"swarm forge objects: write failed; retrying next pass" | stats count() as failures'';
for = "15m";
summary = "swarm-controller has failed to write forge objects on consecutive reconcile passes";
})
# `hive` and `agent` are the resource attributes every
# agent container's log forwarder carries, so each agent
# is its own alert instance.
(logCountRule {
uid = "hh-swarm-term-publish-skipped";
title = "swarm terminal publish skipped";
windowSeconds = 900;
expr = ''_time:15m _msg:"swarm terminal: publish skipped" | stats by (hive, agent) count() as skipped'';
for = "5m";
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
})
];
}
];
};
};
};
};