swarm-grafana: provision alert rules for forge reconcile + swarm terminal publish failures
Two Grafana-managed rules in a "hyperhive" folder, one group evaluated every 1m, each a LogsQL stats count over the VictoriaLogs datasource: - forge reconcile failing: `swarm forge objects: write failed; retrying next pass` over 10m, for 15m. The pass runs every 5m, so one failed pass stays under `for` and a pass failing on every tick fires. - swarm terminal publish skipped: `swarm terminal: publish skipped` over 15m, for 5m, one instance per hive/agent. No contact point or notification policy. Grafana 13 routes to its built-in `empty` receiver when none is provisioned, so firing rules are visible under Alerting and sent nowhere, with no send errors. Refs #4717
This commit is contained in:
parent
10d579ecaf
commit
41d66e2e05
1 changed files with 115 additions and 0 deletions
|
|
@ -72,6 +72,76 @@ let
|
|||
(fromSeverityText "TRACE" "trace")
|
||||
];
|
||||
|
||||
# An alert rule that fires while a LogsQL stats query counts more than zero
|
||||
# lines. `expr` carries its own `_time:` window, so the count covers exactly
|
||||
# that span; `relativeTimeRange` matches it so the rule editor's preview
|
||||
# queries the same span.
|
||||
#
|
||||
# No lines is the healthy state, so no data is `OK`. A query that fails is
|
||||
# `Error`, which the rule list shows separately from both.
|
||||
logCountRule =
|
||||
{
|
||||
uid,
|
||||
title,
|
||||
windowSeconds,
|
||||
expr,
|
||||
for,
|
||||
summary,
|
||||
}:
|
||||
{
|
||||
inherit uid title for;
|
||||
condition = "C";
|
||||
data = [
|
||||
{
|
||||
refId = "A";
|
||||
relativeTimeRange = {
|
||||
from = windowSeconds;
|
||||
to = 0;
|
||||
};
|
||||
datasourceUid = logsDatasourceUid;
|
||||
model = {
|
||||
refId = "A";
|
||||
datasource = {
|
||||
type = "victoriametrics-logs-datasource";
|
||||
uid = logsDatasourceUid;
|
||||
};
|
||||
queryType = "stats";
|
||||
inherit expr;
|
||||
};
|
||||
}
|
||||
{
|
||||
refId = "B";
|
||||
datasourceUid = "__expr__";
|
||||
model = {
|
||||
refId = "B";
|
||||
type = "reduce";
|
||||
expression = "A";
|
||||
reducer = "last";
|
||||
};
|
||||
}
|
||||
{
|
||||
refId = "C";
|
||||
datasourceUid = "__expr__";
|
||||
model = {
|
||||
refId = "C";
|
||||
type = "threshold";
|
||||
expression = "B";
|
||||
conditions = [
|
||||
{
|
||||
evaluator = {
|
||||
type = "gt";
|
||||
params = [ 0 ];
|
||||
};
|
||||
}
|
||||
];
|
||||
};
|
||||
}
|
||||
];
|
||||
noDataState = "OK";
|
||||
execErrState = "Error";
|
||||
annotations = { inherit summary; };
|
||||
};
|
||||
|
||||
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
|
||||
# a real deployment needs the uids above. They are substituted here rather
|
||||
# than committed with the literals so the single binding stays single.
|
||||
|
|
@ -969,6 +1039,51 @@ in
|
|||
}
|
||||
];
|
||||
};
|
||||
|
||||
# Rules only. With no policy provisioned, Grafana routes every alert
|
||||
# to its built-in `empty` receiver, which has no integrations: a
|
||||
# firing rule shows under Alerting → Alert rules and is sent nowhere.
|
||||
provision.alerting.rules.settings = {
|
||||
apiVersion = 1;
|
||||
groups = [
|
||||
{
|
||||
orgId = 1;
|
||||
name = "hyperhive";
|
||||
folder = "hyperhive";
|
||||
interval = "1m";
|
||||
rules = [
|
||||
# The controller re-runs the forge objects pass every 5m
|
||||
# (`RECONCILE_INTERVAL` in swarm-controller's
|
||||
# forge/objects.rs) and warns once per failed write. A
|
||||
# 10m window always spans the previous tick, so a pass
|
||||
# failing on every tick keeps the count above zero
|
||||
# continuously and fires after `for`. One failed pass
|
||||
# keeps it above zero for only 10m, short of the 15m
|
||||
# `for`, so it never fires; about three failed passes in
|
||||
# a row do.
|
||||
(logCountRule {
|
||||
uid = "hh-forge-objects-write-failed";
|
||||
title = "forge reconcile failing";
|
||||
windowSeconds = 600;
|
||||
expr = ''_time:10m _msg:"swarm forge objects: write failed; retrying next pass" | stats count() as failures'';
|
||||
for = "15m";
|
||||
summary = "swarm-controller has failed to write forge objects on consecutive reconcile passes";
|
||||
})
|
||||
# `hive` and `agent` are the resource attributes every
|
||||
# agent container's log forwarder carries, so each agent
|
||||
# is its own alert instance.
|
||||
(logCountRule {
|
||||
uid = "hh-swarm-term-publish-skipped";
|
||||
title = "swarm terminal publish skipped";
|
||||
windowSeconds = 900;
|
||||
expr = ''_time:15m _msg:"swarm terminal: publish skipped" | stats by (hive, agent) count() as skipped'';
|
||||
for = "5m";
|
||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
|
||||
})
|
||||
];
|
||||
}
|
||||
];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
|
|
|
|||
Loading…
Reference in a new issue