swarm-grafana: alert rules for queue, renewal and missing-log WARNs
Adds rules to the `hyperhive` alerting folder next to the two from #4811: - swarm queue connect failed (per hive/agent, 1h window, fires at once: the line is logged once per agent process and the connection is never retried, so the window is how long the rule stays firing) - swarm agent state publish skipped (per hive/agent, 15m, for 5m, as the swarm terminal rule) - swarm queue credential missing (per hive/agent, 15m, for 5m) - agent credential renewal failing (10m, for 15m, as the forge reconcile rule: same 5m pass cadence, one line per failed pass) - no log records from <hive>, one per `swarm.hives` entry: fires when the hive's ungrouped count over 10m is below 1, and on no data. A new `logAbsenceRule` helper wraps `logCountRule` with the inverted threshold and noDataState = Alerting. No contact point and no notification policy: the rules show under Alerting -> Alert rules and are delivered nowhere. Refs #4717 Refs #3900
This commit is contained in:
parent
7ccbeaa161
commit
b9667d5975
1 changed files with 89 additions and 1 deletions
|
|
@ -142,6 +142,37 @@ let
|
|||
annotations = { inherit summary; };
|
||||
};
|
||||
|
||||
# `logCountRule` inverted: fires while the count is below one. `expr` must
|
||||
# be an ungrouped `stats count()`, which answers 0 rather than no row when
|
||||
# nothing matched; a grouped count drops the group instead, and a dropped
|
||||
# instance resolves rather than fires. No data is `Alerting` so a query
|
||||
# that returns nothing at all still fires.
|
||||
logAbsenceRule =
|
||||
args:
|
||||
let
|
||||
rule = logCountRule args;
|
||||
in
|
||||
rule
|
||||
// {
|
||||
data = map (
|
||||
query:
|
||||
if query.refId == "C" then
|
||||
lib.recursiveUpdate query {
|
||||
model.conditions = [
|
||||
{
|
||||
evaluator = {
|
||||
type = "lt";
|
||||
params = [ 1 ];
|
||||
};
|
||||
}
|
||||
];
|
||||
}
|
||||
else
|
||||
query
|
||||
) rule.data;
|
||||
noDataState = "Alerting";
|
||||
};
|
||||
|
||||
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
|
||||
# a real deployment needs the uids above. They are substituted here rather
|
||||
# than committed with the literals so the single binding stays single.
|
||||
|
|
@ -1080,7 +1111,64 @@ in
|
|||
for = "5m";
|
||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
|
||||
})
|
||||
];
|
||||
# Logged once per agent process: the connection is made
|
||||
# once and never retried, so the agent stays unconnected
|
||||
# until it restarts. The window is how long the rule
|
||||
# stays firing after that one line.
|
||||
(logCountRule {
|
||||
uid = "hh-swarm-queue-connect-failed";
|
||||
title = "swarm queue connect failed";
|
||||
windowSeconds = 3600;
|
||||
expr = ''_time:1h _msg:"swarm queue connect failed; this agent publishes nothing upward" | stats by (hive, agent) count() as failures'';
|
||||
for = "0s";
|
||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} could not connect to the swarm queue and publishes nothing upward until it restarts";
|
||||
})
|
||||
(logCountRule {
|
||||
uid = "hh-swarm-agent-state-publish-skipped";
|
||||
title = "swarm agent state publish skipped";
|
||||
windowSeconds = 900;
|
||||
expr = ''_time:15m _msg:"swarm agent state: publish skipped" | stats by (hive, agent) count() as skipped'';
|
||||
for = "5m";
|
||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping agent state updates: the harness is not connected to the queue";
|
||||
})
|
||||
# Logged on every connection attempt, so it repeats under
|
||||
# the reconnect backoff for as long as the path is empty.
|
||||
(logCountRule {
|
||||
uid = "hh-swarm-queue-credential-missing";
|
||||
title = "swarm queue credential missing";
|
||||
windowSeconds = 900;
|
||||
expr = ''_time:15m _msg:"nothing is stored at this agent's queue credential path; its queue connection retries under backoff" | stats by (hive, agent) count() as missing'';
|
||||
for = "5m";
|
||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} finds nothing at its queue credential path and cannot connect to the queue";
|
||||
})
|
||||
# Same cadence as the forge rule: one pass every 5m
|
||||
# (`RECONCILE_INTERVAL` in swarm-controller's
|
||||
# agent_renewal.rs), one line per failed pass, so one
|
||||
# failed pass never fires and about three in a row do.
|
||||
(logCountRule {
|
||||
uid = "hh-agent-credential-renewal-failed";
|
||||
title = "agent credential renewal failing";
|
||||
windowSeconds = 600;
|
||||
expr = ''_time:10m _msg:"agent credential renewal: pass failed; retrying next tick" | stats count() as failures'';
|
||||
for = "15m";
|
||||
summary = "swarm-controller has failed the agent credential renewal pass on consecutive ticks";
|
||||
})
|
||||
]
|
||||
# One rule per declared hive rather than one grouped by
|
||||
# `hive`, so a hive that sends nothing at all still has
|
||||
# an instance to fire. `hive` is stamped by the swarm
|
||||
# collector from the same `swarm.hives` key.
|
||||
++ lib.mapAttrsToList (
|
||||
name: _:
|
||||
logAbsenceRule {
|
||||
uid = "hh-logs-absent-${name}";
|
||||
title = "no log records from ${name}";
|
||||
windowSeconds = 600;
|
||||
expr = ''_time:10m hive:="${name}" | stats count() as records'';
|
||||
for = "5m";
|
||||
summary = "hive ${name} has sent no log records to VictoriaLogs for at least 10m";
|
||||
}
|
||||
) hyperhiveCfg.swarm.hives;
|
||||
}
|
||||
];
|
||||
};
|
||||
|
|
|
|||
Loading…
Reference in a new issue