swarm-grafana: alert rules for queue, renewal and missing-log WARNs
Adds rules to the `hyperhive` alerting folder next to the two from #4811: - swarm queue connect failed (per hive/agent, 1h window, fires at once: the line is logged once per agent process and the connection is never retried, so the window is how long the rule stays firing) - swarm agent state publish skipped (per hive/agent, 15m, for 5m, as the swarm terminal rule) - swarm queue credential missing (per hive/agent, 15m, for 5m) - agent credential renewal failing (10m, for 15m, as the forge reconcile rule: same 5m pass cadence, one line per failed pass) - no log records from <hive>, one per `swarm.hives` entry: fires when the hive's ungrouped count over 10m is below 1, and on no data. A new `logAbsenceRule` helper wraps `logCountRule` with the inverted threshold and noDataState = Alerting. No contact point and no notification policy: the rules show under Alerting -> Alert rules and are delivered nowhere. Refs #4717 Refs #3900
This commit is contained in:
parent
7ccbeaa161
commit
b9667d5975
1 changed files with 89 additions and 1 deletions
|
|
@ -142,6 +142,37 @@ let
|
||||||
annotations = { inherit summary; };
|
annotations = { inherit summary; };
|
||||||
};
|
};
|
||||||
|
|
||||||
|
# `logCountRule` inverted: fires while the count is below one. `expr` must
|
||||||
|
# be an ungrouped `stats count()`, which answers 0 rather than no row when
|
||||||
|
# nothing matched; a grouped count drops the group instead, and a dropped
|
||||||
|
# instance resolves rather than fires. No data is `Alerting` so a query
|
||||||
|
# that returns nothing at all still fires.
|
||||||
|
logAbsenceRule =
|
||||||
|
args:
|
||||||
|
let
|
||||||
|
rule = logCountRule args;
|
||||||
|
in
|
||||||
|
rule
|
||||||
|
// {
|
||||||
|
data = map (
|
||||||
|
query:
|
||||||
|
if query.refId == "C" then
|
||||||
|
lib.recursiveUpdate query {
|
||||||
|
model.conditions = [
|
||||||
|
{
|
||||||
|
evaluator = {
|
||||||
|
type = "lt";
|
||||||
|
params = [ 1 ];
|
||||||
|
};
|
||||||
|
}
|
||||||
|
];
|
||||||
|
}
|
||||||
|
else
|
||||||
|
query
|
||||||
|
) rule.data;
|
||||||
|
noDataState = "Alerting";
|
||||||
|
};
|
||||||
|
|
||||||
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
|
# The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where
|
||||||
# a real deployment needs the uids above. They are substituted here rather
|
# a real deployment needs the uids above. They are substituted here rather
|
||||||
# than committed with the literals so the single binding stays single.
|
# than committed with the literals so the single binding stays single.
|
||||||
|
|
@ -1080,7 +1111,64 @@ in
|
||||||
for = "5m";
|
for = "5m";
|
||||||
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
|
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue";
|
||||||
})
|
})
|
||||||
];
|
# Logged once per agent process: the connection is made
|
||||||
|
# once and never retried, so the agent stays unconnected
|
||||||
|
# until it restarts. The window is how long the rule
|
||||||
|
# stays firing after that one line.
|
||||||
|
(logCountRule {
|
||||||
|
uid = "hh-swarm-queue-connect-failed";
|
||||||
|
title = "swarm queue connect failed";
|
||||||
|
windowSeconds = 3600;
|
||||||
|
expr = ''_time:1h _msg:"swarm queue connect failed; this agent publishes nothing upward" | stats by (hive, agent) count() as failures'';
|
||||||
|
for = "0s";
|
||||||
|
summary = "{{ $labels.hive }}/{{ $labels.agent }} could not connect to the swarm queue and publishes nothing upward until it restarts";
|
||||||
|
})
|
||||||
|
(logCountRule {
|
||||||
|
uid = "hh-swarm-agent-state-publish-skipped";
|
||||||
|
title = "swarm agent state publish skipped";
|
||||||
|
windowSeconds = 900;
|
||||||
|
expr = ''_time:15m _msg:"swarm agent state: publish skipped" | stats by (hive, agent) count() as skipped'';
|
||||||
|
for = "5m";
|
||||||
|
summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping agent state updates: the harness is not connected to the queue";
|
||||||
|
})
|
||||||
|
# Logged on every connection attempt, so it repeats under
|
||||||
|
# the reconnect backoff for as long as the path is empty.
|
||||||
|
(logCountRule {
|
||||||
|
uid = "hh-swarm-queue-credential-missing";
|
||||||
|
title = "swarm queue credential missing";
|
||||||
|
windowSeconds = 900;
|
||||||
|
expr = ''_time:15m _msg:"nothing is stored at this agent's queue credential path; its queue connection retries under backoff" | stats by (hive, agent) count() as missing'';
|
||||||
|
for = "5m";
|
||||||
|
summary = "{{ $labels.hive }}/{{ $labels.agent }} finds nothing at its queue credential path and cannot connect to the queue";
|
||||||
|
})
|
||||||
|
# Same cadence as the forge rule: one pass every 5m
|
||||||
|
# (`RECONCILE_INTERVAL` in swarm-controller's
|
||||||
|
# agent_renewal.rs), one line per failed pass, so one
|
||||||
|
# failed pass never fires and about three in a row do.
|
||||||
|
(logCountRule {
|
||||||
|
uid = "hh-agent-credential-renewal-failed";
|
||||||
|
title = "agent credential renewal failing";
|
||||||
|
windowSeconds = 600;
|
||||||
|
expr = ''_time:10m _msg:"agent credential renewal: pass failed; retrying next tick" | stats count() as failures'';
|
||||||
|
for = "15m";
|
||||||
|
summary = "swarm-controller has failed the agent credential renewal pass on consecutive ticks";
|
||||||
|
})
|
||||||
|
]
|
||||||
|
# One rule per declared hive rather than one grouped by
|
||||||
|
# `hive`, so a hive that sends nothing at all still has
|
||||||
|
# an instance to fire. `hive` is stamped by the swarm
|
||||||
|
# collector from the same `swarm.hives` key.
|
||||||
|
++ lib.mapAttrsToList (
|
||||||
|
name: _:
|
||||||
|
logAbsenceRule {
|
||||||
|
uid = "hh-logs-absent-${name}";
|
||||||
|
title = "no log records from ${name}";
|
||||||
|
windowSeconds = 600;
|
||||||
|
expr = ''_time:10m hive:="${name}" | stats count() as records'';
|
||||||
|
for = "5m";
|
||||||
|
summary = "hive ${name} has sent no log records to VictoriaLogs for at least 10m";
|
||||||
|
}
|
||||||
|
) hyperhiveCfg.swarm.hives;
|
||||||
}
|
}
|
||||||
];
|
];
|
||||||
};
|
};
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue