diff --git a/nix/host-modules/swarm-grafana.nix b/nix/host-modules/swarm-grafana.nix index 954a2abb..a0a09561 100644 --- a/nix/host-modules/swarm-grafana.nix +++ b/nix/host-modules/swarm-grafana.nix @@ -72,6 +72,76 @@ let (fromSeverityText "TRACE" "trace") ]; + # An alert rule that fires while a LogsQL stats query counts more than zero + # lines. `expr` carries its own `_time:` window, so the count covers exactly + # that span; `relativeTimeRange` matches it so the rule editor's preview + # queries the same span. + # + # No lines is the healthy state, so no data is `OK`. A query that fails is + # `Error`, which the rule list shows separately from both. + logCountRule = + { + uid, + title, + windowSeconds, + expr, + for, + summary, + }: + { + inherit uid title for; + condition = "C"; + data = [ + { + refId = "A"; + relativeTimeRange = { + from = windowSeconds; + to = 0; + }; + datasourceUid = logsDatasourceUid; + model = { + refId = "A"; + datasource = { + type = "victoriametrics-logs-datasource"; + uid = logsDatasourceUid; + }; + queryType = "stats"; + inherit expr; + }; + } + { + refId = "B"; + datasourceUid = "__expr__"; + model = { + refId = "B"; + type = "reduce"; + expression = "A"; + reducer = "last"; + }; + } + { + refId = "C"; + datasourceUid = "__expr__"; + model = { + refId = "C"; + type = "threshold"; + expression = "B"; + conditions = [ + { + evaluator = { + type = "gt"; + params = [ 0 ]; + }; + } + ]; + }; + } + ]; + noDataState = "OK"; + execErrState = "Error"; + annotations = { inherit summary; }; + }; + # The shipped dashboards carry `@datasourceUid@` / `@logsDatasourceUid@` where # a real deployment needs the uids above. They are substituted here rather # than committed with the literals so the single binding stays single. @@ -969,6 +1039,51 @@ in } ]; }; + + # Rules only. With no policy provisioned, Grafana routes every alert + # to its built-in `empty` receiver, which has no integrations: a + # firing rule shows under Alerting → Alert rules and is sent nowhere. + provision.alerting.rules.settings = { + apiVersion = 1; + groups = [ + { + orgId = 1; + name = "hyperhive"; + folder = "hyperhive"; + interval = "1m"; + rules = [ + # The controller re-runs the forge objects pass every 5m + # (`RECONCILE_INTERVAL` in swarm-controller's + # forge/objects.rs) and warns once per failed write. A + # 10m window always spans the previous tick, so a pass + # failing on every tick keeps the count above zero + # continuously and fires after `for`. One failed pass + # keeps it above zero for only 10m, short of the 15m + # `for`, so it never fires; about three failed passes in + # a row do. + (logCountRule { + uid = "hh-forge-objects-write-failed"; + title = "forge reconcile failing"; + windowSeconds = 600; + expr = ''_time:10m _msg:"swarm forge objects: write failed; retrying next pass" | stats count() as failures''; + for = "15m"; + summary = "swarm-controller has failed to write forge objects on consecutive reconcile passes"; + }) + # `hive` and `agent` are the resource attributes every + # agent container's log forwarder carries, so each agent + # is its own alert instance. + (logCountRule { + uid = "hh-swarm-term-publish-skipped"; + title = "swarm terminal publish skipped"; + windowSeconds = 900; + expr = ''_time:15m _msg:"swarm terminal: publish skipped" | stats by (hive, agent) count() as skipped''; + for = "5m"; + summary = "{{ $labels.hive }}/{{ $labels.agent }} is dropping swarm terminal rows: the harness is not connected to the queue"; + }) + ]; + } + ]; + }; }; }; };