hive-c0re: serialise the matrix and knowledge sweeps through the job queue
The matrix sweep and the /knowledge pull each had concurrent callers (#4723 item 4). Two overlapping knowledge pulls fail on .git/index.lock and the remote-tracking ref lock: 30 of 30 concurrent replays of the reset/clean/pull sequence in a scratch repo errored, 0 of 10 sequential ones did. Two overlapping matrix sweeps on a hive with no persisted Space / chat-room id both miss the by-name lookup and both createRoom (from reading the code, not reproduced against a homeserver). On every boot the MatrixSweep DAG node and the main.rs loop's immediate first call ran at once. Every sweep now runs as a job node, and each sweep's node holds its own capacity-1 queue resource (Resource::MatrixSweep, Resource::KnowledgeTree), the MetaWindow pattern: the scheduler never starts a second pass of one sweep while the first holds the resource, and different sweeps still run side by side. - templates::matrix_sweep / templates::knowledge_pull build the node with its resource; boot, the periodic loops and the swarm event all use them. - JobQueue::insert_unless_live folds a submission into a live node of the same kind instead of queueing another. Periodic ticks fold into a queued or running pass. The swarm knowledge event folds into a queued pull only, and queues one behind a running pull, which may have fetched before the push. - The main.rs matrix loop no longer sweeps immediately at startup; the boot MatrixSweep node is the startup pass, as KnowledgePull already was for knowledge. - The executors bound each pass (10 min matrix, 5 min knowledge), since a hung pass would otherwise hold its resource against every later one, and own the sweep-health banners, so every pass reports to them. Replaces the SweepLock version of this branch, per review. Refs #4723
This commit is contained in:
parent
2252c55df8
commit
1d4c77d2c8
14 changed files with 352 additions and 105 deletions
|
|
@ -44,9 +44,7 @@ mod webhook_secret;
|
|||
mod workers;
|
||||
|
||||
pub(crate) use agent_config::{capabilities, limits, resource_limits, tool_groups, topology};
|
||||
pub(crate) use stats::{
|
||||
container_stats, hive_stats, host_stats, otel_metrics, sweep_health, warnings,
|
||||
};
|
||||
pub(crate) use stats::{container_stats, hive_stats, host_stats, otel_metrics, warnings};
|
||||
pub(crate) use stores::{approvals, broker, build_logs, db, power, scheduled_prompts};
|
||||
pub(crate) use workers::{
|
||||
agent_sockets, auto_update, crash_watch, knowledge, mcp_sockets, scheduled_prompts_worker,
|
||||
|
|
@ -228,20 +226,24 @@ async fn main() -> Result<()> {
|
|||
}
|
||||
}
|
||||
|
||||
/// Banner message for a failing matrix `ensure_all` sweep, shared by both
|
||||
/// the initial and periodic `record_err` call sites in `cmd_serve` so the
|
||||
/// wording can't drift between them.
|
||||
fn matrix_sweep_banner(ctx: sweep_health::SweepFailure) -> String {
|
||||
let age = ctx.since_last_ok.map_or_else(
|
||||
|| "no success this session".to_owned(),
|
||||
|d| format!("last ok {} ago", sweep_health::fmt_age(d)),
|
||||
);
|
||||
format!(
|
||||
"matrix user/space sweep failing ({} consecutive, {age}) \
|
||||
— some agents may be missing matrix accounts, space membership, \
|
||||
or chat-room invites",
|
||||
ctx.consecutive
|
||||
)
|
||||
/// One periodic sweep tick. It folds into a pass already queued or running
|
||||
/// instead of stacking another behind it, since that pass does the same work.
|
||||
fn submit_periodic_sweep(
|
||||
coord: &Coordinator,
|
||||
kind: &job_queue::NodeKind,
|
||||
declare: impl FnOnce(&job_queue::JobBuilder) -> job_queue::Handle<'_>,
|
||||
) {
|
||||
let live = [job_queue::State::Pending, job_queue::State::Running];
|
||||
match coord.job_queue.insert_unless_live(kind, &live, declare) {
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => tracing::debug!(
|
||||
kind = <&str>::from(kind),
|
||||
"periodic sweep: one already queued or running; folded into it"
|
||||
),
|
||||
Err(e) => {
|
||||
tracing::warn!(kind = <&str>::from(kind), error = ?e, "periodic sweep: submit failed");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Start the coordinator daemon: open the broker, run migrations, spawn
|
||||
|
|
@ -362,35 +364,14 @@ async fn cmd_serve(
|
|||
let mut knowledge_shutdown = coord.shutdown_rx();
|
||||
let knowledge_coord = coord.clone();
|
||||
tokio::spawn(async move {
|
||||
// Persistent-failure → banner. An hourly sweep that keeps failing for
|
||||
// several hours means the operator's `/knowledge` is drifting; raise a
|
||||
// warn banner after 3 consecutive misses so a one-off network blip
|
||||
// self-heals on the next tick without ever bannering. Cleared on the
|
||||
// next successful pull.
|
||||
let mut health = sweep_health::SweepHealth::new("knowledge_pull", "warn", 3);
|
||||
let interval = std::time::Duration::from_hours(1);
|
||||
loop {
|
||||
tokio::select! {
|
||||
() = tokio::time::sleep(interval) => {
|
||||
match knowledge::pull(&knowledge_coord).await {
|
||||
Ok(()) => health.record_ok(),
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "knowledge: periodic pull failed");
|
||||
let err = format!("{e:#}");
|
||||
health.record_err(|ctx| {
|
||||
let age = ctx.since_last_ok.map_or_else(
|
||||
|| "no success this session".to_owned(),
|
||||
|d| format!("last ok {} ago", sweep_health::fmt_age(d)),
|
||||
);
|
||||
format!(
|
||||
"knowledge repo pull failing ({} consecutive, {age}) \
|
||||
— /knowledge is stale until it recovers: {err}",
|
||||
ctx.consecutive
|
||||
)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
() = tokio::time::sleep(interval) => submit_periodic_sweep(
|
||||
&knowledge_coord,
|
||||
&job_queue::NodeKind::KnowledgePull,
|
||||
job_queue::templates::knowledge_pull,
|
||||
),
|
||||
_ = knowledge_shutdown.changed() => {
|
||||
tracing::info!("knowledge pull: shutdown signal received");
|
||||
break;
|
||||
|
|
@ -422,35 +403,23 @@ async fn cmd_serve(
|
|||
// Matrix user sweep: same shape — ensure every container has
|
||||
// an account on the local matrix-tuwunel homeserver with an
|
||||
// access_token persisted to `<state>/matrix-token`. No-op when
|
||||
// the hive-matrix container isn't running. Backgrounded because
|
||||
// UIAA is a two-roundtrip dance per agent.
|
||||
// the hive-matrix container isn't running.
|
||||
//
|
||||
// Runs once at startup AND periodically every 30 minutes so that
|
||||
// token files deleted by `hive-matrix-daemon` (stale-token
|
||||
// recovery — `M_UNKNOWN_TOKEN`) get re-provisioned without
|
||||
// requiring a hive-c0re restart.
|
||||
// Re-submitted every 30 minutes so that token files deleted by
|
||||
// `hive-matrix-daemon` (stale-token recovery — `M_UNKNOWN_TOKEN`)
|
||||
// get re-provisioned without requiring a hive-c0re restart. The
|
||||
// startup pass is the boot `MatrixSweep` node.
|
||||
let mut matrix_shutdown = coord.shutdown_rx();
|
||||
let matrix_coord = coord.clone();
|
||||
tokio::spawn(async move {
|
||||
let interval = std::time::Duration::from_mins(30);
|
||||
// Debounced banner: a lone bad sweep (homeserver mid-restart, a
|
||||
// transient HTTP blip) shouldn't flap the dashboard, but a sweep
|
||||
// that's been failing for hours (missing agent invites, a broken
|
||||
// admin token) should surface. Cleared the moment a sweep is clean.
|
||||
let mut health = sweep_health::SweepHealth::new("matrix_ensure_all", "warn", 2);
|
||||
if matrix::ensure_all().await {
|
||||
health.record_ok();
|
||||
} else {
|
||||
health.record_err(matrix_sweep_banner);
|
||||
}
|
||||
loop {
|
||||
tokio::select! {
|
||||
() = tokio::time::sleep(interval) => {
|
||||
if matrix::ensure_all().await {
|
||||
health.record_ok();
|
||||
} else {
|
||||
health.record_err(matrix_sweep_banner);
|
||||
}
|
||||
}
|
||||
() = tokio::time::sleep(interval) => submit_periodic_sweep(
|
||||
&matrix_coord,
|
||||
&job_queue::NodeKind::MatrixSweep,
|
||||
job_queue::templates::matrix_sweep,
|
||||
),
|
||||
_ = matrix_shutdown.changed() => {
|
||||
tracing::info!("matrix ensure_all: shutdown signal received");
|
||||
break;
|
||||
|
|
|
|||
Loading…
Reference in a new issue