diff --git a/Cargo.lock b/Cargo.lock index f7bf45d2..8ede80f9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1620,7 +1620,6 @@ dependencies = [ "forgejo-api", "hive-core-agent-sock", "hive-host-sock", - "hive-jobq", "hive-priv-sock", "hive-sh4re", "hive-types", diff --git a/Cargo.toml b/Cargo.toml index b4e4848f..5ff5da9c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -53,7 +53,6 @@ clap_complete = "4" indicatif = "0.18" hive-sh4re = { path = "hive-sh4re" } hive-agent-sock = { path = "hive-agent-sock" } -hive-jobq = { path = "hive-jobq" } hive-core-agent-sock = { path = "hive-core-agent-sock" } hive-claude = { path = "hive-claude" } hive-host-sock = { path = "hive-host-sock" } diff --git a/hive-c0re/Cargo.toml b/hive-c0re/Cargo.toml index 9078843f..8809dc9a 100644 --- a/hive-c0re/Cargo.toml +++ b/hive-c0re/Cargo.toml @@ -33,7 +33,6 @@ indicatif.workspace = true hive-core-agent-sock.workspace = true hive-sh4re.workspace = true hive-host-sock.workspace = true -hive-jobq.workspace = true hive-priv-sock.workspace = true hive-types.workspace = true libc.workspace = true diff --git a/hive-c0re/src/coordinator.rs b/hive-c0re/src/coordinator.rs index e5f8b3b5..9d40377d 100644 --- a/hive-c0re/src/coordinator.rs +++ b/hive-c0re/src/coordinator.rs @@ -369,8 +369,7 @@ impl Drop for MetaUpdateGuard { } } -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize)] -#[serde(rename_all = "snake_case")] +#[derive(Debug, Clone, Copy)] pub enum TransientKind { /// `lifecycle::spawn` is running (nixos-container create + update + start). Spawning, diff --git a/hive-c0re/src/dashboard/schedules.rs b/hive-c0re/src/dashboard/schedules.rs index 6b990f93..8ad220d9 100644 --- a/hive-c0re/src/dashboard/schedules.rs +++ b/hive-c0re/src/dashboard/schedules.rs @@ -127,10 +127,8 @@ pub(super) async fn post_rebuild_queue_cancel( State(state): State, AxumPath(id): AxumPath, ) -> Response { - if let Some(terminal) = state.coord.job_queue.cancel(id) { - // Fire the DAG's inline terminal hook (power-op intent revert / approval - // resolution) off the cancel roll-up, then surface the flip live. - crate::job_queue::exec::run_terminal_hook(&state.coord, &terminal).await; + let cancelled = state.coord.job_queue.cancel(id); + if cancelled { state.coord.emit_rebuild_queue_snapshot(); axum::Json(serde_json::json!({"cancelled": true})).into_response() } else { diff --git a/hive-c0re/src/job_queue/exec.rs b/hive-c0re/src/job_queue/exec.rs index 3a232a3b..66e5b53c 100644 --- a/hive-c0re/src/job_queue/exec.rs +++ b/hive-c0re/src/job_queue/exec.rs @@ -10,8 +10,8 @@ use std::sync::Arc; use anyhow::{Context as _, Result}; -use super::Claim; -use super::model::{NodeKind, NodeSpec, State}; +use super::model::{NodeKind, NodeSpec, State, Template}; +use super::{Claim, TerminalDag}; use crate::coordinator::Coordinator; use crate::power::{ReconcileAction, reconcile_action}; @@ -82,86 +82,24 @@ pub(super) async fn run_node(coord: &Arc, claim: &Claim) -> Result< node_id: claim.node_id, }; match &claim.kind { - NodeKind::Prebuild { relock, .. } => run_prebuild(coord, claim, &ctx, *relock).await, - NodeKind::Swap { .. } => run_swap(coord, claim, &ctx).await, - NodeKind::PostSwap { .. } => run_post_swap(coord, claim, &ctx).await, - NodeKind::Provision { .. } => run_provision(coord, claim, &ctx).await, - NodeKind::Create { .. } => run_create(claim, &ctx).await, + NodeKind::Prebuild { relock } => run_prebuild(coord, claim, &ctx, *relock).await, + NodeKind::Swap => run_swap(coord, claim, &ctx).await, + NodeKind::PostSwap => run_post_swap(coord, claim, &ctx).await, + NodeKind::Provision => run_provision(coord, claim, &ctx).await, + NodeKind::Create => run_create(claim, &ctx).await, NodeKind::MetaLock { sweep, fanout } => { run_meta_lock(coord, claim, &ctx, *sweep, fanout.clone()).await } - NodeKind::Reconcile { .. } => run_reconcile(coord, claim).await, - NodeKind::Start { .. } => run_start(coord, claim, &ctx).await, - NodeKind::Stop { .. } => run_stop(coord, claim, &ctx).await, - NodeKind::StopForUpdate { .. } => run_stop_for_update(coord, claim, &ctx).await, - NodeKind::Signal { .. } => Ok(run_signal(coord, claim, &ctx)), - NodeKind::Drain { .. } => run_drain(coord, claim, &ctx).await, - NodeKind::WriteDropin { .. } => run_write_dropin(coord, claim).await, - NodeKind::WritePermFile { .. } => run_write_perm_file(coord, claim, &ctx).await, - NodeKind::ApprovalDeploy { .. } => run_approval_deploy(coord, claim).await, - NodeKind::SetWanted { up, .. } => run_set_wanted(coord, claim, *up), - // Pure grouping container — no work; completing it lets it reach - // `Finishing` so its child template nodes start. The DAG's terminal - // hook fires (inline, via `run_terminal_hook`) when the container itself - // rolls up terminal — not as a scheduled node. - NodeKind::Dag { .. } => Ok(NodeOutput::default()), - } -} - -/// Run a settled DAG's inline terminal hook, dispatched off its rolled-up -/// summary — the container-terminal replacement for the old per-DAG hook node. -/// Always best-effort: a hook failure is logged inside, never surfaced. -pub(crate) async fn run_terminal_hook(coord: &Arc, terminal: &super::TerminalDag) { - match super::terminal_hook(terminal.template, terminal.approval_id) { - Some(super::HookKind::ResolveApproval) => { - crate::actions::resolve_approval_dag(coord, terminal).await; - } - Some(super::HookKind::EmitRebuilt) => emit_rebuilt(coord, terminal), - Some(super::HookKind::RevertIntent) => revert_intent(coord, terminal).await, - None => {} - } -} - -/// Rebuild / perm-change hook: emit one `Rebuilt` manager event per targeted -/// agent — `ok` on `Done`, `!ok` on `Failed`, none on cancel. -fn emit_rebuilt(coord: &Arc, terminal: &super::TerminalDag) { - for agent in &terminal.agents { - match terminal.state { - State::Done => coord.notify_manager(&hive_sh4re::HelperEvent::Rebuilt { - agent: agent.clone(), - ok: true, - note: None, - sha: None, - tag: None, - }), - State::Failed => coord.notify_manager(&hive_sh4re::HelperEvent::Rebuilt { - agent: agent.clone(), - ok: false, - note: terminal.error.clone(), - sha: None, - tag: None, - }), - _ => {} - } - } -} - -/// Power-op hook: on a *cancelled* DAG, revert each targeted agent's `wanted` -/// intent to its observed state — the operator's cancel means "don't do it", so -/// the intent snaps back instead of the flip executing as a surprise side effect -/// of some later reconcile. Noop on any non-cancelled outcome. -async fn revert_intent(coord: &Arc, terminal: &super::TerminalDag) { - if terminal.state != State::Cancelled { - return; - } - for agent in &terminal.agents { - let running = crate::lifecycle::is_running(agent).await; - if let Err(e) = coord - .power - .set(agent, crate::power::Wanted::from_running(running)) - { - tracing::warn!(%agent, error = ?e, "agent_power: cancel revert failed"); - } + NodeKind::Reconcile => run_reconcile(coord, claim).await, + NodeKind::Start => run_start(coord, claim, &ctx).await, + NodeKind::Stop => run_stop(coord, claim, &ctx).await, + NodeKind::StopForUpdate => run_stop_for_update(coord, claim, &ctx).await, + NodeKind::Signal => Ok(run_signal(coord, claim, &ctx)), + NodeKind::Drain => run_drain(coord, claim, &ctx).await, + NodeKind::WriteDropin => run_write_dropin(coord, claim).await, + NodeKind::WritePermFile => run_write_perm_file(coord, claim, &ctx).await, + NodeKind::ApprovalDeploy => run_approval_deploy(coord, claim).await, + NodeKind::SetWanted { up } => run_set_wanted(coord, claim, *up), } } @@ -398,17 +336,14 @@ async fn run_reconcile(coord: &Arc, claim: &Claim) -> Result sub(NodeKind::Start { - agent: name.clone(), - }), - ReconcileAction::Stop => sub(NodeKind::Stop { - agent: name.clone(), - }), + ReconcileAction::Start => sub(NodeKind::Start), + ReconcileAction::Stop => sub(NodeKind::Stop), ReconcileAction::Noop => { tracing::debug!(%name, wanted = wanted.as_str(), running, "reconcile: noop"); Vec::new() @@ -540,30 +475,26 @@ async fn run_write_perm_file( ) -> Result { use super::model::PermPayload; let name = &claim.agent; - // The perm file payload rides the node itself (the only consumer). - let NodeKind::WritePermFile { payload, .. } = &claim.kind else { - anyhow::bail!("run_write_perm_file on a non-WritePermFile node"); - }; ctx.step("writing + committing perm file"); // Deploy-window gate: a perm commit landing inside another node's // staged prepare→finalize window would sweep the staged deploy // lock into its commit (the commits are also path-limited in // meta.rs — belt and braces). let _window = crate::meta::exclusive().await; - match payload { - PermPayload::ToolGroups { groups } => { + match &claim.perm_payload { + Some(PermPayload::ToolGroups { groups }) => { crate::meta::commit_tool_groups(name, groups) .await .with_context(|| format!("commit tool-groups for {name}"))?; coord.emit_tool_groups_snapshot(); } - PermPayload::Capabilities { caps } => { + Some(PermPayload::Capabilities { caps }) => { crate::meta::commit_capabilities(name, caps) .await .with_context(|| format!("commit capabilities for {name}"))?; coord.emit_capabilities_snapshot(); } - PermPayload::Combined { groups, caps } => { + Some(PermPayload::Combined { groups, caps }) => { crate::meta::commit_perms(name, groups.as_deref(), caps.as_deref()) .await .with_context(|| format!("commit perms for {name}"))?; @@ -574,6 +505,10 @@ async fn run_write_perm_file( coord.emit_capabilities_snapshot(); } } + None => anyhow::bail!( + "perm_change dag {} for {name} is missing perm_payload", + claim.dag_id + ), } Ok(NodeOutput::default()) } @@ -596,6 +531,71 @@ async fn run_approval_deploy(coord: &Arc, claim: &Claim) -> Result< .map(|()| NodeOutput::default()) } +/// Terminal-roll-up hook, fired exactly once per DAG (node completion +/// and cancel paths alike — the queue buffers roll-ups and the +/// scheduler drains them). Three concerns: +/// - approval DAGs resolve their approval row (except the opaque +/// deploy pipeline, which resolves inside its node — unless it was +/// cancelled while still queued and the node never ran); +/// - non-approval rebuild-shaped DAGs emit exactly one `Rebuilt` +/// manager event: ok on `Done`, !ok on `Failed`, none on cancel; +/// - a cancelled power-op DAG reverts the `wanted` intent its submit +/// wrote: the operator's cancel means "don't do it", so intent +/// snaps back to the observed state instead of the flip executing +/// as a surprise side effect of some later reconcile. +pub(super) async fn on_dag_terminal(coord: &Arc, terminal: &TerminalDag) { + if terminal.state == State::Cancelled + && matches!( + terminal.template, + Template::Start + | Template::Stop + | Template::GracefulStop + | Template::Restart + | Template::GracefulRestart + ) + { + // Revert each targeted agent's power intent to its observed state — + // the operator's cancel means "don't do it". Single-agent power-op + // DAGs have one agent here. + for agent in &terminal.agents { + let running = crate::lifecycle::is_running(agent).await; + if let Err(e) = coord + .power + .set(agent, crate::power::Wanted::from_running(running)) + { + tracing::warn!(%agent, error = ?e, "agent_power: cancel revert failed"); + } + } + } + if terminal.approval_id.is_some() { + crate::actions::resolve_approval_dag(coord, terminal).await; + return; + } + if matches!(terminal.template, Template::Rebuild | Template::PermChange) { + // Rebuild / PermChange are single-agent; emit one `Rebuilt` per + // targeted agent (exactly one today). + for agent in &terminal.agents { + match terminal.state { + State::Done => coord.notify_manager(&hive_sh4re::HelperEvent::Rebuilt { + agent: agent.clone(), + ok: true, + note: None, + sha: None, + tag: None, + }), + State::Failed => coord.notify_manager(&hive_sh4re::HelperEvent::Rebuilt { + agent: agent.clone(), + ok: false, + note: terminal.error.clone(), + sha: None, + tag: None, + }), + _ => {} + } + } + } +} + /// Compute which agents a `nix flake update ` on the meta /// flake affects — the fan-out set for `MetaUpdate` DAGs. Empty /// `inputs` or any input under `hyperhive` → every container; diff --git a/hive-c0re/src/job_queue/mod.rs b/hive-c0re/src/job_queue/mod.rs index e2068cb7..c7362cea 100644 --- a/hive-c0re/src/job_queue/mod.rs +++ b/hive-c0re/src/job_queue/mod.rs @@ -1,94 +1,91 @@ -//! Generic job-DAG queue + desired-state reconciliation — the host-side -//! wrapper over the domain-agnostic [`hive_jobq`] scheduler. Jobs are nodes in -//! per-request DAGs (see [`templates`]); the special cases (graceful-stop -//! watcher, deferred-start follow-up, meta-update cascade) collapse into DAG -//! *shapes* over the shared node primitives ([`model::NodeKind`]). +//! Generic job-DAG queue + desired-state reconciliation — replaces the +//! old flat `rebuild_queue`. Jobs are nodes in per-request DAGs (see +//! [`templates`]); the special cases (graceful-stop watcher thread, +//! deferred-start follow-up, meta-update cascade) collapse into DAG +//! *shapes* over a shared set of primitive nodes ([`model::NodeKind`]). //! -//! [`hive_jobq`] owns the graph, the two-class resource pool, and the roll-up -//! settle loop; this module maps hive-c0re's concepts onto it: -//! - [`model::NodeKind`] **is** the crate payload `N` directly — each variant -//! carries the agent it targets ([`NodeKind::agent`]); the two resource -//! classes are [`resource::Resource`] (`BuildSlot` node-held, `Agent` lease -//! subtree-held), derived per node by [`NodeKind::resource_deps`]; -//! - a **DAG is a single container node** ([`NodeKind::Dag`], `parent = None`) -//! carrying the group's metadata, with the template's nodes hung under it as -//! its subtree (the **parent axis** groups; `deps` order). So the container's -//! `NodeId` is the DAG id, its rolled-up state is the DAG state, and membership -//! is a graph walk — there are no host grouping side-tables. The lease is owned -//! by a subtree root and borrowed by its descendants (continuity); -//! - per-DAG terminal work runs **inline** ([`exec::run_terminal_hook`]) when the -//! container rolls up terminal — dispatched off its template -//! ([`terminal_hook`]): approval-resolve, `Rebuilt`-emit, or power-intent -//! revert. No terminal-hook node, no drained event stream. +//! Concurrency is gated by two resource classes: +//! 1. **Build slots** — N permits (`services.hyperhive.c0re.buildSlots`, +//! default 1) held by nix-heavy nodes for the node's duration. +//! 2. **Per-agent lifecycle lease** — keyed on the *node's* agent and +//! globally exclusive per agent across all DAGs: acquired before a +//! container-affecting node runs, held (by the owning DAG) until no +//! live node of that DAG still targets the agent, so two DAGs never +//! interleave container ops on the same agent. A DAG spanning multiple +//! agents holds one lease per agent it touches. //! -//! The queue is runtime-only (no persistence): an empty graph on boot; desired -//! state is re-derived by the reconcile sweep. A single scheduler task -//! ([`scheduler::run_worker`]) drives it; concurrency comes from the build-slot -//! capacity, not multiple workers. Design: `docs/coordinator.md::Job queue`. +//! The meta *repo* is serialized by `meta::META_LOCK` inside the +//! executors themselves. Per-agent power *intent* (`wanted`) lives in +//! the durable [`crate::power`] store; the DAGs are the reconcile +//! mechanism. Design + rationale: `docs/coordinator.md::Job queue`. pub mod exec; pub mod model; -pub mod resource; pub mod scheduler; pub mod submit; pub mod templates; #[cfg(test)] mod tests; -use std::collections::HashMap; +use std::collections::{HashMap, VecDeque}; use std::sync::Mutex; -use hive_jobq::resources::ResourceTable; -use hive_jobq::scheduler::{Outcome, Scheduler}; -use hive_jobq::{Dep, DepWhen as JobDepWhen, Graph, NodeId, State as JobState}; -use hive_sh4re::jobs::NodeView; use hive_sh4re::wire_time::now_unix; use tokio::sync::Notify; -use crate::coordinator::TransientKind; pub use model::{ - DagSpec, DagView, DepWhen, NodeKind, NodeSpec, PermPayload, Source, State, Template, + Dag, DagSpec, DagView, DepWhen, Node, NodeId, NodeKind, NodeSpec, PermPayload, Source, State, + Template, }; -use resource::Resource; -/// How many terminal DAGs (`Done` / `Failed` / `Cancelled`) to retain per -/// template in the snapshot, matching the old per-kind history cap. +/// How many terminal DAGs (`Done` / `Failed` / `Cancelled`) to retain +/// per template in the snapshot, matching the old per-kind history cap. const MAX_HISTORY_PER_TEMPLATE: usize = 5; -/// Terminal DAGs younger than this are exempt from the per-template history -/// cap, so a burst of same-template DAGs that settle within one `QueueDag` -/// poll interval isn't evicted before the poller observes their terminal state. +/// Terminal DAGs younger than this are exempt from the per-template +/// history cap. A broad `hivectl stop`/`start` submits many +/// same-template DAGs that can all settle within one poll interval — +/// without the grace, the cap would evict some before the ~1s +/// `QueueDag` poller ever observes their terminal state, silently +/// swallowing failures. const HISTORY_GRACE_SECS: i64 = 300; /// Cap on stored node error strings. const MAX_ERROR_LEN: usize = 2_000; -/// A node claimed for execution — everything the executor needs, snapshotted at -/// claim time. +/// A node claimed for execution — everything the executor needs, +/// snapshotted at claim time. #[derive(Debug, Clone)] pub struct Claim { pub dag_id: u64, pub node_id: NodeId, pub kind: NodeKind, - /// The agent this node targets (its own, not a DAG-level field). Empty for - /// the agentless [`NodeKind::MetaLock`] + [`NodeKind::Dag`] container nodes. + /// The agent this node targets (its own, not a DAG-level field). The + /// executor operates on this agent's container; the lease is keyed on + /// it. pub agent: String, pub template: Template, pub approval_id: Option, pub inputs: Vec, - /// Transient pill kind for the lease window (from the spec). Whether the - /// pill is currently shown is derived from live lease ownership - /// ([`JobQueue::held_transients`]), not a per-claim edge. - pub transient: Option, + pub perm_payload: Option, + /// True when claiming this node newly acquired its agent's lease — + /// the scheduler creates the per-`(dag, agent)` transient guard on + /// this edge. + pub lease_acquired: bool, + /// Transient pill kind for the lease window (from the spec). + pub transient: Option, } -/// Summary of a DAG's terminal roll-up — the input to the terminal node's -/// executor (approval resolution, `Rebuilt` emission, cancelled-power-op intent -/// revert). Computed on demand from live graph state, not drained. +/// Summary of a DAG that just reached its terminal roll-up state — +/// input to the approval-resolution hook and the lease/transient +/// release. #[derive(Debug, Clone)] pub struct TerminalDag { + pub dag_id: u64, pub template: Template, - /// Distinct agents this DAG's nodes targeted (one for a single-agent DAG). + /// Distinct agents this DAG's nodes targeted (one for a single-agent + /// DAG). The cancel-revert hook walks these to snap each agent's + /// power intent back on a cancelled power-op DAG. pub agents: Vec, pub approval_id: Option, pub state: State, @@ -96,781 +93,559 @@ pub struct TerminalDag { pub error: Option, } -/// Per-node runtime metadata the crate graph doesn't carry (kind + agent live -/// in the node payload; state lives in the node). -#[derive(Debug, Default, Clone)] -struct NodeRuntime { - step: Option, - build_log_id: Option, - started_at: Option, - finished_at: Option, - error: Option, +/// A single per-agent lease release: agent `agent`'s subgraph within DAG +/// `dag_id` just reached terminal, so the scheduler drops that agent's +/// `(dag_id, agent)` transient guard — ahead of (or coinciding with) the +/// whole-DAG [`TerminalDag`]. The lease itself is freed inside `settle`; +/// this only carries the transient-drop signal out to the scheduler. +#[derive(Debug, Clone)] +pub struct AgentRelease { + pub dag_id: u64, + pub agent: String, } -/// An owned read-view of a DAG container's carried metadata ([`NodeKind::Dag`]). -/// Derived on read from the container node — the data has a single home (the -/// node payload); this is not a stored side-table. -struct DagMeta { - template: Template, - source: Source, - reason: String, - transient: Option, - approval_id: Option, - inputs: Vec, - created_at: i64, +#[derive(Debug, Default)] +struct Inner { + dags: VecDeque, + next_id: u64, + build_slots: usize, + slots_used: usize, + /// agent → dag id currently holding that agent's lifecycle lease. + leases: HashMap, + /// Terminal roll-ups not yet consumed by the scheduler + /// ([`JobQueue::drain_terminal`]). Fed by every path that settles + /// state — node completion AND the cancel surfaces — so the + /// terminal hooks (approval resolution, intent revert, transient + /// release) fire exactly once per DAG no matter how it ended. + pending_terminal: Vec, + /// Per-agent lease releases not yet consumed by the scheduler + /// ([`JobQueue::drain_agent_releases`]). Fed by `settle` the moment + /// an agent's subgraph within a DAG goes terminal — earlier than the + /// whole-DAG `pending_terminal` for a multi-agent DAG. Drives the + /// per-agent transient-guard drop. + pending_agent_release: Vec, } -/// The mutable queue state behind the mutex: the crate scheduler plus the -/// per-node runtime metadata the graph can't carry. A **DAG is a single -/// container node** ([`NodeKind::Dag`], `parent = None`) whose subtree is the -/// DAG's work — so the container's `NodeId` is the DAG id, its rolled-up state -/// is the DAG state, and there are no grouping side-tables: membership + meta -/// are graph queries ([`QueueInner::container`] / [`QueueInner::subtree`] / -/// [`QueueInner::dag_meta`]). One shared crate [`Graph`] holds every DAG. -struct QueueInner { - sched: Scheduler, - /// Per-node runtime metadata (build-log id, step, timestamps, error) — - /// mutable after insert, so it can't ride the immutable node payload. - node_rt: HashMap, -} - -/// The queue. Lives on `Coordinator` (one per hive-c0re process); a single -/// scheduler task ([`scheduler::run_worker`]) drives it. +/// The queue. Lives on `Coordinator` (one per hive-c0re process); a +/// single scheduler task ([`scheduler::run_worker`]) drives it — +/// concurrency comes from the build-slot count, not multiple workers. +#[derive(Debug)] pub struct JobQueue { - inner: Mutex, + inner: Mutex, /// Wakes the scheduler when something new arrives or state changed. pub(crate) notify: Notify, } -impl std::fmt::Debug for JobQueue { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("JobQueue").finish_non_exhaustive() - } -} - impl Default for JobQueue { fn default() -> Self { Self::new(1) } } -/// Map a spec dependency edge kind onto the crate's. -fn to_crate_when(when: DepWhen) -> JobDepWhen { - match when { - DepWhen::AfterOk => JobDepWhen::AfterOk, - DepWhen::AfterAny => JobDepWhen::AfterAny, - } -} - -/// The inline terminal-hook a settled DAG fires — dispatched off its container's -/// template + approval id when the container rolls up terminal (no hook node). -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum HookKind { - /// Approval-driven DAG (spawn / opaque deploy): resolve the approval row. - ResolveApproval, - /// Rebuild / perm-change: emit one `Rebuilt` manager event per agent. - EmitRebuilt, - /// Power-op: on a *cancelled* DAG, revert each agent's `wanted` intent. - RevertIntent, -} - -/// The terminal hook a DAG needs, from its template + approval id — or `None` -/// for a DAG with no terminal side effect (meta-update, boot, bare reconcile). -#[must_use] -pub fn terminal_hook(template: Template, approval_id: Option) -> Option { - if approval_id.is_some() { - return Some(HookKind::ResolveApproval); - } - match template { - Template::Rebuild | Template::PermChange => Some(HookKind::EmitRebuilt), - Template::Start - | Template::Stop - | Template::GracefulStop - | Template::Restart - | Template::GracefulRestart => Some(HookKind::RevertIntent), - _ => None, - } -} - -/// Map a crate node state onto the wire state (`Pending` ↔ `Queued`; -/// `Finishing` — own logic done, sub-nodes still running — reads as `Running`). -fn to_wire_state(state: JobState) -> State { - match state { - JobState::Pending => State::Queued, - JobState::Running | JobState::Finishing => State::Running, - JobState::Done => State::Done, - JobState::Failed => State::Failed, - JobState::Cancelled => State::Cancelled, - } -} - -/// Insert `nodes` into the shared graph, honouring the spec's explicit **parent -/// axis**: a node with `parent = None` is a top-level group root (re-parented to -/// `group_parent`, which is `None` for `submit` and the emitting node for -/// `append_subgraph`); a node with `parent = Some(idx)` becomes a child of the -/// already-inserted node at spec index `idx`. `deps` are translated to crate -/// `Dep::Node` edges verbatim — templates declare the parent axis + sibling -/// ordering directly, so there is no dep-on-root to drop and no lease to hoist: -/// each node declares its own `Dep::Resource`, and the crate's borrow model -/// keeps a resource continuous across a subtree (a root owns it, descendants -/// borrow it). Independent group roots (multiple `parent = None` nodes) carry no -/// cross-links, so a multi-agent DAG's per-agent subgraphs run concurrently, each -/// on its own lease. Records per-node `node_rt`. Returns the inserted ids -/// (index-aligned with `nodes`). A node with `parent = None` is re-parented to -/// `group_parent` (the DAG container for a template, or the emitting node for a -/// runtime-appended subgraph); a node's `parent` / dep targets must precede it -/// in `nodes` (submit-time `validate` enforces density + acyclicity). -/// -/// # Errors -/// Propagates a crate graph-insert error (malformed dep/parent / dep-scope). -fn insert_group( - inner: &mut QueueInner, - nodes: &[NodeSpec], - group_parent: Option, -) -> anyhow::Result> { - let mut ids: Vec = Vec::with_capacity(nodes.len()); - for ns in nodes { - let payload = ns.kind.clone(); - let mut deps = payload.resource_deps(); - for d in &ns.deps { - deps.push(Dep::Node { - id: ids[dep_index(d.on)], - when: to_crate_when(d.when), - }); - } - let parent = match ns.parent { - Some(idx) => Some(ids[dep_index(idx)]), - None => group_parent, - }; - let id = inner - .sched - .append(payload, deps, parent) - .map_err(|e| anyhow::anyhow!("job_queue: graph insert failed: {e}"))?; - ids.push(id); - inner.node_rt.insert(id, NodeRuntime::default()); - } - Ok(ids) -} - impl JobQueue { - #[must_use] pub fn new(build_slots: usize) -> Self { - let mut table = ResourceTable::new(); - table.set_capacity( - Resource::BuildSlot, - u32::try_from(build_slots.max(1)).unwrap_or(u32::MAX), - ); Self { - inner: Mutex::new(QueueInner { - sched: Scheduler::new(Graph::new(), table), - node_rt: HashMap::new(), + inner: Mutex::new(Inner { + build_slots: build_slots.max(1), + ..Inner::default() }), notify: Notify::new(), } } - fn lock(&self) -> std::sync::MutexGuard<'_, QueueInner> { - self.inner.lock().expect("job_queue mutex poisoned") - } - - /// Submit a DAG. Validates the spec, inserts a [`NodeKind::Dag`] **container - /// node** carrying the group's metadata, then inserts the template's nodes as - /// its subtree (their roots re-parented to the container). Returns the - /// container's id as the DAG id — its rolled-up state is the DAG state and it - /// reaching terminal fires the DAG's inline hook. + /// Submit a DAG. Validates the spec (cycle rejection) and returns the + /// newly-allocated DAG id. /// - /// # Errors - /// Propagates the spec-validation error (empty / cyclic / bad parent) or a - /// graph-insert error (dependencies that aren't dependency-topological). + /// Submit-time dedup was removed with the agent-per-node refactor + /// (a multi-agent DAG has no single agent to key a dedup on) — every + /// submit now enqueues a fresh DAG. Whether any dedup needs + /// reintroducing (and in what form) is tracked as a follow-up; see the + /// dedup re-evaluation issue. pub fn submit(&self, spec: DagSpec) -> anyhow::Result { templates::validate(&spec)?; - let mut inner = self.lock(); - let container = inner - .sched - .append( - NodeKind::Dag { - template: spec.template, - source: spec.source, - reason: spec.reason, - transient: spec.transient, - approval_id: spec.approval_id, - inputs: spec.inputs, - created_at: now_unix(), - }, - Vec::new(), - None, - ) - .map_err(|e| anyhow::anyhow!("job_queue: container insert failed: {e}"))?; - inner.node_rt.insert(container, NodeRuntime::default()); - insert_group(&mut inner, &spec.nodes, Some(container))?; - // Settle the container's own (no-op) logic immediately so it parks in - // `Finishing` and its children become runnable — it never needs claiming - // or executing, and stays out of `claim_ready`. It rolls up terminal when - // its whole subtree settles (that's the DAG-done signal). - inner.sched.complete(container, Outcome::Done); + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let id = Self::push_dag(&mut inner, spec); drop(inner); self.notify.notify_one(); - Ok(container.get()) + Ok(id) } - /// Append a whole *subgraph* into a live DAG at runtime — the single - /// in-DAG-growth primitive. The subgraph is inserted as a [`insert_group`] - /// rooted under `dep_on` (the emitting node): the subgraph's own root becomes - /// a *child* of `dep_on`, its steps children of that root, and the group's - /// agent lease is hoisted onto that root. Ordering root→`dep_on` is the parent - /// gate — the children run once `dep_on` reaches `Finishing`. Because the - /// emitting node stays `Finishing` until this appended subtree is terminal and - /// the DAG's terminal node deps on the top root, roll-up keeps the DAG from - /// settling early with no explicit wiring. Returns the new node ids; empty if - /// the DAG is gone or `nodes` is empty. - pub fn append_subgraph(&self, dag_id: u64, nodes: &[NodeSpec], dep_on: NodeId) -> Vec { + /// Append a whole *subgraph* into a live (non-terminal) DAG at runtime — + /// the single in-DAG-growth primitive. Each [`NodeSpec`] carries its own + /// `agent` and subgraph-relative `deps` (indices into `nodes`); this + /// rebases those onto the DAG's node-id space (`id == index`) and attaches + /// every subgraph *root* — a node with no internal deps — to `dep_on` with + /// an `AfterOk` edge. Used both for multi-node growth (the `MetaLock` + /// growing per-agent rebuild subgraphs into the same boot / meta-update + /// DAG instead of fanning out child DAGs) and the single-node case (a + /// `Reconcile` planner's `Start` / `Stop` as a one-node subgraph). Must be + /// called *before* the emitting node's [`Self::complete_node`] so the DAG + /// can't roll terminal with the appended work still pending. Returns the + /// new node ids; empty if the DAG is gone or `nodes` is empty. + pub fn append_subgraph( + &self, + dag_id: u64, + nodes: Vec, + dep_on: NodeId, + ) -> Vec { if nodes.is_empty() { return Vec::new(); } - let mut inner = self.lock(); - if inner.container(dag_id).is_none() { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(dag) = inner.dags.iter_mut().find(|d| d.id == dag_id) else { return Vec::new(); - } - // Insert the subgraph as a group rooted under the emitting node: the - // subgraph's own root becomes a child of `dep_on`, its steps children of - // that root. No terminal-node wiring — roll-up carries terminality: the - // emitter stays `Finishing` until this appended subtree settles, and the - // container node rolls up terminal only once its whole subtree (incl. this - // appended work) has settled, so the DAG hook waits for free. - let ids = match insert_group(&mut inner, nodes, Some(dep_on)) { - Ok(ids) => ids, - Err(e) => { - tracing::error!( - dag = dag_id, - error = %e, - "job_queue: append_subgraph insert failed" - ); - return Vec::new(); - } }; + let base: NodeId = u32::try_from(dag.nodes.len()).unwrap_or(u32::MAX); + let mut new_ids = Vec::with_capacity(nodes.len()); + for (i, spec) in nodes.into_iter().enumerate() { + let new_id: NodeId = base + u32::try_from(i).unwrap_or(u32::MAX); + // Subgraph roots (no internal deps) hang off the emitting node; + // internal deps rebase from subgraph-relative onto the DAG id + // space (both start at `base`). + let deps = if spec.deps.is_empty() { + vec![model::Dep { + on: dep_on, + when: DepWhen::AfterOk, + }] + } else { + spec.deps + .into_iter() + .map(|d| model::Dep { + on: base + d.on, + when: d.when, + }) + .collect() + }; + dag.nodes.push(Node { + id: new_id, + agent: spec.agent, + kind: spec.kind, + deps, + state: State::Queued, + step: None, + build_log_id: None, + started_at: None, + finished_at: None, + error: None, + }); + new_ids.push(new_id); + } drop(inner); self.notify.notify_one(); - ids + new_ids } - /// Claim every currently-runnable node, acquiring its resources, and mark it - /// `Running`. Delegates readiness + resource acquisition to the crate's - /// settle loop; builds a [`Claim`] per started node from its payload + its - /// DAG container's metadata. The container node itself is claimed like any - /// other (its executor is an instant no-op that lets its subtree start). + fn push_dag(inner: &mut Inner, spec: DagSpec) -> u64 { + inner.next_id += 1; + let id = inner.next_id; + let nodes = spec + .nodes + .into_iter() + .enumerate() + .map(|(i, n)| Node { + id: u32::try_from(i).unwrap_or(u32::MAX), + agent: n.agent, + kind: n.kind, + deps: n.deps, + state: State::Queued, + step: None, + build_log_id: None, + started_at: None, + finished_at: None, + error: None, + }) + .collect(); + inner.dags.push_back(Dag { + id, + template: spec.template, + source: spec.source, + reason: spec.reason, + approval_id: spec.approval_id, + inputs: spec.inputs, + perm_payload: spec.perm_payload, + transient: spec.transient, + created_at: now_unix(), + nodes, + terminal_reported: false, + }); + id + } + + /// Claim every currently-ready node, acquiring resources, and mark + /// them `Running`. A node is ready when it's `Queued`, every dep is + /// satisfied (`AfterOk`: dep `Done`; `AfterAny`: dep terminal), and + /// its resources are free (build slot; agent lease free or already + /// held by this DAG). Iteration is in DAG-submit order, so + /// simultaneously-ready nodes compete FIFO — bulk operations drain + /// predictably. pub fn claim_ready(&self) -> Vec { - let mut inner = self.lock(); + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + Self::propagate_cancellations(&mut inner); + let mut claims = Vec::new(); let inner = &mut *inner; - let started = inner.sched.settle(); - let now = now_unix(); - let mut claims = Vec::with_capacity(started.len()); - for id in started { - let Some(node) = inner.sched.graph().node(id) else { - continue; - }; - let kind = node.payload.clone(); - let agent = node.payload.agent().to_owned(); - let Some(container) = inner.dag_of(id) else { - continue; - }; - let Some(meta) = inner.dag_meta(container) else { - continue; - }; - claims.push(Claim { - dag_id: container.get(), - node_id: id, - kind, - agent, - template: meta.template, - approval_id: meta.approval_id, - inputs: meta.inputs, - transient: meta.transient, - }); - if let Some(rt) = inner.node_rt.get_mut(&id) { - rt.started_at = Some(now); + for di in 0..inner.dags.len() { + // Split-borrow dance: deps are checked against the same + // DAG's other nodes, so snapshot the states first. + let dag = &inner.dags[di]; + let dag_id = dag.id; + let ready_ids: Vec = dag + .nodes + .iter() + .filter(|n| n.state == State::Queued && Self::deps_satisfied(dag, n)) + .map(|n| n.id) + .collect(); + for node_id in ready_ids { + let dag = &inner.dags[di]; + let node = dag.node(node_id).expect("node id from same dag"); + let needs_slot = node.kind.needs_build_slot(); + let needs_lease = node.kind.needs_lease(); + // The lifecycle lease is keyed on the *node's* agent, held + // by this DAG (dag_id) — still globally exclusive per agent + // across all DAGs. A multi-agent DAG acquires one lease per + // agent it touches; each is released in `settle` once no + // live node of this DAG still targets that agent. + let node_agent = node.agent.clone(); + if needs_slot && inner.slots_used >= inner.build_slots { + continue; + } + let mut lease_acquired = false; + if needs_lease { + match inner.leases.get(node_agent.as_str()) { + Some(&holder) if holder != dag_id => continue, + Some(_) => {} + None => { + inner.leases.insert(node_agent.clone(), dag_id); + lease_acquired = true; + } + } + } + if needs_slot { + inner.slots_used += 1; + } + let dag = &mut inner.dags[di]; + let claim = Claim { + dag_id, + node_id, + kind: dag.node(node_id).expect("node").kind.clone(), + agent: node_agent, + template: dag.template, + approval_id: dag.approval_id, + inputs: dag.inputs.clone(), + perm_payload: dag.perm_payload.clone(), + lease_acquired, + transient: dag.transient, + }; + let node = dag.node_mut(node_id).expect("node"); + node.state = State::Running; + node.started_at = Some(now_unix()); + claims.push(claim); } } claims } - /// Mark a claimed node terminal, recording its outcome + (truncated) error. - /// The crate releases the node's build slot immediately and cascades the - /// `AfterOk` failure cancellation + subtree lease release. Returns the DAG's - /// terminal summary **iff** this completion rolled its container terminal — - /// the scheduler runs the DAG's inline hook off it. - pub fn complete_node( - &self, - _dag_id: u64, - node_id: NodeId, - result: Result<(), String>, - ) -> Option { - let mut inner = self.lock(); - let now = now_unix(); - let (error, outcome) = match result { - Ok(()) => (None, Outcome::Done), - Err(e) => (Some(truncate_error(&e)), Outcome::Failed), - }; - if let Some(rt) = inner.node_rt.get_mut(&node_id) { - rt.finished_at = Some(now); - rt.step = None; - if let Some(e) = error { - rt.error = Some(e); + fn deps_satisfied(dag: &Dag, node: &Node) -> bool { + node.deps.iter().all(|dep| { + dag.node(dep.on).is_some_and(|d| match dep.when { + DepWhen::AfterOk => d.state == State::Done, + DepWhen::AfterAny => d.state.is_terminal(), + }) + }) + } + + /// Cancel-downstream: a `Queued` node with an `AfterOk` dep that + /// `Failed` / `Cancelled` becomes `Cancelled` itself. Loops to a + /// fixpoint so the cancellation cascades through chains. + fn propagate_cancellations(inner: &mut Inner) { + for dag in &mut inner.dags { + loop { + let doomed: Vec = dag + .nodes + .iter() + .filter(|n| { + n.state == State::Queued + && n.deps.iter().any(|dep| { + dep.when == DepWhen::AfterOk + && dag.node(dep.on).is_some_and(|d| { + matches!(d.state, State::Failed | State::Cancelled) + }) + }) + }) + .map(|n| n.id) + .collect(); + if doomed.is_empty() { + break; + } + let now = now_unix(); + for id in doomed { + if let Some(n) = dag.node_mut(id) { + n.state = State::Cancelled; + n.finished_at = Some(now); + } + } } } - let container = inner.dag_of(node_id); - inner.sched.complete(node_id, outcome); - // If this completion rolled the DAG's container up to a terminal state, - // hand its summary back so the scheduler fires the inline hook once. - let terminal = container - .filter(|&c| c != node_id && inner.dag_is_terminal(c)) - .and_then(|c| inner.terminal_dag(c)); - drop(inner); - self.notify.notify_one(); - terminal } - /// Cancel a DAG that hasn't started yet: every work node is still `Pending`, - /// so each is cancelled. `None` once any work node is running or terminal — - /// an in-flight nix build isn't interruptible. Otherwise the container is - /// rolled up so the DAG settles (wire state `Cancelled`) and its terminal - /// summary is returned — the caller fires the inline hook (power-intent - /// revert / approval resolution) off it. - pub fn cancel(&self, dag_id: u64) -> Option { - let mut inner = self.lock(); - let container = inner.container(dag_id)?; - let work = inner.subtree(container); - let all_pending = work.iter().all(|&id| { - inner - .sched - .graph() - .node(id) - .is_some_and(|n| n.state == JobState::Pending) - }); - if !all_pending { - return None; + /// Mark a claimed node terminal, release its build slot, cascade + /// cancellations, and settle terminal DAGs (lease release + history + /// trim; the terminal roll-up lands in the [`Self::drain_terminal`] + /// buffer). `error` is stored (truncated) when `result` is `Err`. + pub fn complete_node(&self, dag_id: u64, node_id: NodeId, result: Result<(), String>) { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + if let Some(dag) = inner.dags.iter_mut().find(|d| d.id == dag_id) + && let Some(node) = dag.node_mut(node_id) + && node.state == State::Running + { + let needs_slot = node.kind.needs_build_slot(); + node.finished_at = Some(now_unix()); + node.step = None; + match result { + Ok(()) => node.state = State::Done, + Err(e) => { + node.state = State::Failed; + let mut msg = e; + if msg.len() > MAX_ERROR_LEN { + msg.truncate( + (0..=MAX_ERROR_LEN) + .rev() + .find(|i| msg.is_char_boundary(*i)) + .unwrap_or(0), + ); + msg.push('…'); + } + node.error = Some(msg); + } + } + if needs_slot { + inner.slots_used = inner.slots_used.saturating_sub(1); + } } - for id in work { - inner.sched.cancel_node(id); - } - // The container was settled to `Finishing` at submit; completing it again - // now re-runs the roll-up with its children all `Cancelled`, driving it to - // a terminal state synchronously within this lock — so the caller reads - // the terminal summary immediately instead of waiting for the scheduler - // loop to observe the cancellation. `dag_rollup` reports `Cancelled` to - // the wire (a container whose children all cancelled). - inner.sched.complete(container, Outcome::Done); - let terminal = inner.terminal_dag(container); + Self::settle(&mut inner); drop(inner); self.notify.notify_one(); - terminal } - /// Set the step label on a `Running` node. Returns `true` when it changed. - pub fn set_step(&self, dag_id: u64, node_id: NodeId, step: &str) -> bool { - let mut inner = self.lock(); - if inner.dag_of(node_id).map(NodeId::get) != Some(dag_id) || !inner.node_running(node_id) { + /// Take the terminal roll-ups accumulated since the last drain. + /// The scheduler calls this after every wakeup and runs the + /// terminal hooks on each entry. + pub fn drain_terminal(&self) -> Vec { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + std::mem::take(&mut inner.pending_terminal) + } + + /// Take the per-agent lease releases accumulated since the last + /// drain. The scheduler calls this every wakeup and drops the + /// matching `(dag_id, agent)` transient guard for each — freeing an + /// agent's dashboard pill the moment its subgraph settles, ahead of + /// the whole-DAG terminal for a multi-agent DAG. + pub fn drain_agent_releases(&self) -> Vec { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + std::mem::take(&mut inner.pending_agent_release) + } + + /// Propagate cancellations, release each agent's lease the moment its + /// own subgraph settles (not at whole-DAG terminal), buffer each + /// terminal roll-up exactly once (the `terminal_reported` flag) for + /// [`Self::drain_terminal`], and trim history. + fn settle(inner: &mut Inner) { + Self::propagate_cancellations(inner); + + // Per-agent early lease release. A lease gates an agent's + // container globally, so it must be held for exactly as long as + // that agent's work in the DAG is in flight — no longer. The + // moment no live node of a DAG still targets an agent, free that + // agent's lease so a concurrent DAG wanting the same agent can + // proceed, even while the rest of this DAG runs on. For a + // single-agent DAG the agent's subgraph goes terminal exactly + // when the whole DAG does, so this reduces to the old behaviour. + // Runs over ALL dags, gated on this dag actually holding the + // lease, so it fires exactly once per (dag, agent). + let mut released: Vec = Vec::new(); + for dag in &inner.dags { + for agent in dag.agents() { + if inner.leases.get(agent.as_str()) == Some(&dag.id) + && dag.agent_subgraph_terminal(&agent) + { + released.push(AgentRelease { + dag_id: dag.id, + agent, + }); + } + } + } + for rel in &released { + inner.leases.remove(&rel.agent); + } + inner.pending_agent_release.extend(released); + + // Whole-DAG terminal roll-up (approval resolution, intent revert, + // `Rebuilt` events) — still fires once per DAG. Leases are already + // freed by the per-agent pass above by the time we get here. + let mut reports: Vec = Vec::new(); + for dag in &mut inner.dags { + if !dag.is_terminal() || dag.terminal_reported { + continue; + } + dag.terminal_reported = true; + reports.push(TerminalDag { + dag_id: dag.id, + template: dag.template, + agents: dag.agents(), + approval_id: dag.approval_id, + state: dag.rollup(), + error: dag.first_error().map(str::to_owned), + }); + } + inner.pending_terminal.append(&mut reports); + Self::trim_history(inner, now_unix() - HISTORY_GRACE_SECS); + } + + /// Cancel a DAG that hasn't started yet (roll-up `Queued`): every + /// node flips to `Cancelled`. No-op (returns `false`) once any node + /// is running or terminal — an in-flight nix build isn't + /// interruptible, matching the old queue's rule. + pub fn cancel(&self, dag_id: u64) -> bool { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(dag) = inner.dags.iter_mut().find(|d| d.id == dag_id) else { + return false; + }; + if dag.rollup() != State::Queued { return false; } - let rt = inner.node_rt.entry(node_id).or_default(); - if rt.step.as_deref() == Some(step) { - return false; + let now = now_unix(); + for n in &mut dag.nodes { + n.state = State::Cancelled; + n.finished_at = Some(now); } - rt.step = Some(step.to_owned()); + // Settle buffers the terminal roll-up; the notify wakes the + // scheduler, which drains it and fires the terminal hooks + // (approval resolution, power-intent revert). + Self::settle(&mut inner); + drop(inner); + self.notify.notify_one(); true } - /// Set the step label on the DAG's currently-running node — the DAG-id-only - /// compatibility surface for the opaque approval pipeline. - pub fn set_step_running(&self, dag_id: u64, step: &str) -> bool { - let mut inner = self.lock(); - let Some(node_id) = inner.running_node_of(dag_id) else { + /// Set the step label on a `Running` node. Returns `true` when the + /// label actually changed (callers emit a snapshot only then). + pub fn set_step(&self, dag_id: u64, node_id: NodeId, step: &str) -> bool { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(node) = inner + .dags + .iter_mut() + .find(|d| d.id == dag_id) + .and_then(|d| d.node_mut(node_id)) + else { return false; }; - let rt = inner.node_rt.entry(node_id).or_default(); - if rt.step.as_deref() == Some(step) { + if node.state != State::Running || node.step.as_deref() == Some(step) { return false; } - rt.step = Some(step.to_owned()); + node.step = Some(step.to_owned()); + true + } + + /// Set the step label on the DAG's currently-running node — + /// compatibility surface for the opaque approval pipeline, whose + /// callbacks only know the DAG id. Single-node approval DAGs make + /// this exact. + pub fn set_step_running(&self, dag_id: u64, step: &str) -> bool { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(node) = inner + .dags + .iter_mut() + .find(|d| d.id == dag_id) + .and_then(|d| d.nodes.iter_mut().find(|n| n.state == State::Running)) + else { + return false; + }; + if node.step.as_deref() == Some(step) { + return false; + } + node.step = Some(step.to_owned()); true } /// Link a `build_logs` row to a specific `Running` node. pub fn set_build_log_id(&self, dag_id: u64, node_id: NodeId, log_id: i64) -> bool { - let mut inner = self.lock(); - if inner.dag_of(node_id).map(NodeId::get) != Some(dag_id) || !inner.node_running(node_id) { - return false; - } - inner.node_rt.entry(node_id).or_default().build_log_id = Some(log_id); - true - } - - /// Link a `build_logs` row to the DAG's currently-running node — DAG-id-only - /// compatibility surface (approval pipeline callbacks). - pub fn set_build_log_id_running(&self, dag_id: u64, log_id: i64) -> bool { - let mut inner = self.lock(); - let Some(node_id) = inner.running_node_of(dag_id) else { - return false; - }; - inner.node_rt.entry(node_id).or_default().build_log_id = Some(log_id); - true - } - - /// A DAG's terminal roll-up summary, computed on demand from its container. - /// `None` if the DAG id is unknown. Test-only — production reads the summary - /// `complete_node` returns when the container rolls up terminal. - #[cfg(test)] - #[must_use] - pub(crate) fn terminal_summary(&self, dag_id: u64) -> Option { - let inner = self.lock(); - let container = inner.container(dag_id)?; - inner.terminal_dag(container) - } - - /// The `(dag_id, agent, kind)` triples for every per-agent lease currently - /// held by a DAG that carries a transient pill — the live transient-pill - /// set, a pull query over crate resource ownership (replaces the old - /// lease-release event stream). A DAG with no transient kind is omitted. - #[must_use] - pub fn held_transients(&self) -> Vec<(u64, String, TransientKind)> { - let inner = self.lock(); - inner - .sched - .resource_state() - .into_iter() - .filter_map(|(res, holder)| { - let Resource::Agent(agent) = res else { - return None; - }; - let container = inner.dag_of(holder)?; - let kind = inner.dag_meta(container)?.transient?; - Some((container.get(), agent, kind)) - }) - .collect() - } - - /// Snapshot every live + retained DAG for `/api/state` + `RebuildQueueChanged`. - #[must_use] - pub fn snapshot(&self) -> Vec { - self.snapshot_capped(now_unix() - HISTORY_GRACE_SECS) - } - - /// Snapshot the visible DAG set (live + newest-per-template terminal, terminal - /// ones after `grace_cutoff` always kept), sorted by container id. - fn snapshot_capped(&self, grace_cutoff: i64) -> Vec { - let inner = self.lock(); - let mut ids = inner.visible_dags(grace_cutoff); - ids.sort_unstable_by_key(|c| c.get()); - ids.into_iter().filter_map(|c| inner.dag_view(c)).collect() - } - - /// Number of live (non-terminal) DAGs — tests + diagnostics. - #[cfg(test)] - #[must_use] - pub fn live_count(&self) -> usize { - let inner = self.lock(); - inner - .containers() - .into_iter() - .filter(|&c| !inner.dag_is_terminal(c)) - .count() - } - - /// Test hook: snapshot with the history grace window disabled, so the - /// per-template cap applies to just-finished terminal DAGs too. - #[cfg(test)] - #[must_use] - pub(crate) fn snapshot_no_grace(&self) -> Vec { - self.snapshot_capped(i64::MAX) - } -} - -impl QueueInner { - /// Whether `id` is a `Running` node. - fn node_running(&self, id: NodeId) -> bool { - self.sched - .graph() - .node(id) - .is_some_and(|n| n.state == JobState::Running) - } - - /// The container node of `dag_id` — the `NodeKind::Dag` root whose id equals - /// `dag_id`. `NodeId` is un-fabricable from a raw `u64`, so this is a search. - fn container(&self, dag_id: u64) -> Option { - self.sched.graph().nodes().find_map(|n| { - (n.parent.is_none() - && n.id.get() == dag_id - && matches!(n.payload, NodeKind::Dag { .. })) - .then_some(n.id) - }) - } - - /// The DAG container a node belongs to — walk its parent chain to the root - /// (`parent == None`), which is the container. Returns `id` itself for a - /// container node. - fn dag_of(&self, id: NodeId) -> Option { - let mut cur = id; - loop { - match self.sched.graph().node(cur)?.parent { - Some(p) => cur = p, - None => return Some(cur), - } - } - } - - /// The DAG's work nodes — its `container`'s subtree, excluding the container. - fn subtree(&self, container: NodeId) -> Vec { - self.sched - .graph() - .nodes() - .filter(|n| n.id != container && self.dag_of(n.id) == Some(container)) - .map(|n| n.id) - .collect() - } - - /// The container's carried domain metadata as an owned read-view. The data - /// lives solely in the [`NodeKind::Dag`] payload — this is a derived read, - /// not a stored side-table. - fn dag_meta(&self, container: NodeId) -> Option { - let NodeKind::Dag { - template, - source, - reason, - transient, - approval_id, - inputs, - created_at, - } = &self.sched.graph().node(container)?.payload + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(node) = inner + .dags + .iter_mut() + .find(|d| d.id == dag_id) + .and_then(|d| d.node_mut(node_id)) else { - return None; + return false; }; - Some(DagMeta { - template: *template, - source: *source, - reason: reason.clone(), - transient: *transient, - approval_id: *approval_id, - inputs: inputs.clone(), - created_at: *created_at, - }) - } - - /// The DAG's currently-running work node, if any (the opaque approval - /// pipeline's single-node DAGs make this exact). - fn running_node_of(&self, dag_id: u64) -> Option { - let container = self.container(dag_id)?; - self.subtree(container) - .into_iter() - .find(|&id| self.node_running(id)) - } - - /// Roll-up state over a DAG's work nodes: `Failed` if any failed; else - /// `Running` if any running; else `Queued` if any queued; else `Cancelled` - /// if any cancelled; else `Done`. (Kept eager over the subtree — a failed - /// child shows `Failed` immediately, before the container finishes rolling - /// up — matching the pre-container behaviour.) - fn dag_rollup(&self, container: NodeId) -> State { - let mut any_running = false; - let mut any_queued = false; - let mut any_cancelled = false; - for id in self.subtree(container) { - match self.sched.graph().node(id).map(|n| n.state) { - Some(JobState::Failed) => return State::Failed, - Some(JobState::Running | JobState::Finishing) => any_running = true, - Some(JobState::Pending) => any_queued = true, - Some(JobState::Cancelled) => any_cancelled = true, - Some(JobState::Done) | None => {} - } - } - if any_running { - State::Running - } else if any_queued { - State::Queued - } else if any_cancelled { - State::Cancelled - } else { - State::Done + if node.state != State::Running { + return false; } + node.build_log_id = Some(log_id); + true } - /// True when the DAG has settled — its container has rolled up terminal - /// (equivalent to every work node being terminal). - fn dag_is_terminal(&self, container: NodeId) -> bool { - self.sched - .graph() - .node(container) - .is_some_and(|n| n.state.is_terminal()) + /// Link a `build_logs` row to the DAG's currently-running node — + /// DAG-id-only compatibility surface (approval pipeline callbacks). + pub fn set_build_log_id_running(&self, dag_id: u64, log_id: i64) -> bool { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + let Some(node) = inner + .dags + .iter_mut() + .find(|d| d.id == dag_id) + .and_then(|d| d.nodes.iter_mut().find(|n| n.state == State::Running)) + else { + return false; + }; + node.build_log_id = Some(log_id); + true } - /// Distinct agents a DAG's work nodes target, in first-seen order. - fn dag_agents(&self, container: NodeId) -> Vec { - let mut seen: Vec = Vec::new(); - for id in self.subtree(container) { - if let Some(n) = self.sched.graph().node(id) { - let agent = n.payload.agent(); - if !agent.is_empty() && !seen.iter().any(|s| s == agent) { - seen.push(agent.to_owned()); - } - } - } - seen + /// Snapshot every DAG for `/api/state` + `RebuildQueueChanged`. + pub fn snapshot(&self) -> Vec { + let inner = self.inner.lock().expect("job_queue mutex poisoned"); + inner.dags.iter().map(Dag::view).collect() } - /// First failed work node's stored error, for the roll-up `error` field. - fn dag_first_error(&self, container: NodeId) -> Option { - for id in self.subtree(container) { - if self - .sched - .graph() - .node(id) - .is_some_and(|n| n.state == JobState::Failed) - && let Some(e) = self.node_rt.get(&id).and_then(|r| r.error.clone()) - { - return Some(e); - } - } - None + /// Number of live (non-terminal) DAGs — used by tests and + /// diagnostics. + #[cfg(test)] + pub fn live_count(&self) -> usize { + let inner = self.inner.lock().expect("job_queue mutex poisoned"); + inner.dags.iter().filter(|d| !d.is_terminal()).count() } - /// A DAG's terminal roll-up summary — the input to its inline hook. - fn terminal_dag(&self, container: NodeId) -> Option { - let meta = self.dag_meta(container)?; - Some(TerminalDag { - template: meta.template, - agents: self.dag_agents(container), - approval_id: meta.approval_id, - state: self.dag_rollup(container), - error: self.dag_first_error(container), - }) - } - - /// Rebuild the wire [`DagView`] for a DAG from its container metadata + work - /// nodes + per-node runtime. - fn dag_view(&self, container: NodeId) -> Option { - let meta = self.dag_meta(container)?; - let node_ids = self.subtree(container); - let mut nodes = Vec::with_capacity(node_ids.len()); - let mut started: Vec = Vec::new(); - let mut finished: Vec = Vec::new(); - for &id in &node_ids { - let Some(node) = self.sched.graph().node(id) else { - continue; - }; - let rt = self.node_rt.get(&id); - let deps: Vec = node - .deps - .iter() - .filter_map(|d| match d { - Dep::Node { id, .. } => Some(id.get()), - Dep::Resource { .. } => None, - }) - .collect(); - if let Some(s) = rt.and_then(|r| r.started_at) { - started.push(s); - } - if let Some(fin) = rt.and_then(|r| r.finished_at) { - finished.push(fin); - } - nodes.push(NodeView { - id: id.get(), - agent: node.payload.agent().to_owned(), - kind: node.payload.as_str().to_owned(), - deps, - state: to_wire_state(node.state), - step: rt.and_then(|r| r.step.clone()), - build_log_id: rt.and_then(|r| r.build_log_id), - started_at: rt.and_then(|r| r.started_at), - finished_at: rt.and_then(|r| r.finished_at), - error: rt.and_then(|r| r.error.clone()), - }); - } - let is_terminal = self.dag_is_terminal(container); - Some(DagView { - id: container.get(), - kind: meta.template, - state: self.dag_rollup(container), - source: meta.source, - reason: meta.reason.clone(), - enqueued_at: meta.created_at, - started_at: started.into_iter().min(), - finished_at: if is_terminal { - finished.into_iter().max() - } else { - None - }, - inputs: meta.inputs.clone(), - approval_id: meta.approval_id, - nodes, - }) - } - - /// When a DAG's work node finishes on `finished_at` — the max over its - /// subtree, for the history cap ordering. - fn dag_finished_at(&self, container: NodeId) -> i64 { - self.subtree(container) - .iter() - .filter_map(|id| self.node_rt.get(id).and_then(|r| r.finished_at)) - .max() - .unwrap_or(0) - } - - /// Every DAG container node id in the graph. - fn containers(&self) -> Vec { - self.sched - .graph() - .nodes() - .filter(|n| n.parent.is_none() && matches!(n.payload, NodeKind::Dag { .. })) - .map(|n| n.id) - .collect() - } - - /// The **visible** DAG set for the snapshot: every live (non-terminal) DAG, - /// plus the newest [`MAX_HISTORY_PER_TEMPLATE`] terminal DAGs per template - /// (terminal DAGs finished after `grace_cutoff` are always kept). Crate nodes - /// for evicted DAGs linger in the graph (bounded-prune is a Stage-C - /// follow-up); this filter is what bounds what the dashboard sees. - fn visible_dags(&self, grace_cutoff: i64) -> Vec { - let mut live: Vec = Vec::new(); - let mut terminal: Vec<(NodeId, Template, i64)> = Vec::new(); - for c in self.containers() { - if self.dag_is_terminal(c) { - if let Some(meta) = self.dag_meta(c) { - terminal.push((c, meta.template, self.dag_finished_at(c))); - } - } else { - live.push(c); - } - } - // Newest first so the per-template cap keeps the most recent. - terminal.sort_by(|a, b| b.2.cmp(&a.2).then(b.0.get().cmp(&a.0.get()))); + /// Keep only the newest `MAX_HISTORY_PER_TEMPLATE` terminal DAGs + /// per template. Never evicted: live DAGs; and terminal DAGs that + /// finished after `grace_cutoff` (see [`HISTORY_GRACE_SECS`]). + fn trim_history(inner: &mut Inner, grace_cutoff: i64) { let mut counts: HashMap = HashMap::new(); - let mut kept = live; - for (c, template, finished) in terminal { - let n = counts.entry(template).or_insert(0); - *n += 1; - if *n <= MAX_HISTORY_PER_TEMPLATE || finished > grace_cutoff { - kept.push(c); - } - } - kept + let kept: Vec = inner + .dags + .iter() + .rev() + .filter(|d| { + if !d.is_terminal() { + return true; + } + let finished = d.nodes.iter().filter_map(|n| n.finished_at).max(); + if finished.is_none_or(|t| t > grace_cutoff) { + return true; + } + let n = counts.entry(d.template).or_insert(0); + *n += 1; + *n <= MAX_HISTORY_PER_TEMPLATE + }) + .cloned() + .collect(); + inner.dags = kept.into_iter().rev().collect(); + } + + /// Test hook: trim with the grace window disabled, so eviction + /// behavior is assertable without aging real timestamps. + #[cfg(test)] + pub(crate) fn trim_ignoring_grace(&self) { + let mut inner = self.inner.lock().expect("job_queue mutex poisoned"); + Self::trim_history(&mut inner, i64::MAX); } } - -/// A spec dependency index (`Dep.on`, a wire `u64`) as a `usize` for indexing -/// into the node/id vectors. `templates::validate` guarantees it's in range. -fn dep_index(on: u64) -> usize { - usize::try_from(on).unwrap_or(usize::MAX) -} - -/// Truncate a node error to [`MAX_ERROR_LEN`] on a char boundary, appending `…`. -fn truncate_error(e: &str) -> String { - if e.len() <= MAX_ERROR_LEN { - return e.to_owned(); - } - let cut = (0..=MAX_ERROR_LEN) - .rev() - .find(|i| e.is_char_boundary(*i)) - .unwrap_or(0); - let mut msg = e[..cut].to_owned(); - msg.push('…'); - msg -} diff --git a/hive-c0re/src/job_queue/model.rs b/hive-c0re/src/job_queue/model.rs index bfc66aa0..a20a1ed8 100644 --- a/hive-c0re/src/job_queue/model.rs +++ b/hive-c0re/src/job_queue/model.rs @@ -11,11 +11,9 @@ //! DAG can span agents). See `docs/coordinator.md::Job queue` for the //! full design. -pub use hive_sh4re::jobs::{DagView, NodeId, PermPayload, Source, State, Template}; +pub use hive_sh4re::jobs::{DagView, NodeId, NodeView, PermPayload, Source, State, Template}; use serde::Serialize; -use crate::coordinator::TransientKind; - /// When a dependency edge is considered satisfied. #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] #[serde(rename_all = "snake_case")] @@ -56,12 +54,12 @@ pub enum NodeKind { /// skipped when the container is already down — it only exists to /// shrink the swap's downtime, which a stopped agent doesn't need /// (the sync + dir prep still run; `Swap` builds inline). - Prebuild { agent: String, relock: bool }, + Prebuild { relock: bool }, /// `nixos-container update` profile-swap (requires the container /// stopped). Re-applies nspawn flags + resource limits first — /// rebuild is the reconcile verb. The post-rebuild bookkeeping tail /// lives in the sibling `PostSwap` node. - Swap { agent: String }, + Swap, /// The post-`Swap` bookkeeping tail as a first-class node: rev marker, /// forge + matrix sync, manager kick, container rescan, meta-inputs /// snapshot. Split out of `Swap` for dashboard visibility + retry @@ -71,16 +69,16 @@ pub enum NodeKind { /// recovery still runs. Store/forge/matrix work only — no nix build, so /// build-slot-exempt; the agent lease taken at `Swap` is held across the /// whole chain until `Reconcile` settles, so it's not re-declared here. - PostSwap { agent: String }, + PostSwap, /// First-spawn pre-create provisioning: proposed/applied repos, /// state subvolume, and meta registration (`sync_agents`). Runs /// ahead of `Create` so the `nixos-container create --flake /// meta#` ref resolves. Store/meta-only — no container yet — /// so it's lease- and build-slot-exempt like `Prebuild`. - Provision { agent: String }, + Provision, /// First-spawn `nixos-container create` proper. Assumes the /// upstream `Provision` node already registered the agent in meta. - Create { agent: String }, + Create, /// Meta flake lock bump. `sweep = false`: `meta::lock_update` /// (commit fused, under `META_LOCK`) with the DAG's `inputs`; /// `sweep = true`: `meta::lock_update_hyperhive`, *non-fatal* (a @@ -97,37 +95,36 @@ pub enum NodeKind { /// `Offline` & up, else noop). The mechanical work is not done in /// this node — it fans a child [`NodeKind::Start`] / [`NodeKind::Stop`] /// DAG out at runtime so the sub-step is a first-class DAG node. - Reconcile { agent: String }, + Reconcile, /// Mechanical container start: the start preamble (runtime dir + /// drop-ins), `start_with_fallback`, MCP listener registration, and /// the manager kick. Fanned out by a [`NodeKind::Reconcile`] that /// observed `wanted = Up` and the container down. - Start { agent: String }, + Start, /// Mechanical container stop: `nixos-container` kill, MCP listener /// unregister, and the `Killed` manager notify. Fanned out by a /// [`NodeKind::Reconcile`] that observed `wanted = Offline` and up. - Stop { agent: String }, + Stop, /// Mechanical `nixos-container stop` for the profile swap. Never /// touches `wanted`. Noop if already stopped. - StopForUpdate { agent: String }, + StopForUpdate, /// Set the graceful-stop fence + kick the harness so it runs one /// stop-checkpoint turn. - Signal { agent: String }, + Signal, /// Await the harness clearing the fence, bounded by /// `GRACEFUL_STOP_TIMEOUT`. Resolves ok either way — the /// downstream `Reconcile` performs the actual stop. - Drain { agent: String }, + Drain, /// `set_nspawn_flags` + `set_resource_limits` + daemon-reload. - WriteDropin { agent: String }, - /// Commit `tool-groups.json` / `capabilities.json` per its `payload` - /// (commit fused under `META_LOCK`). The payload rides this node — the only - /// consumer — rather than the generic DAG container. - WritePermFile { agent: String, payload: PermPayload }, + WriteDropin, + /// Commit `tool-groups.json` / `capabilities.json` per the DAG's + /// `perm_payload` (commit fused under `META_LOCK`). + WritePermFile, /// Opaque approval deploy pipeline (`MergeConfigPr`): the two-phase /// prepare/finalize/abort meta deploy stays inside `actions.rs` in v1 — /// deliberately not /// modeled as scheduler nodes (see the design doc §9). - ApprovalDeploy { agent: String }, + ApprovalDeploy, /// Write the agent's durable power intent (`wanted = Up` when `up`, else /// `Offline`) as a first-class DAG node, at the head of a power-op /// template so the downstream `Reconcile` reads it. Replaces the old @@ -141,24 +138,7 @@ pub enum NodeKind { /// the DAG. (In `stale_start` the lease is thus held across the head /// `Prebuild`, but that's a no-op there — the agent is down, so prebuild /// is skipped.) - SetWanted { agent: String, up: bool }, - /// The **DAG container** node: one per submitted DAG, carrying the group's - /// domain metadata. Every template node hangs *under* it (its subtree), so - /// the container's `NodeId` **is** the DAG id, its rolled-up state **is** the - /// DAG state, and it reaching terminal **is** the completion signal that - /// fires the DAG's inline hook (approval-resolve / rebuilt-emit / - /// intent-revert, dispatched off `template`). Pure grouping — lease- and - /// build-slot-exempt; the executor instant-completes it (`Done`) so it - /// reaches `Finishing` and its children start. - Dag { - template: Template, - source: Source, - reason: String, - transient: Option, - approval_id: Option, - inputs: Vec, - created_at: i64, - }, + SetWanted { up: bool }, } impl NodeKind { @@ -166,47 +146,21 @@ impl NodeKind { pub fn as_str(&self) -> &'static str { match self { NodeKind::Prebuild { .. } => "prebuild", - NodeKind::Swap { .. } => "swap", - NodeKind::PostSwap { .. } => "post_swap", - NodeKind::Provision { .. } => "provision", - NodeKind::Create { .. } => "create", + NodeKind::Swap => "swap", + NodeKind::PostSwap => "post_swap", + NodeKind::Provision => "provision", + NodeKind::Create => "create", NodeKind::MetaLock { .. } => "meta_lock", - NodeKind::Reconcile { .. } => "reconcile", - NodeKind::Start { .. } => "start", - NodeKind::Stop { .. } => "stop", - NodeKind::StopForUpdate { .. } => "stop_for_update", - NodeKind::Signal { .. } => "signal", - NodeKind::Drain { .. } => "drain", - NodeKind::WriteDropin { .. } => "write_dropin", - NodeKind::WritePermFile { .. } => "write_perm_file", - NodeKind::ApprovalDeploy { .. } => "approval_deploy", + NodeKind::Reconcile => "reconcile", + NodeKind::Start => "start", + NodeKind::Stop => "stop", + NodeKind::StopForUpdate => "stop_for_update", + NodeKind::Signal => "signal", + NodeKind::Drain => "drain", + NodeKind::WriteDropin => "write_dropin", + NodeKind::WritePermFile => "write_perm_file", + NodeKind::ApprovalDeploy => "approval_deploy", NodeKind::SetWanted { .. } => "set_wanted", - NodeKind::Dag { .. } => "dag", - } - } - - /// The agent this node targets, or `""` for agentless kinds - /// ([`NodeKind::MetaLock`] on the `hyperhive` pseudo-agent, and the - /// [`NodeKind::Dag`] container). - #[must_use] - pub fn agent(&self) -> &str { - match self { - NodeKind::Prebuild { agent, .. } - | NodeKind::Swap { agent } - | NodeKind::PostSwap { agent } - | NodeKind::Provision { agent } - | NodeKind::Create { agent } - | NodeKind::Reconcile { agent } - | NodeKind::Start { agent } - | NodeKind::Stop { agent } - | NodeKind::StopForUpdate { agent } - | NodeKind::Signal { agent } - | NodeKind::Drain { agent } - | NodeKind::WriteDropin { agent } - | NodeKind::WritePermFile { agent, .. } - | NodeKind::ApprovalDeploy { agent } - | NodeKind::SetWanted { agent, .. } => agent, - NodeKind::MetaLock { .. } | NodeKind::Dag { .. } => "", } } @@ -216,10 +170,10 @@ impl NodeKind { matches!( self, NodeKind::Prebuild { .. } - | NodeKind::Swap { .. } - | NodeKind::Create { .. } + | NodeKind::Swap + | NodeKind::Create | NodeKind::MetaLock { .. } - | NodeKind::ApprovalDeploy { .. } + | NodeKind::ApprovalDeploy ) } @@ -234,44 +188,58 @@ impl NodeKind { pub fn needs_lease(&self) -> bool { matches!( self, - NodeKind::Swap { .. } - | NodeKind::Create { .. } - | NodeKind::Reconcile { .. } - | NodeKind::StopForUpdate { .. } - | NodeKind::Signal { .. } - | NodeKind::Drain { .. } - | NodeKind::WriteDropin { .. } - | NodeKind::ApprovalDeploy { .. } + NodeKind::Swap + | NodeKind::Create + | NodeKind::Reconcile + | NodeKind::StopForUpdate + | NodeKind::Signal + | NodeKind::Drain + | NodeKind::WriteDropin + | NodeKind::ApprovalDeploy | NodeKind::SetWanted { .. } ) } } +/// One schedulable unit inside a DAG. +#[derive(Debug, Clone)] +pub struct Node { + pub id: NodeId, + /// The agent this node's work targets. Per-node so a single DAG can + /// span agents (e.g. a hive-wide restart); the lifecycle lease is + /// acquired against *this* agent (still globally exclusive per agent + /// across all DAGs). `"hyperhive"` for meta-level nodes. + pub agent: String, + pub kind: NodeKind, + pub deps: Vec, + pub state: State, + /// Live sub-label while `Running` (kept for parity with the old + /// per-entry `step`). + pub step: Option, + /// Row id of the `build_logs` entry this node opened (`Prebuild` / + /// `Swap` / `ApprovalDeploy`), for the dashboard's live-stream link. + pub build_log_id: Option, + pub started_at: Option, + pub finished_at: Option, + /// Populated when `state == Failed` (truncated by the queue). + pub error: Option, +} + /// Submit-time spec for one node. #[derive(Debug, Clone)] pub struct NodeSpec { - /// The node's payload — [`NodeKind`] is the queue's payload type directly, - /// and each variant carries the agent it targets (a DAG can span agents; - /// the queue derives per-agent leasing from [`NodeKind::agent`]). + /// The agent this node targets (see [`Node::agent`]). Built by the + /// `templates.rs` `node` helper, which stamps the template's agent + /// onto every node. + pub agent: String, pub kind: NodeKind, pub deps: Vec, - /// The **structural parent** axis — the spec-local index of this node's - /// group parent, or `None` for a top-level (group-root) node. Independent - /// of `deps`: `deps` order execution, `parent` groups nodes into a subtree - /// whose resource the whole subtree borrows (the agent lease is owned by a - /// group root and re-entered by its descendants for continuity). A child - /// runs once its parent reaches `Finishing` (the parent gate), so a child - /// never `deps` on its own parent (that would deadlock — dep-scope - /// validation rejects it). - pub parent: Option, } /// Submit-time spec for a whole DAG. Built by `templates.rs`; validated /// (cycle rejection) by `JobQueue::submit`. No DAG-level `agent` — every /// node carries its own (a DAG can span agents), and the queue derives -/// per-agent leasing from [`NodeKind::agent`]. Type-specific payloads -/// (`PermChange`'s file payload) ride the node that consumes them -/// ([`NodeKind::WritePermFile`]), not this generic spec. +/// per-agent leasing from [`NodeSpec::agent`]. #[derive(Debug, Clone)] pub struct DagSpec { pub template: Template, @@ -283,8 +251,147 @@ pub struct DagSpec { /// `MetaUpdate`-only: the inputs to bump (also part of the dedup /// key for that template). Display copy lives on the DAG. pub inputs: Vec, + /// `PermChange`-only payload. + pub perm_payload: Option, /// Dashboard transient pill (and crash-watch suppression) held for /// the lease window — from lease acquisition to DAG terminal. pub transient: Option, pub nodes: Vec, } + +/// A live DAG in the queue. No DAG-level `agent`: agent is per-[`Node`], +/// so a DAG can span agents. Per-agent leasing is derived from the +/// nodes' agents. +#[derive(Debug, Clone)] +pub struct Dag { + pub id: u64, + pub template: Template, + pub source: Source, + pub reason: String, + pub approval_id: Option, + pub inputs: Vec, + pub perm_payload: Option, + pub transient: Option, + pub created_at: i64, + pub nodes: Vec, + /// Terminal roll-up already reported to the scheduler's hooks + /// (approval resolution, transient release). Internal bookkeeping, + /// never serialized. + pub terminal_reported: bool, +} + +impl Dag { + /// Roll-up state: `Failed` if any node failed; else `Running` if + /// any running; else `Queued` if any queued; else `Cancelled` if + /// any cancelled; else `Done`. + pub fn rollup(&self) -> State { + let mut any_cancelled = false; + let mut any_queued = false; + let mut any_running = false; + for n in &self.nodes { + match n.state { + State::Failed => return State::Failed, + State::Running => any_running = true, + State::Queued => any_queued = true, + State::Cancelled => any_cancelled = true, + State::Done => {} + } + } + if any_running { + State::Running + } else if any_queued { + State::Queued + } else if any_cancelled { + State::Cancelled + } else { + State::Done + } + } + + /// True when every node is terminal. + pub fn is_terminal(&self) -> bool { + self.nodes.iter().all(|n| n.state.is_terminal()) + } + + /// True when no live (non-terminal) node of this DAG still targets + /// `agent` — i.e. that agent's subgraph within the DAG has settled. + /// Used to release an agent's lifecycle lease the moment its own + /// work is done, rather than waiting for the whole DAG to terminate. + /// Vacuously true for an agent the DAG has no node for; callers gate + /// on actually holding that agent's lease first. + pub fn agent_subgraph_terminal(&self, agent: &str) -> bool { + self.nodes + .iter() + .filter(|n| n.agent == agent) + .all(|n| n.state.is_terminal()) + } + + /// First failed node's error, for the roll-up `error` field. + pub fn first_error(&self) -> Option<&str> { + self.nodes + .iter() + .find(|n| n.state == State::Failed) + .and_then(|n| n.error.as_deref()) + } + + pub fn node(&self, id: NodeId) -> Option<&Node> { + self.nodes.iter().find(|n| n.id == id) + } + + pub fn node_mut(&mut self, id: NodeId) -> Option<&mut Node> { + self.nodes.iter_mut().find(|n| n.id == id) + } + + /// Distinct agents this DAG's nodes target, in first-seen order. + /// Used for per-agent lease release and the terminal cancel-revert — + /// a single-agent DAG yields one, a multi-agent DAG yields several. + pub fn agents(&self) -> Vec { + let mut seen: Vec = Vec::new(); + for n in &self.nodes { + if !seen.iter().any(|a| a == &n.agent) { + seen.push(n.agent.clone()); + } + } + seen + } +} + +impl Dag { + pub fn view(&self) -> DagView { + let started_at = self.nodes.iter().filter_map(|n| n.started_at).min(); + let finished_at = if self.is_terminal() { + self.nodes.iter().filter_map(|n| n.finished_at).max() + } else { + None + }; + DagView { + id: self.id, + kind: self.template, + state: self.rollup(), + source: self.source, + reason: self.reason.clone(), + enqueued_at: self.created_at, + started_at, + finished_at, + inputs: self.inputs.clone(), + approval_id: self.approval_id, + perm_payload: self.perm_payload.clone(), + nodes: self + .nodes + .iter() + .map(|n| NodeView { + id: n.id, + agent: n.agent.clone(), + kind: n.kind.as_str().to_owned(), + deps: n.deps.iter().map(|d| d.on).collect(), + state: n.state, + step: n.step.clone(), + build_log_id: n.build_log_id, + started_at: n.started_at, + finished_at: n.finished_at, + error: n.error.clone(), + }) + .collect(), + } + } +} diff --git a/hive-c0re/src/job_queue/resource.rs b/hive-c0re/src/job_queue/resource.rs deleted file mode 100644 index a59a428f..00000000 --- a/hive-c0re/src/job_queue/resource.rs +++ /dev/null @@ -1,51 +0,0 @@ -//! The concrete resource type the rebuild queue schedules over — the bridge -//! from hive-c0re's [`NodeKind`] onto the domain-agnostic `hive-jobq` crate. -//! `hive-jobq` is generic over a resource type `R: Clone + Eq + Hash` and a node -//! payload `N`; here `R` is [`Resource`] and `N` is [`NodeKind`] directly (each -//! variant carries the agent it targets). - -use hive_jobq::Dep; - -use super::model::NodeKind; - -/// The two resource classes the queue gates concurrency on, as the crate's -/// generic resource type `R`. -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub enum Resource { - /// One of the `buildSlots` permits, held by a nix-heavy node for its - /// duration. Capacity is `services.hyperhive.c0re.buildSlots` (default 1), - /// set on the [`hive_jobq::resources::ResourceTable`] at construction. - BuildSlot, - /// The per-agent lifecycle lease — globally exclusive per agent across all - /// DAGs (unconfigured, so the crate's default capacity 1 applies). Held by - /// a DAG's first container-affecting node for that agent and re-entered by - /// the rest of that agent's subtree via the crate's recursive lock, so two - /// DAGs never interleave container ops on one agent. - Agent(String), -} - -impl NodeKind { - /// The [`Dep::Resource`] edges this node must acquire to run, derived from - /// its kind + agent: a build slot for nix-heavy kinds - /// ([`NodeKind::needs_build_slot`]) and the agent lease for - /// container-affecting kinds ([`NodeKind::needs_lease`]). Lease-exempt - /// container ops (`Start` / `Stop`, fanned out by a lease-holding - /// `Reconcile`) hold no lease of their own — they re-enter the ancestor's - /// `Agent` lock through the crate's recursive re-entrancy. - pub fn resource_deps(&self) -> Vec> { - let mut deps = Vec::new(); - if self.needs_build_slot() { - deps.push(Dep::Resource { - name: Resource::BuildSlot, - count: 1, - }); - } - if self.needs_lease() { - deps.push(Dep::Resource { - name: Resource::Agent(self.agent().to_owned()), - count: 1, - }); - } - deps - } -} diff --git a/hive-c0re/src/job_queue/scheduler.rs b/hive-c0re/src/job_queue/scheduler.rs index 1c2022f9..3e9959d5 100644 --- a/hive-c0re/src/job_queue/scheduler.rs +++ b/hive-c0re/src/job_queue/scheduler.rs @@ -1,25 +1,18 @@ -//! The single scheduler task that drives all DAGs: claim every ready node (as -//! many as the build slots / leases allow), spawn one executor task per claim, -//! and on any completion re-evaluate. Concurrency comes from the build-slot -//! count, not multiple workers. +//! The single scheduler task that drives all DAGs: claim every ready +//! node (as many as the build slots / leases allow), spawn one +//! executor task per claim, and on any completion re-evaluate. +//! Concurrency comes from the build-slot count, not multiple workers. //! -//! Owns the per-DAG transient guard (dashboard pill + crash-watch suppression) -//! that the sync queue core can't hold itself. The guard set is *reconciled* -//! from live lease ownership ([`super::JobQueue::held_transients`]) each loop: -//! a `(dag, agent)` pill exists for exactly as long as that agent's lease is -//! held, so it appears when the agent's owner node starts and disappears when -//! its subgraph settles — one pill per agent a DAG touches. +//! Also owns the per-DAG transient guard (dashboard pill + crash-watch +//! suppression) that the sync queue core can't hold itself — created when a +//! DAG acquires its agent lease, dropped when the DAG settles terminal. //! -//! Per-DAG terminal work (approval resolution, `Rebuilt`, cancelled-power-op -//! intent revert) is not drained here: it runs as the DAG's focused terminal -//! node (`ResolveApproval` / `EmitRebuilt` / `RevertIntent`), dispatched through -//! `exec::run_node` like any other node once the DAG settles. -//! -//! In-DAG growth (a `MetaLock` growing rebuild subgraphs, a `Reconcile` fanning -//! its `Start`/`Stop`) flows through `NodeOutput.append_subgraph`, applied -//! before the emitting node completes — see `handle_completion`. +//! In-DAG growth (a `MetaLock` growing rebuild subgraphs after the lock +//! bump, a `Reconcile` fanning its `Start`/`Stop`) flows through +//! `NodeOutput.append_subgraph`, applied before the emitting node +//! completes — see `handle_completion`. -use std::collections::{HashMap, HashSet}; +use std::collections::HashMap; use std::sync::Arc; use super::Claim; @@ -33,24 +26,38 @@ struct NodeDone { /// Scheduler loop. Spawned once at hive-c0re startup from `main.rs`. /// -/// Shutdown semantics: subscribes to `coord.shutdown_rx()`. On a true signal -/// the loop exits immediately; already-running node tasks ride the runtime down -/// with the process, and pending `Queued` DAGs are dropped — desired state is -/// re-derived on next boot (boot sweep + reconcile), so the in-memory queue is -/// deliberately not durable. +/// Shutdown semantics: subscribes to `coord.shutdown_rx()`. On a true +/// signal the loop exits immediately; already-running node tasks ride +/// the runtime down with the process, and pending `Queued` DAGs are +/// dropped — desired state is re-derived on next boot (boot sweep + +/// reconcile), so the in-memory queue is deliberately not durable. pub async fn run_worker(coord: Arc) { let mut shutdown = coord.shutdown_rx(); let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel::(); - // (DAG id, agent) → transient guard held for that agent's lease window. + // (DAG id, agent) → transient guard held for that agent's lease + // window. Keyed per-agent so a multi-agent DAG shows one transient + // pill per agent it touches. let mut transients: HashMap<(u64, String), crate::coordinator::TransientGuard> = HashMap::new(); loop { - reconcile_transients(&coord, &mut transients); + // Terminal roll-ups can appear without a node completion — + // the cancel surfaces settle DAGs directly and wake this loop + // via notify — so drain on every iteration, not just inside + // handle_completion. + process_terminals(&coord, &mut transients).await; let claims = coord.job_queue.claim_ready(); if !claims.is_empty() { for claim in claims { + if claim.lease_acquired + && let Some(kind) = claim.transient + { + transients.insert( + (claim.dag_id, claim.agent.clone()), + coord.transient_guard(&claim.agent, kind), + ); + } tracing::info!( dag = claim.dag_id, - node = claim.node_id.get(), + node = claim.node_id, kind = claim.kind.as_str(), agent = %claim.agent, template = claim.template.as_str(), @@ -64,8 +71,6 @@ pub async fn run_worker(coord: Arc) { let _ = tx.send(NodeDone { claim, result }); }); } - // Newly-started owner nodes now hold their leases — surface the pills. - reconcile_transients(&coord, &mut transients); coord.emit_rebuild_queue_snapshot(); continue; } @@ -78,85 +83,82 @@ pub async fn run_worker(coord: Arc) { } } Some(done) = rx.recv() => { - handle_completion(&coord, done); + handle_completion(&coord, &mut transients, done).await; } () = coord.job_queue.notify.notified() => {} } } } -fn handle_completion(coord: &Arc, done: NodeDone) { +async fn handle_completion( + coord: &Arc, + transients: &mut HashMap<(u64, String), crate::coordinator::TransientGuard>, + done: NodeDone, +) { let NodeDone { claim, result } = done; match result { Ok(output) => { tracing::info!( dag = claim.dag_id, - node = claim.node_id.get(), + node = claim.node_id, "job_queue: node done" ); // Append any in-DAG subgraphs BEFORE completing this node, so // completing it doesn't roll the DAG terminal while the appended - // work is still pending. Each subgraph roots on this node - // (`AfterOk`), so it becomes ready the instant this one settles - // `Done` just below — covers both the multi-node case (a `MetaLock` - // growing per-agent rebuild subgraphs) and the single-node case (a - // `Reconcile` planner's `Start` / `Stop`). - for subgraph in &output.append_subgraph { + // work is still pending — that keeps the lease-window transient + // held across it. Each subgraph is independent, rooted on this + // node (`AfterOk`), so it becomes ready the instant this one + // settles `Done` just below. Covers both the multi-node case (a + // `MetaLock` growing per-agent rebuild subgraphs) and the + // single-node case (a `Reconcile` planner's `Start` / `Stop`). + for subgraph in output.append_subgraph { coord .job_queue .append_subgraph(claim.dag_id, subgraph, claim.node_id); } - let terminal = coord + coord .job_queue .complete_node(claim.dag_id, claim.node_id, Ok(())); - fire_terminal_hook(coord, terminal); } Err(e) => { let msg = format!("{e:#}"); tracing::warn!( dag = claim.dag_id, - node = claim.node_id.get(), + node = claim.node_id, kind = claim.kind.as_str(), agent = %claim.agent, error = %msg, "job_queue: node failed" ); - let terminal = coord + coord .job_queue .complete_node(claim.dag_id, claim.node_id, Err(msg)); - fire_terminal_hook(coord, terminal); } } - // The next loop iteration re-reconciles the transient pills against the - // post-completion lease state (a settled subgraph drops its pill). + process_terminals(coord, transients).await; coord.emit_rebuild_queue_snapshot(); } -/// Fire a settled DAG's inline terminal hook (approval-resolve / rebuilt-emit / -/// intent-revert) off the container-terminal summary `complete_node` returned — -/// spawned so the async hook doesn't block the scheduler loop. -fn fire_terminal_hook(coord: &Arc, terminal: Option) { - let Some(terminal) = terminal else { - return; - }; - let coord = Arc::clone(coord); - tokio::spawn(async move { - exec::run_terminal_hook(&coord, &terminal).await; - }); -} - -/// Reconcile the transient-guard set against live lease ownership: drop pills -/// whose lease is no longer held, create one for each newly-held `(dag, agent)`. -fn reconcile_transients( +/// Drain per-agent lease releases and buffered terminal roll-ups. +/// +/// Per-agent first: an agent's subgraph within a DAG went terminal (its +/// lease was freed in `settle`), so drop that agent's `(dag, agent)` +/// transient pill now — ahead of whole-DAG terminal for a multi-agent +/// DAG. Then the whole-DAG terminals: drop any remaining transient the +/// DAG still held and run the terminal hook (approval resolution, +/// `Rebuilt` events, cancelled-power-op intent revert). +async fn process_terminals( coord: &Arc, transients: &mut HashMap<(u64, String), crate::coordinator::TransientGuard>, ) { - let held = coord.job_queue.held_transients(); - let keys: HashSet<(u64, String)> = held.iter().map(|(d, a, _)| (*d, a.clone())).collect(); - transients.retain(|k, _| keys.contains(k)); - for (dag_id, agent, kind) in held { - transients - .entry((dag_id, agent.clone())) - .or_insert_with(|| coord.transient_guard(&agent, kind)); + for rel in coord.job_queue.drain_agent_releases() { + transients.remove(&(rel.dag_id, rel.agent)); + } + for terminal in coord.job_queue.drain_terminal() { + // Drop any per-agent transient guard the DAG still held (the + // per-agent pass above already dropped the ones whose subgraphs + // settled early). + transients.retain(|(dag_id, _), _| *dag_id != terminal.dag_id); + exec::on_dag_terminal(coord, &terminal).await; } } diff --git a/hive-c0re/src/job_queue/submit.rs b/hive-c0re/src/job_queue/submit.rs index b4d85415..805e4d78 100644 --- a/hive-c0re/src/job_queue/submit.rs +++ b/hive-c0re/src/job_queue/submit.rs @@ -25,7 +25,7 @@ use std::sync::Arc; use super::model::{DagSpec, Dep, NodeKind, NodeSpec, Template}; -use super::templates::{after_ok, child, node, rebuild_nodes}; +use super::templates::{after_ok, node, rebuild_nodes}; use super::{Source, templates}; use crate::coordinator::{Coordinator, TransientKind}; use crate::lifecycle; @@ -60,23 +60,13 @@ pub fn rebuild(coord: &Arc, agent: &str, source: Source, reason: St /// stays even for a down agent so a race-up between the state read and exec /// is still stopped in-DAG. fn stop_chain(agent: &str, graceful: bool, running: bool) -> Vec { - // `SetWanted` is the group root and owns the agent lease; the mechanical - // steps are its children (borrow the lease, run once it reaches `Finishing`, - // dep-ordered among themselves). - let a = || agent.to_owned(); - let mut n = vec![node( - NodeKind::SetWanted { - agent: a(), - up: false, - }, - Vec::new(), - )]; + let mut n = vec![node(agent, NodeKind::SetWanted { up: false }, Vec::new())]; if graceful && running { - n.push(child(0, NodeKind::Signal { agent: a() }, Vec::new())); - n.push(child(0, NodeKind::Drain { agent: a() }, after_ok(1))); - n.push(child(0, NodeKind::Reconcile { agent: a() }, after_ok(2))); + n.push(node(agent, NodeKind::Signal, after_ok(0))); + n.push(node(agent, NodeKind::Drain, after_ok(1))); + n.push(node(agent, NodeKind::Reconcile, after_ok(2))); } else { - n.push(child(0, NodeKind::Reconcile { agent: a() }, Vec::new())); + n.push(node(agent, NodeKind::Reconcile, after_ok(0))); } n } @@ -86,26 +76,13 @@ fn stop_chain(agent: &str, graceful: bool, running: bool) -> Vec { /// current derivations), otherwise a plain `Reconcile` (which starts a down /// agent and noops an already-running one). fn start_chain(agent: &str, running: bool, stale: bool) -> Vec { - let mut n = vec![node( - NodeKind::SetWanted { - agent: agent.to_owned(), - up: true, - }, - Vec::new(), - )]; + let mut n = vec![node(agent, NodeKind::SetWanted { up: true }, Vec::new())]; if !running && stale { - // Rebuild subtree after the SetWanted head (base = 1, so the rebuild's - // `Prebuild` root deps `after_ok(0)` = the head). `Prebuild` + - // `Reconcile` are their own group roots (top-level, per `rebuild_nodes`). + // Rebuild subgraph rooted at the SetWanted head (base = 1, so + // `Prebuild` deps `after_ok(0)` = the head). n.extend(rebuild_nodes(agent, true, 1)); } else { - n.push(child( - 0, - NodeKind::Reconcile { - agent: agent.to_owned(), - }, - Vec::new(), - )); + n.push(node(agent, NodeKind::Reconcile, after_ok(0))); } n } @@ -120,39 +97,22 @@ fn start_chain(agent: &str, running: bool, stale: bool) -> Vec { /// converges to intent — a stopped (`wanted = Off`) agent stays stopped, /// a crashed (`wanted = Up`) agent comes back up. fn restart_chain(agent: &str, graceful: bool, running: bool) -> Vec { - let a = || agent.to_owned(); if !running { // Nothing to bounce — a lone Reconcile converges to intent. - return vec![node(NodeKind::Reconcile { agent: a() }, Vec::new())]; + return vec![node(agent, NodeKind::Reconcile, Vec::new())]; } - // Running: mechanical stop then Reconcile. The first stop node is the group - // ROOT (no SetWanted head) and owns the agent lease; the rest are its - // children (borrow the lease, dep-ordered), so the bounce holds one - // continuous lease and `Reconcile` cancel-cascades if a stop step fails. - let mut n = vec![if graceful { - node(NodeKind::Signal { agent: a() }, Vec::new()) - } else { - node(NodeKind::StopForUpdate { agent: a() }, Vec::new()) - }]; + // Running: mechanical stop then Reconcile. The first stop node is the + // subgraph root (no SetWanted head) and acquires the agent lease. + let mut n = Vec::new(); if graceful { - n.push(child(0, NodeKind::Drain { agent: a() }, Vec::new())); - n.push(child( - 0, - NodeKind::StopForUpdate { agent: a() }, - after_ok(1), - )); - } - // `Reconcile` gates on the last mechanical step. When the only step is the - // root itself (non-graceful, `StopForUpdate` == index 0), the parent gate - // already orders `Reconcile` after it — a child must NOT dep on its own - // parent (dep-scope). So the sibling dep is added only for a graceful - // bounce, where the last step is a sibling child. - let deps = if n.len() > 1 { - after_ok(u64::try_from(n.len() - 1).unwrap_or(0)) + n.push(node(agent, NodeKind::Signal, Vec::new())); + n.push(node(agent, NodeKind::Drain, after_ok(0))); + n.push(node(agent, NodeKind::StopForUpdate, after_ok(1))); } else { - Vec::new() - }; - n.push(child(0, NodeKind::Reconcile { agent: a() }, deps)); + n.push(node(agent, NodeKind::StopForUpdate, Vec::new())); + } + let stop_idx = u32::try_from(n.len() - 1).unwrap_or(0); + n.push(node(agent, NodeKind::Reconcile, after_ok(stop_idx))); n } @@ -164,7 +124,7 @@ fn restart_chain(agent: &str, graceful: bool, running: bool) -> Vec { fn concat_subgraphs(chains: Vec>) -> Vec { let mut out: Vec = Vec::new(); for chain in chains { - let base = u64::try_from(out.len()).unwrap_or(u64::MAX); + let base = u32::try_from(out.len()).unwrap_or(u32::MAX); for spec in chain { let deps = spec .deps @@ -175,12 +135,9 @@ fn concat_subgraphs(chains: Vec>) -> Vec { }) .collect(); out.push(NodeSpec { + agent: spec.agent, kind: spec.kind, deps, - // Rebase the structural parent by the same offset (a subgraph - // root keeps `parent = None`, so the per-agent groups stay - // independent + concurrent). - parent: spec.parent.map(|p| base + p), }); } } @@ -201,6 +158,7 @@ fn power_dag( reason, approval_id: None, inputs: Vec::new(), + perm_payload: None, transient: Some(transient), nodes, } diff --git a/hive-c0re/src/job_queue/templates.rs b/hive-c0re/src/job_queue/templates.rs index f456da8e..898f7b96 100644 --- a/hive-c0re/src/job_queue/templates.rs +++ b/hive-c0re/src/job_queue/templates.rs @@ -35,78 +35,52 @@ use crate::coordinator::TransientKind; /// After-ok edge on the previous node — the common chain link. Shared with /// the async power-op builders in `submit.rs` (which assemble per-agent /// chains dynamically from live container state). -pub(crate) fn after_ok(on: u64) -> Vec { +pub(crate) fn after_ok(on: u32) -> Vec { vec![Dep { on, when: DepWhen::AfterOk, }] } -/// Build one **top-level (group-root)** node — `parent = None`. `kind` carries -/// the agent it targets ([`NodeKind`] is the payload directly). Shared with -/// `submit.rs`'s dynamic power-op builders. A root owns whatever resource it -/// declares for its whole subtree; its descendants borrow it (agent-lease / -/// build-slot continuity). Ordering vs other nodes is `deps`; grouping is -/// `parent`. -pub(crate) fn node(kind: NodeKind, deps: Vec) -> NodeSpec { +/// Build one node targeting `agent`. The single place a node's agent is +/// stamped. Shared with `submit.rs`'s dynamic power-op builders. +pub(crate) fn node(agent: &str, kind: NodeKind, deps: Vec) -> NodeSpec { NodeSpec { + agent: agent.to_owned(), kind, deps, - parent: None, } } -/// Build a **child** node whose structural parent is spec-index `parent`. The -/// child runs once its parent reaches `Finishing` (the parent gate), so it must -/// NOT `deps` on `parent` (dep-scope validation rejects a dep on one's own -/// parent). `deps` here order the child against its *siblings* only. -pub(crate) fn child(parent: u64, kind: NodeKind, deps: Vec) -> NodeSpec { - NodeSpec { - kind, - deps, - parent: Some(parent), - } -} - -/// The rebuild node subtree (nested, two group roots). `base` is the spec index -/// of the first node (`Prebuild`). Structure: -/// - `Prebuild` (base+0, **root**): owns the build slot for the whole subtree. -/// Lease-exempt — the nix build overlaps other DAGs on the same agent. -/// - `StopForUpdate` (base+1, child of `Prebuild`): owns the agent lease. Runs -/// once `Prebuild` reaches `Finishing` (parent gate). -/// - `Swap` (base+2, child of `StopForUpdate`): borrows the agent lease from its -/// parent and the build slot from grand-ancestor `Prebuild` — both continuous. -/// - `PostSwap` (base+3, child of `StopForUpdate`): the swap's Ok-only -/// bookkeeping tail (rev marker, forge/matrix sync, kick, rescan), `AfterOk` -/// its sibling `Swap`. -/// - `Reconcile` (base+4, **root**): `AfterAny` `Prebuild`, which rolls up -/// terminal only once its whole mechanical subtree (SFU→Swap→PostSwap) has -/// settled — so `Reconcile` runs after the swap regardless of outcome, and as -/// a top-level root it survives the cancel-cascade of a failed `Prebuild` -/// (recovery-start invariant). It takes a fresh lease; the tiny gap is -/// harmless — `Reconcile` converges to the persisted `wanted` idempotently. -pub(crate) fn rebuild_nodes(agent: &str, relock: bool, base: u64) -> Vec { - let a = || agent.to_owned(); +/// The rebuild node chain. `PostSwap` carries the swap's Ok-only +/// bookkeeping tail (rev marker, forge/matrix sync, kick, rescan) and deps +/// `Swap` with `AfterOk`. `Reconcile` then deps on `PostSwap` with +/// `AfterAny`: it must run even when the swap failed, so a previously-up +/// agent comes back on its old config (today's recovery-start). On swap +/// failure the `AfterOk` `PostSwap` is cancel-cascaded to a terminal state, +/// which still satisfies `Reconcile`'s `AfterAny` edge — the only `AfterAny` +/// edge in v1. Pointing `Reconcile` at `PostSwap` (not `Swap`) also +/// serializes the tail ahead of the reconcile, so there's no double +/// rescan/kick race. +pub(crate) fn rebuild_nodes(agent: &str, relock: bool, base: u32) -> Vec { vec![ node( - NodeKind::Prebuild { agent: a(), relock }, + agent, + NodeKind::Prebuild { relock }, if base == 0 { Vec::new() } else { after_ok(base - 1) }, ), - child(base, NodeKind::StopForUpdate { agent: a() }, Vec::new()), - child(base + 1, NodeKind::Swap { agent: a() }, Vec::new()), - child( - base + 1, - NodeKind::PostSwap { agent: a() }, - after_ok(base + 2), - ), + node(agent, NodeKind::StopForUpdate, after_ok(base)), + node(agent, NodeKind::Swap, after_ok(base + 1)), + node(agent, NodeKind::PostSwap, after_ok(base + 2)), node( - NodeKind::Reconcile { agent: a() }, + agent, + NodeKind::Reconcile, vec![Dep { - on: base, + on: base + 3, when: DepWhen::AfterAny, }], ), @@ -125,6 +99,7 @@ pub fn rebuild(agent: &str, source: Source, reason: String, relock: bool) -> Dag reason, approval_id: None, inputs: Vec::new(), + perm_payload: None, transient: Some(TransientKind::Rebuilding), nodes: rebuild_nodes(agent, relock, 0), } @@ -140,13 +115,9 @@ pub fn approval_deploy(agent: &str, approval_id: i64, reason: String) -> DagSpec reason, approval_id: Some(approval_id), inputs: Vec::new(), + perm_payload: None, transient: Some(TransientKind::Rebuilding), - nodes: vec![node( - NodeKind::ApprovalDeploy { - agent: agent.to_owned(), - }, - Vec::new(), - )], + nodes: vec![node(agent, NodeKind::ApprovalDeploy, Vec::new())], } } @@ -169,25 +140,16 @@ pub fn reconcile_only( reason, approval_id: None, inputs: Vec::new(), + perm_payload: None, transient, - nodes: vec![node( - NodeKind::Reconcile { - agent: agent.to_owned(), - }, - Vec::new(), - )], + nodes: vec![node(agent, NodeKind::Reconcile, Vec::new())], } } /// First-deploy spawn (approval-driven): `Provision` (proposed/applied /// repos, state subvolume, meta registration) then `Create` /// (`nixos-container create`), drop-in write, then `Reconcile` starts -/// the container (`wanted = Up` written at approve time). All-or-nothing: -/// `Provision` (lease-exempt, precedes the container) is the group root; -/// `Create` (child) owns the agent lease; `WriteDropin` + `Reconcile` -/// (children of `Create`) borrow it. A failure cancel-cascades the rest — -/// unlike rebuild there's no recovery-reconcile (nothing to converge if the -/// container was never created). +/// the container (`wanted = Up` written at approve time). pub fn spawn(agent: &str, approval_id: i64, reason: String) -> DagSpec { DagSpec { template: Template::Spawn, @@ -195,16 +157,14 @@ pub fn spawn(agent: &str, approval_id: i64, reason: String) -> DagSpec { reason, approval_id: Some(approval_id), inputs: Vec::new(), + perm_payload: None, transient: Some(TransientKind::Spawning), - nodes: { - let a = || agent.to_owned(); - vec![ - node(NodeKind::Provision { agent: a() }, Vec::new()), - child(0, NodeKind::Create { agent: a() }, Vec::new()), - child(1, NodeKind::WriteDropin { agent: a() }, Vec::new()), - child(1, NodeKind::Reconcile { agent: a() }, after_ok(2)), - ] - }, + nodes: vec![ + node(agent, NodeKind::Provision, Vec::new()), + node(agent, NodeKind::Create, after_ok(0)), + node(agent, NodeKind::WriteDropin, after_ok(1)), + node(agent, NodeKind::Reconcile, after_ok(2)), + ], } } @@ -212,13 +172,7 @@ pub fn spawn(agent: &str, approval_id: i64, reason: String) -> DagSpec { /// the updated `HIVE_TOOL_GROUPS` / `HIVE_CAPABILITIES` env var takes /// effect in the container. pub fn perm_change(agent: &str, source: Source, reason: String, payload: PermPayload) -> DagSpec { - let mut nodes = vec![node( - NodeKind::WritePermFile { - agent: agent.to_owned(), - payload, - }, - Vec::new(), - )]; + let mut nodes = vec![node(agent, NodeKind::WritePermFile, Vec::new())]; nodes.extend(rebuild_nodes(agent, true, 1)); DagSpec { template: Template::PermChange, @@ -226,6 +180,7 @@ pub fn perm_change(agent: &str, source: Source, reason: String, payload: PermPay reason, approval_id: None, inputs: Vec::new(), + perm_payload: Some(payload), transient: Some(TransientKind::Rebuilding), nodes, } @@ -253,8 +208,10 @@ pub fn meta_update( reason, approval_id, inputs, + perm_payload: None, transient: Some(TransientKind::Rebuilding), nodes: vec![node( + "hyperhive", NodeKind::MetaLock { sweep: false, fanout: None, @@ -270,9 +227,9 @@ pub fn meta_update( // per-agent child DAGs. /// Validate a spec before it enters the queue: node ids are dense -/// (index = id), deps + parents reference existing *earlier* nodes, and the -/// dep graph is acyclic (petgraph `toposort`). Rejecting cycles here fixes the -/// old queue's documented "circular dep silently deadlocks forever" caveat. +/// (index = id), deps reference existing nodes, and the dep graph is +/// acyclic (petgraph `toposort`). Rejecting cycles here fixes the old +/// queue's documented "circular dep silently deadlocks forever" caveat. pub fn validate(spec: &DagSpec) -> Result<()> { if spec.nodes.is_empty() { bail!("dag spec {:?} has no nodes", spec.template); @@ -283,19 +240,8 @@ pub fn validate(spec: &DagSpec) -> Result<()> { .map(|i| graph.add_node(u32::try_from(i).unwrap_or(u32::MAX))) .collect(); for (i, node) in spec.nodes.iter().enumerate() { - // A `parent` must index an earlier node — `insert_group` resolves it to - // an already-inserted `NodeId`, so a forward/out-of-bounds parent would - // otherwise panic there. - if let Some(p) = node.parent - && usize::try_from(p).is_ok_and(|p| p >= i) - { - bail!( - "dag spec {:?} node {i} has invalid parent {p} (must be an earlier node)", - spec.template - ); - } for dep in &node.deps { - let Some(&dep_idx) = usize::try_from(dep.on).ok().and_then(|i| idx.get(i)) else { + let Some(&dep_idx) = idx.get(dep.on as usize) else { bail!( "dag spec {:?} node {i} depends on unknown node {}", spec.template, diff --git a/hive-c0re/src/job_queue/tests.rs b/hive-c0re/src/job_queue/tests.rs index b385e420..eae41d91 100644 --- a/hive-c0re/src/job_queue/tests.rs +++ b/hive-c0re/src/job_queue/tests.rs @@ -109,24 +109,20 @@ fn cyclic_dag_is_rejected_at_submit() { // 0 → 1 → 0 cycle. spec.nodes = vec![ NodeSpec { - kind: NodeKind::StopForUpdate { - agent: "agent-a".to_owned(), - }, + agent: "agent-a".to_owned(), + kind: NodeKind::StopForUpdate, deps: vec![Dep { on: 1, when: DepWhen::AfterOk, }], - parent: None, }, NodeSpec { - kind: NodeKind::Reconcile { - agent: "agent-a".to_owned(), - }, + agent: "agent-a".to_owned(), + kind: NodeKind::Reconcile, deps: vec![Dep { on: 0, when: DepWhen::AfterOk, }], - parent: None, }, ]; assert!(q.submit(spec).is_err(), "cyclic spec must be refused"); @@ -138,30 +134,12 @@ fn unknown_dep_is_rejected_at_submit() { let q = JobQueue::new(1); let mut spec = rebuild("agent-a", "bad dep"); spec.nodes = vec![NodeSpec { - kind: NodeKind::Reconcile { - agent: "agent-a".to_owned(), - }, + agent: "agent-a".to_owned(), + kind: NodeKind::Reconcile, deps: vec![Dep { on: 9, when: DepWhen::AfterOk, }], - parent: None, - }]; - assert!(q.submit(spec).is_err()); -} - -#[test] -fn invalid_parent_is_rejected_at_submit() { - let q = JobQueue::new(1); - let mut spec = rebuild("agent-a", "bad parent"); - // A forward/out-of-bounds parent index must be refused at validate, not - // panic in `insert_group`. - spec.nodes = vec![NodeSpec { - kind: NodeKind::Reconcile { - agent: "agent-a".to_owned(), - }, - deps: Vec::new(), - parent: Some(3), }]; assert!(q.submit(spec).is_err()); } @@ -202,16 +180,13 @@ fn build_slot_serializes_nix_heavy_nodes() { assert_eq!(first.dag_id, a); assert_eq!(first.kind.as_str(), "prebuild"); q.complete_node(a, first.node_id, Ok(())); - // Uniform hold: agent-a keeps the build slot across its whole build chain - // (Swap re-enters it), so a's StopForUpdate (lease, slot-free) runs but b's - // Prebuild must wait for a's slot-needers (through Swap) to finish. + // With the slot free again, FIFO gives... a's StopForUpdate is + // slot-free (lease) and b's Prebuild takes the slot — both run. let claims = q.claim_ready(); let kinds: Vec<(u64, &str)> = claims.iter().map(|c| (c.dag_id, c.kind.as_str())).collect(); - assert_eq!(kinds, vec![(a, "stop_for_update")]); - assert!( - !kinds.iter().any(|&(d, _)| d == b), - "b's build waits — slot held across a's chain" - ); + assert!(kinds.contains(&(a, "stop_for_update"))); + assert!(kinds.contains(&(b, "prebuild"))); + assert_eq!(claims.len(), 2); } #[test] @@ -233,27 +208,9 @@ fn fifo_fairness_for_the_slot() { let first = claim_one(&q); assert_eq!(first.dag_id, a, "submit order wins the slot"); q.complete_node(a, first.node_id, Ok(())); - // Uniform hold: the slot stays with agent-a until its Swap (the last - // slot-needer) completes. Drive a's chain; the moment its slot frees, - // submit order (b before c) wins it. - let mut freed_to = None; - for _ in 0..6 { - let claims = q.claim_ready(); - if let Some(nb) = claims.iter().find(|cl| cl.dag_id == b || cl.dag_id == c) { - freed_to = Some(nb.dag_id); - break; - } - for cl in claims { - if cl.dag_id == a { - q.complete_node(a, cl.node_id, Ok(())); - } - } - } - assert_eq!( - freed_to, - Some(b), - "b's prebuild wins the freed slot before c's" - ); + let next: Vec = q.claim_ready().iter().map(|cl| cl.dag_id).collect(); + assert!(next.contains(&b), "b's prebuild before c's"); + assert!(!next.contains(&c)); } // ---- per-agent lease ---- @@ -277,19 +234,17 @@ fn lease_serializes_two_lifecycle_dags_for_same_agent() { let first = claim_one(&q); assert_eq!(first.dag_id, restart); assert_eq!(first.kind.as_str(), "stop_for_update"); + assert!(first.lease_acquired); q.complete_node(restart, first.node_id, Ok(())); - // Same DAG keeps the lease through the tail Reconcile (re-entered from the - // dep graph — no fresh acquire), since stop's Reconcile can't re-enter it. + // Same DAG keeps the lease through the tail Reconcile. let second = claim_one(&q); assert_eq!(second.dag_id, restart); assert_eq!(second.kind.as_str(), "reconcile"); + assert!(!second.lease_acquired, "lease already held by this DAG"); q.complete_node(restart, second.node_id, Ok(())); - // Restart's work is terminal → its lease releases, so stop's now-unblocked - // Reconcile becomes ready (restart's inline hook fired off the returned - // summary — no terminal-hook node). + // Restart terminal → lease released → stop's Reconcile runs. let third = claim_one(&q); assert_eq!(third.dag_id, stop); - assert_eq!(third.kind.as_str(), "reconcile"); q.complete_node(stop, third.node_id, Ok(())); assert_eq!(state_of(&q, restart), State::Done); assert_eq!(state_of(&q, stop), State::Done); @@ -333,15 +288,8 @@ fn lease_exempt_prebuild_overlaps_other_dag_on_same_agent() { .expect("reconcile claim") .clone(); q.complete_node(stop, reconcile.node_id, Ok(())); - // stop's Reconcile done → its lease frees, so rebuild's StopForUpdate - // unblocks. (stop's DAG rolls up terminal; its inline hook fires off the - // returned summary — no terminal-hook node in the claim set.) - let after = q.claim_ready(); - let sfu = after - .iter() - .find(|c| c.kind.as_str() == "stop_for_update") - .expect("rebuild StopForUpdate unblocked once the lease frees"); - assert_eq!(sfu.agent, "agent-a"); + let next = claim_one(&q); + assert_eq!(next.kind.as_str(), "stop_for_update"); } #[test] @@ -367,16 +315,16 @@ fn multi_agent_restart_is_one_dag_with_concurrent_per_agent_subgraphs() { // lease (no contention across distinct agents), all inside the single DAG. let claims = q.claim_ready(); assert!(claims.iter().all(|c| c.dag_id == id)); - let mut heads: Vec<(&str, &str)> = claims + let mut heads: Vec<(&str, &str, bool)> = claims .iter() - .map(|c| (c.agent.as_str(), c.kind.as_str())) + .map(|c| (c.agent.as_str(), c.kind.as_str(), c.lease_acquired)) .collect(); heads.sort_unstable(); assert_eq!( heads, vec![ - ("agent-a", "stop_for_update"), - ("agent-b", "stop_for_update"), + ("agent-a", "stop_for_update", true), + ("agent-b", "stop_for_update", true), ], "both per-agent subgraphs start concurrently, each acquiring its own lease" ); @@ -439,14 +387,17 @@ fn multi_agent_stop_is_one_dag_with_concurrent_per_agent_subgraphs() { assert_eq!(q.snapshot().len(), 1); let claims = q.claim_ready(); assert!(claims.iter().all(|c| c.dag_id == id)); - let mut heads: Vec<(&str, &str)> = claims + let mut heads: Vec<(&str, &str, bool)> = claims .iter() - .map(|c| (c.agent.as_str(), c.kind.as_str())) + .map(|c| (c.agent.as_str(), c.kind.as_str(), c.lease_acquired)) .collect(); heads.sort_unstable(); assert_eq!( heads, - vec![("agent-a", "set_wanted"), ("agent-b", "set_wanted")], + vec![ + ("agent-a", "set_wanted", true), + ("agent-b", "set_wanted", true), + ], "both per-agent stop subgraphs start concurrently, each on its own lease" ); } @@ -561,14 +512,15 @@ fn append_subgraph_roots_on_emitter_and_rebases_local_deps() { reason: "sweep".to_owned(), approval_id: None, inputs: Vec::new(), + perm_payload: None, transient: None, nodes: vec![NodeSpec { + agent: "hyperhive".to_owned(), kind: NodeKind::MetaLock { sweep: true, fanout: None, }, deps: Vec::new(), - parent: None, }], }; let id = submit(&q, spec); @@ -580,8 +532,8 @@ fn append_subgraph_roots_on_emitter_and_rebases_local_deps() { // tracks any drift in that builder's root-first (`base = 0`) shape. let subgraph = |agent: &str| templates::rebuild_nodes(agent, true, 0); // Must append BEFORE completing the emitter (the documented contract). - q.append_subgraph(id, &subgraph("a"), emitter.node_id); - q.append_subgraph(id, &subgraph("b"), emitter.node_id); + q.append_subgraph(id, subgraph("a"), emitter.node_id); + q.append_subgraph(id, subgraph("b"), emitter.node_id); q.complete_node(id, emitter.node_id, Ok(())); // Still ONE DAG; both subgraph roots become ready once the emitter is // Done (rooted on it), each on its own agent lease. @@ -628,7 +580,7 @@ fn meta_update_carries_rebuilding_transient_and_grows_cascade_in_dag() { for agent in ["alice", "bob"] { q.append_subgraph( id, - &templates::rebuild_nodes(agent, false, 0), + templates::rebuild_nodes(agent, false, 0), meta_lock.node_id, ); } @@ -774,10 +726,7 @@ fn failed_reconcile_marks_dag_failed() { fn cancel_clears_queued_dag() { let q = JobQueue::new(1); let id = submit(&q, rebuild("agent-a", "r")); - // Cancel returns the terminal summary (state `Cancelled`) — the inline hook - // fires off it at the caller; there's no terminal-hook node to claim. - let terminal = q.cancel(id).expect("cancelled"); - assert_eq!(terminal.state, State::Cancelled); + assert!(q.cancel(id)); assert_eq!(state_of(&q, id), State::Cancelled); assert!(q.claim_ready().is_empty()); } @@ -787,32 +736,29 @@ fn cancel_refuses_running_dag() { let q = JobQueue::new(1); let id = submit(&q, rebuild("agent-a", "r")); let _ = claim_one(&q); - assert!(q.cancel(id).is_none()); + assert!(!q.cancel(id)); assert_eq!(state_of(&q, id), State::Running); } // ---- terminal reporting + lease release ---- #[test] -fn dag_settles_terminal_and_releases_lease_after_work() { +fn terminal_dag_reported_exactly_once_and_lease_released() { let q = JobQueue::new(1); let id = submit(&q, restart_online(&["agent-a"], false, "r")); - // restart = StopForUpdate → Reconcile. + // restart = StopForUpdate → Reconcile; not terminal until the last + // node completes. let stop = claim_one(&q); - assert_eq!(stop.kind.as_str(), "stop_for_update"); q.complete_node(id, stop.node_id, Ok(())); + assert!(q.drain_terminal().is_empty(), "dag not terminal yet"); let rec = claim_one(&q); - assert_eq!(rec.kind.as_str(), "reconcile"); - // Completing the last work node rolls the container up terminal and returns - // the summary the inline hook consumes — there is no terminal-hook node. - let summary = q - .complete_node(id, rec.node_id, Ok(())) - .expect("terminal summary"); - assert_eq!(summary.state, State::Done); - assert!(q.claim_ready().is_empty(), "no terminal-hook node to claim"); - assert_eq!(state_of(&q, id), State::Done); - // Lease released when the work chain settled: a new DAG for the agent claims - // immediately. + q.complete_node(id, rec.node_id, Ok(())); + let reports = q.drain_terminal(); + assert_eq!(reports.len(), 1); + assert_eq!(reports[0].dag_id, id); + assert_eq!(reports[0].state, State::Done); + assert!(q.drain_terminal().is_empty(), "reported exactly once"); + // Lease released: a new DAG for the agent can claim immediately. let next = submit( &q, templates::reconcile_only( @@ -825,33 +771,31 @@ fn dag_settles_terminal_and_releases_lease_after_work() { ); let c = claim_one(&q); assert_eq!(c.dag_id, next); + assert!(c.lease_acquired); } /// A DAG cancelled while fully queued must still surface a terminal /// roll-up for the scheduler's hooks — otherwise a queued approval /// DAG cancelled by the operator would dangle its approval forever. #[test] -fn cancelled_dag_finalizes_with_terminal_rollup() { +fn cancelled_dag_reports_terminal_once() { let q = JobQueue::new(1); let id = submit( &q, templates::approval_deploy("agent-a", 7, "approval #7".to_owned()), ); - // Cancel rolls the DAG up terminal and returns its summary — the inline hook - // (approval resolution) runs off it at the caller. Cancelled + approval id 7. - let summary = q.cancel(id).expect("cancelled"); - assert_eq!(summary.state, State::Cancelled); - assert_eq!(summary.approval_id, Some(7)); - // The cancelled DAG's summary stays available (until history-trimmed) and - // unrelated later activity doesn't disturb it. + assert!(q.cancel(id)); + let reports = q.drain_terminal(); + assert_eq!(reports.len(), 1); + assert_eq!(reports[0].dag_id, id); + assert_eq!(reports[0].state, State::Cancelled); + assert_eq!(reports[0].approval_id, Some(7)); + // Never re-reported by later activity. let other = submit(&q, rebuild("agent-b", "r")); let c = claim_one(&q); assert_eq!(c.dag_id, other); q.complete_node(other, c.node_id, Err("boom".to_owned())); - assert_eq!( - q.terminal_summary(id).map(|t| t.state), - Some(State::Cancelled) - ); + assert!(q.drain_terminal().iter().all(|t| t.dag_id != id)); } // ---- steps, build logs, history ---- @@ -860,10 +804,7 @@ fn cancelled_dag_finalizes_with_terminal_rollup() { fn set_step_only_on_running_and_signals_change() { let q = JobQueue::new(1); let id = submit(&q, rebuild("agent-a", "r")); - assert!( - !q.set_step_running(id, "too early"), - "no running node yet → refused" - ); + assert!(!q.set_step(id, 0, "too early"), "queued node refuses step"); let c = claim_one(&q); assert!(q.set_step(id, c.node_id, "nix build")); assert!( @@ -882,10 +823,7 @@ fn set_step_only_on_running_and_signals_change() { fn set_build_log_id_links_running_node() { let q = JobQueue::new(1); let id = submit(&q, rebuild("agent-a", "r")); - assert!( - !q.set_build_log_id_running(id, 41), - "no running node yet → refused" - ); + assert!(!q.set_build_log_id(id, 0, 41), "queued node refuses log id"); let c = claim_one(&q); assert!(q.set_build_log_id(id, c.node_id, 42)); assert!(q.set_build_log_id_running(id, 43)); @@ -910,8 +848,6 @@ fn history_evicts_old_terminals_per_template() { ), ); let c = claim_one(&q); - // Completing the single work node rolls the container up terminal (its - // inline hook fires off the returned summary — no terminal-hook node). q.complete_node(id, c.node_id, Ok(())); } // Fresh terminals are inside the grace window: nothing evicts yet, @@ -923,7 +859,8 @@ fn history_evicts_old_terminals_per_template() { "grace window protects fresh terminals" ); // Past the grace window the per-template cap applies. - assert_eq!(q.snapshot_no_grace().len(), 5, "per-template history cap"); + q.trim_ignoring_grace(); + assert_eq!(q.snapshot().len(), 5, "per-template history cap"); assert_eq!(q.live_count(), 0); } diff --git a/hive-c0re/src/workers/auto_update.rs b/hive-c0re/src/workers/auto_update.rs index 8db07795..d115e232 100644 --- a/hive-c0re/src/workers/auto_update.rs +++ b/hive-c0re/src/workers/auto_update.rs @@ -325,20 +325,20 @@ fn submit_boot_tree( // MetaLock into `run_meta_lock`, which appends the rebuild subgraphs. if any_stale { nodes.push(NodeSpec { + agent: "hyperhive".to_owned(), kind: NodeKind::MetaLock { sweep: true, fanout: Some(fanout), }, deps: Vec::new(), - parent: None, }); } // One boot Reconcile per drifted agent — independent roots. for name in drifted { nodes.push(NodeSpec { - kind: NodeKind::Reconcile { agent: name }, + agent: name, + kind: NodeKind::Reconcile, deps: Vec::new(), - parent: None, }); } @@ -348,6 +348,7 @@ fn submit_boot_tree( reason, approval_id: None, inputs: Vec::new(), + perm_payload: None, // Rebuilding when the sweep will grow rebuild subgraphs (per-agent // crash-watch suppression during their Swap, applied at claim time); // a reconcile-only boot needs no transient. diff --git a/hive-sh4re/src/jobs.rs b/hive-sh4re/src/jobs.rs index a429b72b..d060a1ba 100644 --- a/hive-sh4re/src/jobs.rs +++ b/hive-sh4re/src/jobs.rs @@ -134,11 +134,8 @@ pub enum PermPayload { }, } -/// Node id. Carries the scheduler crate's globally-monotonic node id -/// (`hive_jobq::NodeId`) verbatim on the wire — unique across all DAGs, not -/// just within one. Consumers treat it opaquely (grouping + dep matching), -/// so the widening from the old dag-local `u32` is transparent. -pub type NodeId = u64; +/// Node id, unique within its DAG. +pub type NodeId = u32; /// One node of a queued DAG, as serialized. Step labels, build-log /// links, errors, and timestamps are per-node; the DAG-level `state` @@ -195,5 +192,7 @@ pub struct DagView { pub inputs: Vec, #[serde(default, skip_serializing_if = "Option::is_none")] pub approval_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub perm_payload: Option, pub nodes: Vec, } diff --git a/hivectl/src/dag_progress.rs b/hivectl/src/dag_progress.rs index c726773f..8fb6d76a 100644 --- a/hivectl/src/dag_progress.rs +++ b/hivectl/src/dag_progress.rs @@ -325,7 +325,7 @@ mod tests { use super::render_dag_line; - fn node(id: u64, agent: &str, kind: &str, state: State, step: Option<&str>) -> NodeView { + fn node(id: u32, agent: &str, kind: &str, state: State, step: Option<&str>) -> NodeView { NodeView { id, agent: agent.to_owned(), @@ -353,6 +353,7 @@ mod tests { finished_at: None, inputs: vec![], approval_id: None, + perm_payload: None, nodes: vec![ node(0, "alice", "prebuild", State::Done, None), node(1, "alice", "stop_for_update", State::Done, None), @@ -391,6 +392,7 @@ mod tests { finished_at: Some(2), inputs: vec![], approval_id: None, + perm_payload: None, nodes: vec![failed], }; let line = render_dag_line(&dag);