hive-c0re: fail on a malformed agent name and on an unreadable container list
An agent name that is not a valid Ident made `Coordinator::agent_paths` panic. Job payloads carry names as plain strings (the swarm's published wanted state is one source), and a panic inside a job-queue node never reaches `complete_growing`, so the node's resources (the deploy window included) were held until hive-c0re restarted. `agent_paths` now returns an error; the job-queue nodes, the admin-socket spawn and set-limits paths, the root-agent spawn and the dashboard set-limits handler propagate it. `lifecycle::list().await.unwrap_or_default()` turned a failed container list into "no agents": - meta-update cascade: the lock bump committed and zero rebuilds fanned out, reported as success. The cascade is now resolved before the lock bump and a list failure fails the node. - dashboard update-all: queued nothing and returned 200 "ok". Now 500 with the error. - container rescan: every row was emitted as removed and the cache emptied. Now the last snapshot stands; `hivectl status` gets an error. - dashboard journal: answered 404 "no managed container". Now 500. - spawn/rebuild port-collision check: silently skipped. Now fails. - startup migration: the per-agent phases ran over nothing, and phase 3 handed an empty agent list to `meta::sync_agents`, which renders the meta flake with exactly the agents it is given. Both now log the list failure and skip. The hive-jobq scheduler still leaks a node's resources on any executor panic; that root is not addressed here. Refs #4723
This commit is contained in:
parent
ee25b7de20
commit
7ac6819652
10 changed files with 153 additions and 67 deletions
|
|
@ -293,7 +293,7 @@ async fn handle_spawn(coord: &Arc<Coordinator>, name: &str) -> Result<HostRespon
|
|||
tracing::info!(%name, "spawn");
|
||||
let agent_dir = crate::paths::agent_runtime_dir(name);
|
||||
let hive = coord.hive_env();
|
||||
let paths = Coordinator::agent_paths(name, agent_dir);
|
||||
let paths = Coordinator::agent_paths(name, agent_dir)?;
|
||||
// lifecycle::spawn creates the runtime dir internally before start.
|
||||
// MCP listener registration is event-driven: bind immediately on
|
||||
// success so the harness can connect on its first turn without
|
||||
|
|
@ -375,12 +375,15 @@ async fn handle_set_paused(
|
|||
|
||||
/// Collect per-agent status rows for `hivectl status` and the dashboard.
|
||||
async fn handle_agent_status(coord: &Arc<Coordinator>) -> HostResponse {
|
||||
let rows = crate::container_view::build_all(&coord.hive_env())
|
||||
.await
|
||||
.into_iter()
|
||||
.map(hive_sh4re::container::AgentStatusRow::from)
|
||||
.collect();
|
||||
HostResponse::agent_statuses(rows)
|
||||
match crate::container_view::build_all(&coord.hive_env()).await {
|
||||
Ok(views) => HostResponse::agent_statuses(
|
||||
views
|
||||
.into_iter()
|
||||
.map(hive_sh4re::container::AgentStatusRow::from)
|
||||
.collect(),
|
||||
),
|
||||
Err(e) => HostResponse::error(format!("listing containers: {e:#}")),
|
||||
}
|
||||
}
|
||||
|
||||
/// `SubscribeAgentStatus` — ack once, then push one single-row
|
||||
|
|
@ -502,7 +505,7 @@ async fn handle_set_resource_limits(
|
|||
// in the JSON until the agent's next spawn or rebuild.
|
||||
let agent_dir = crate::paths::agent_runtime_dir(name.as_str());
|
||||
let hive = coord.hive_env();
|
||||
let paths = Coordinator::agent_paths(name.as_str(), agent_dir);
|
||||
let paths = Coordinator::agent_paths(name.as_str(), agent_dir)?;
|
||||
crate::lifecycle::write_dropins(name.as_str(), &hive, &paths).await?;
|
||||
|
||||
let (cpu, mem) = crate::resource_limits::effective(
|
||||
|
|
|
|||
Loading…
Reference in a new issue