hive-c0re: fail on a malformed agent name and on an unreadable container list
An agent name that is not a valid Ident made `Coordinator::agent_paths` panic. Job payloads carry names as plain strings (the swarm's published wanted state is one source), and a panic inside a job-queue node never reaches `complete_growing`, so the node's resources (the deploy window included) were held until hive-c0re restarted. `agent_paths` now returns an error; the job-queue nodes, the admin-socket spawn and set-limits paths, the root-agent spawn and the dashboard set-limits handler propagate it. `lifecycle::list().await.unwrap_or_default()` turned a failed container list into "no agents": - meta-update cascade: the lock bump committed and zero rebuilds fanned out, reported as success. The cascade is now resolved before the lock bump and a list failure fails the node. - dashboard update-all: queued nothing and returned 200 "ok". Now 500 with the error. - container rescan: every row was emitted as removed and the cache emptied. Now the last snapshot stands; `hivectl status` gets an error. - dashboard journal: answered 404 "no managed container". Now 500. - spawn/rebuild port-collision check: silently skipped. Now fails. - startup migration: the per-agent phases ran over nothing, and phase 3 handed an empty agent list to `meta::sync_agents`, which renders the meta flake with exactly the agents it is given. Both now log the list failure and skip. The hive-jobq scheduler still leaks a node's resources on any executor panic; that root is not addressed here. Refs #4723
This commit is contained in:
parent
ee25b7de20
commit
7ac6819652
10 changed files with 153 additions and 67 deletions
|
|
@ -392,7 +392,10 @@ pub(super) async fn post_resource_limits(
|
|||
}
|
||||
let agent_dir = crate::paths::agent_runtime_dir(ident.as_str());
|
||||
let hive = state.coord.hive_env();
|
||||
let paths = crate::coordinator::Coordinator::agent_paths(ident.as_str(), agent_dir);
|
||||
let paths = match crate::coordinator::Coordinator::agent_paths(ident.as_str(), agent_dir) {
|
||||
Ok(p) => p,
|
||||
Err(e) => return error_response(&format!("write_dropins {logical}: {e:#}")),
|
||||
};
|
||||
if let Err(e) = crate::lifecycle::write_dropins(ident.as_str(), &hive, &paths).await {
|
||||
return error_response(&format!("write_dropins {logical}: {e:#}"));
|
||||
}
|
||||
|
|
@ -405,18 +408,18 @@ pub(super) async fn post_resource_limits(
|
|||
#[utoipa::path(
|
||||
post,
|
||||
path = "/api/update-all",
|
||||
responses((status = 200, description = "rebuilds queued", body = String)),
|
||||
responses(
|
||||
(status = 200, description = "rebuilds queued", body = String),
|
||||
(status = 500, description = "container list unreadable, nothing queued"),
|
||||
),
|
||||
tag = "lifecycle_ops"
|
||||
)]
|
||||
pub(super) async fn post_update_all(State(state): State<AppState>) -> Response {
|
||||
let containers = lifecycle::list().await.unwrap_or_default();
|
||||
for container in containers {
|
||||
let Some(logical) = container
|
||||
.strip_prefix(lifecycle::AGENT_PREFIX)
|
||||
.map(str::to_owned)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let agents = match lifecycle::agent_names(lifecycle::list().await) {
|
||||
Ok(agents) => agents,
|
||||
Err(e) => return error_response(&format!("update-all: listing containers: {e:#}")),
|
||||
};
|
||||
for logical in agents {
|
||||
if let Err(e) = state.coord.job_queue.insert_job(|b| {
|
||||
crate::job_queue::templates::rebuild(b, &logical, true);
|
||||
Vec::new()
|
||||
|
|
|
|||
Loading…
Reference in a new issue