hyperhive/hive-c0re/src/dashboard/lifecycle_ops.rs
atlas 7ac6819652 hive-c0re: fail on a malformed agent name and on an unreadable container list
An agent name that is not a valid Ident made `Coordinator::agent_paths`
panic. Job payloads carry names as plain strings (the swarm's published
wanted state is one source), and a panic inside a job-queue node never
reaches `complete_growing`, so the node's resources (the deploy
window included) were held until hive-c0re restarted. `agent_paths` now
returns an error; the job-queue nodes, the admin-socket spawn and
set-limits paths, the root-agent spawn and the dashboard set-limits
handler propagate it.

`lifecycle::list().await.unwrap_or_default()` turned a failed container
list into "no agents":
- meta-update cascade: the lock bump committed and zero rebuilds fanned
  out, reported as success. The cascade is now resolved before the lock
  bump and a list failure fails the node.
- dashboard update-all: queued nothing and returned 200 "ok". Now 500
  with the error.
- container rescan: every row was emitted as removed and the cache
  emptied. Now the last snapshot stands; `hivectl status` gets an error.
- dashboard journal: answered 404 "no managed container". Now 500.
- spawn/rebuild port-collision check: silently skipped. Now fails.
- startup migration: the per-agent phases ran over nothing, and phase 3
  handed an empty agent list to `meta::sync_agents`, which renders the
  meta flake with exactly the agents it is given. Both now log the list
  failure and skip.

The hive-jobq scheduler still leaks a node's resources on any executor
panic; that root is not addressed here.

Refs #4723
2026-09-27 05:13:21 +02:00

472 lines
17 KiB
Rust

//! Container lifecycle endpoints for the dashboard.
//!
//! Rebuild / restart / start / stop (hard + graceful) / update-all all
//! insert DAGs into the job queue — the power ops via
//! [`crate::job_queue::power`], the static shapes straight through
//! `JobQueue::insert` — so each shows a visible queued→running transient on
//! the dashboard; a direct sub-second start/stop only flashed the badge.
//! Start/stop also persist the agent's `wanted` power intent before
//! inserting; the DAG's `Reconcile` converges to it. Destroy delegates to
//! `actions::destroy` (optionally purging).
use axum::{
extract::{Form, Path as AxumPath, Query, State},
http::StatusCode,
response::{IntoResponse, Response},
};
use serde::Deserialize;
use utoipa::{IntoParams, ToSchema};
/// Query params for `post_kill` / `post_restart`. `?graceful=1` routes to
/// the graceful-stop/-restart orchestration (quiesce the harness, flush
/// `/state`, then container stop/restart) instead of an immediate hard
/// action. Defaults false → today's hard kill/restart.
#[derive(Deserialize, IntoParams)]
pub(super) struct GracefulParams {
#[serde(default)]
graceful: bool,
}
use super::{AppState, Ident, error_response, guard_agent_name, strip_container_prefix};
use crate::{actions, lifecycle};
/// Queue a rebuild DAG for `name`.
#[utoipa::path(
post,
path = "/api/rebuild/{name}",
params(("name" = String, Path, description = "agent name")),
responses(
(status = 200, description = "rebuild queued", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_rebuild(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if let Err(e) = state.coord.job_queue.insert_job(|b| {
crate::job_queue::templates::rebuild(b, &logical, true);
Vec::new()
}) {
tracing::error!(agent = %logical, error = ?e, "rebuild: insert failed");
}
state.coord.emit_rebuild_queue_snapshot();
(StatusCode::OK, "ok").into_response()
}
/// Stop `name`, hard by default or
/// gracefully when `graceful=1`.
///
/// Graceful mode: quiesce → drain → stop.
#[utoipa::path(
post,
path = "/api/kill/{name}",
params(
("name" = String, Path, description = "agent name"),
GracefulParams,
),
responses(
(status = 200, description = "stop queued/performed", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_kill(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
Query(params): Query<GracefulParams>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if params.graceful {
// Graceful stop: submit the quiesce DAG (signal the harness →
// one stop-checkpoint turn → drain → container stop, with a
// timeout fallback to a hard stop). The agent's lifecycle
// lease keeps it from racing an in-flight rebuild for the same
// agent, and per-node progress surfaces on the queue snapshot.
if let Err(e) =
crate::job_queue::power::stop_many(&state.coord, std::slice::from_ref(&logical), true)
.await
{
tracing::error!(agent = %logical, error = ?e, "graceful stop: insert failed");
}
return (StatusCode::OK, "ok").into_response();
}
// Manager is stoppable from the dashboard like any other
// agent. The host's dashboard server keeps running (it's
// hive-c0re, not the manager container), per-agent approvals
// submitted by other sub-agents still process through the
// host-side approval queue without the manager up, and
// operator-driven meta-input updates work from the dashboard
// either way.
if let Err(e) =
crate::job_queue::power::stop_many(&state.coord, std::slice::from_ref(&logical), false)
.await
{
tracing::error!(agent = %logical, error = ?e, "stop: insert failed");
}
(StatusCode::OK, "ok").into_response()
}
/// Restart `name`, hard by default
/// or gracefully when `graceful=1`.
///
/// Graceful mode: quiesce → drain → restart.
#[utoipa::path(
post,
path = "/api/restart/{name}",
params(
("name" = String, Path, description = "agent name"),
GracefulParams,
),
responses(
(status = 200, description = "restart queued/performed", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_restart(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
Query(params): Query<GracefulParams>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if params.graceful {
if let Err(e) = crate::job_queue::power::restart_many(
&state.coord,
std::slice::from_ref(&logical),
true,
)
.await
{
tracing::error!(agent = %logical, error = ?e, "graceful restart: insert failed");
}
return (StatusCode::OK, "ok").into_response();
}
if let Err(e) =
crate::job_queue::power::restart_many(&state.coord, std::slice::from_ref(&logical), false)
.await
{
tracing::error!(agent = %logical, error = ?e, "restart: insert failed");
}
(StatusCode::OK, "ok").into_response()
}
/// Query params for `post_start`. `?paused=1` writes the pause marker
/// before (or instead of) starting — see `post_start`'s doc.
#[derive(Deserialize, IntoParams)]
pub(super) struct StartParams {
#[serde(default)]
paused: bool,
}
/// Start `name`, optionally paused.
///
/// Plain `?paused=1` mirrors `hivectl agent <name> start --paused`: if
/// `name` is already running, this just writes the pause marker in place
/// and returns without submitting a start DAG (nothing to start). If it's
/// down, the marker is written *before* the start DAG is submitted, so
/// the container comes up paused rather than racing the harness's own
/// pause-gate poll against an already-in-flight start.
#[utoipa::path(
post,
path = "/api/start/{name}",
params(
("name" = String, Path, description = "agent name"),
StartParams,
),
responses(
(status = 200, description = "start queued (or paused in place)", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
(status = 500, description = "pause marker write failed"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_start(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
Query(params): Query<StartParams>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if params.paused {
let ident = match Ident::parse(&logical) {
Ok(i) => i,
Err(e) => {
return (StatusCode::BAD_REQUEST, format!("bad agent name: {e}")).into_response();
}
};
let already_running = lifecycle::is_running(&logical).await;
if let Err(e) = crate::coordinator::Coordinator::set_paused(&ident, true).await {
return error_response(&format!("pause {logical}: {e}"));
}
state.coord.rescan_containers_and_emit().await;
if already_running {
// Already up — pausing in place is the whole request, no DAG
// to submit.
return (StatusCode::OK, "ok").into_response();
}
}
if let Err(e) =
crate::job_queue::power::start_many(&state.coord, std::slice::from_ref(&logical)).await
{
tracing::error!(agent = %logical, error = ?e, "start: insert failed");
}
(StatusCode::OK, "ok").into_response()
}
/// Write the pause marker for `name`.
///
/// Unlike the lifecycle ops above this is not a DAG: it writes a single
/// marker file, which the harness stats at the top of its serve loop.
/// Works on stopped containers too (the marker is sticky and takes effect
/// when the container next boots). Triggers an immediate rescan so the
/// `paused` badge flips on the dashboard without waiting for the next
/// periodic sweep.
#[utoipa::path(
post,
path = "/api/pause/{name}",
params(("name" = String, Path, description = "agent name")),
responses(
(status = 200, description = "pause marker written", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
(status = 500, description = "marker write failed"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_pause(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if let Err(e) = crate::coordinator::Coordinator::set_paused_by_name(&logical, true).await {
return set_paused_error_response(&logical, &e);
}
state.coord.rescan_containers_and_emit().await;
(StatusCode::OK, "ok").into_response()
}
/// Remove the pause marker for `name`.
///
/// The inverse of `post_pause`. Removing a non-existent marker is a no-op
/// (idempotent). Triggers an immediate rescan so the paused badge clears.
#[utoipa::path(
post,
path = "/api/resume/{name}",
params(("name" = String, Path, description = "agent name")),
responses(
(status = 200, description = "pause marker removed", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
(status = 500, description = "marker removal failed"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_resume(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
if let Err(e) = crate::coordinator::Coordinator::set_paused_by_name(&logical, false).await {
return set_paused_error_response(&logical, &e);
}
state.coord.rescan_containers_and_emit().await;
(StatusCode::OK, "ok").into_response()
}
/// Render a [`crate::coordinator::SetPausedByNameError`] as the response
/// `post_pause`/`post_resume` both need: 400 for a bad name (the caller's
/// mistake), 500 for a write failure (this host's).
fn set_paused_error_response(
logical: &str,
e: &crate::coordinator::SetPausedByNameError,
) -> Response {
match e {
crate::coordinator::SetPausedByNameError::BadName(_) => {
(StatusCode::BAD_REQUEST, e.to_string()).into_response()
}
crate::coordinator::SetPausedByNameError::Write(_) => {
error_response(&format!("{logical}: {e}"))
}
}
}
/// Form fields for `post_resource_limits`. Both fields are optional strings;
/// an empty value clears the per-agent override for that field, falling back
/// to the hive-wide default.
#[derive(Deserialize, Default, ToSchema)]
pub(super) struct ResourceLimitsForm {
#[serde(default)]
cpu_quota: String,
#[serde(default)]
memory_max: String,
}
/// Write per-agent CPU/memory limit
/// overrides for `name`.
///
/// An empty `cpu_quota` or `memory_max` field clears that field's override,
/// falling back to the hive-wide default. Both empty together removes the
/// agent's entry entirely. The new drop-in is written immediately — the
/// limits take effect on the next container start or restart. Triggers an
/// immediate rescan so `ContainerView.cpu_quota`/`memory_max` update on
/// the dashboard via SSE without waiting for the next periodic sweep.
#[utoipa::path(
post,
path = "/api/resource-limits/{name}",
params(("name" = String, Path, description = "agent name")),
request_body(content = ResourceLimitsForm, content_type = "application/x-www-form-urlencoded"),
responses(
(status = 200, description = "limits written", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
(status = 422, description = "invalid cpu_quota/memory_max value"),
(status = 500, description = "commit or drop-in write failed"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_resource_limits(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
Form(form): Form<ResourceLimitsForm>,
) -> Response {
let logical = strip_container_prefix(&name);
if let Some(reject) = guard_agent_name(&state, &logical).await {
return reject;
}
let ident = match Ident::parse(&logical) {
Ok(i) => i,
Err(e) => return (StatusCode::BAD_REQUEST, format!("bad agent name: {e}")).into_response(),
};
let cpu_quota = if form.cpu_quota.is_empty() {
None
} else {
Some(form.cpu_quota.as_str())
};
let memory_max = if form.memory_max.is_empty() {
None
} else {
Some(form.memory_max.as_str())
};
if let Some(v) = cpu_quota
&& let Err(e) = crate::resource_limits::validate_cpu_quota(v)
{
return (StatusCode::UNPROCESSABLE_ENTITY, e).into_response();
}
// Returns the value to store, not just an ok/err verdict — see
// `validate_memory_max`'s own doc for why that isn't always `v`.
let memory_max = match memory_max.map(crate::resource_limits::validate_memory_max) {
Some(Err(e)) => return (StatusCode::UNPROCESSABLE_ENTITY, e).into_response(),
Some(Ok(v)) => Some(v),
None => None,
};
let limits = crate::resource_limits::AgentLimits {
cpu_quota: cpu_quota.map(str::to_owned),
memory_max,
};
if let Err(e) = crate::meta::commit_resource_limits(ident.as_str(), &limits).await {
return error_response(&format!("set limits {logical}: {e:#}"));
}
let agent_dir = crate::paths::agent_runtime_dir(ident.as_str());
let hive = state.coord.hive_env();
let paths = match crate::coordinator::Coordinator::agent_paths(ident.as_str(), agent_dir) {
Ok(p) => p,
Err(e) => return error_response(&format!("write_dropins {logical}: {e:#}")),
};
if let Err(e) = crate::lifecycle::write_dropins(ident.as_str(), &hive, &paths).await {
return error_response(&format!("write_dropins {logical}: {e:#}"));
}
state.coord.rescan_containers_and_emit().await;
(StatusCode::OK, "ok").into_response()
}
/// Queue a rebuild DAG for every live agent
/// container.
#[utoipa::path(
post,
path = "/api/update-all",
responses(
(status = 200, description = "rebuilds queued", body = String),
(status = 500, description = "container list unreadable, nothing queued"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_update_all(State(state): State<AppState>) -> Response {
let agents = match lifecycle::agent_names(lifecycle::list().await) {
Ok(agents) => agents,
Err(e) => return error_response(&format!("update-all: listing containers: {e:#}")),
};
for logical in agents {
if let Err(e) = state.coord.job_queue.insert_job(|b| {
crate::job_queue::templates::rebuild(b, &logical, true);
Vec::new()
}) {
tracing::error!(agent = %logical, error = ?e, "update-all: insert failed");
}
}
state.coord.emit_rebuild_queue_snapshot();
(StatusCode::OK, "ok").into_response()
}
#[derive(Deserialize, Default, ToSchema)]
pub(super) struct DestroyForm {
#[serde(default)]
purge: Option<String>,
}
/// Destroy `name`'s container.
///
/// Form field `purge` (any non-empty value, e.g. `"on"`) also wipes the
/// retained state dir instead of leaving a tombstone.
#[utoipa::path(
post,
path = "/api/destroy/{name}",
params(("name" = String, Path, description = "agent name")),
request_body(content = DestroyForm, content_type = "application/x-www-form-urlencoded"),
responses(
(status = 200, description = "destroy queued", body = String),
(status = 400, description = "bad agent name"),
(status = 404, description = "no such agent"),
),
tag = "lifecycle_ops"
)]
pub(super) async fn post_destroy(
State(state): State<AppState>,
AxumPath(name): AxumPath<String>,
Form(form): Form<DestroyForm>,
) -> Response {
if let Some(reject) = guard_agent_name(&state, &name).await {
return reject;
}
// Checkbox semantics: any non-empty value (axum sends "on") = purge.
let purge = form.purge.as_deref().is_some_and(|v| !v.is_empty());
// Submit-and-return, like every other lifecycle endpoint here. The
// container rescan now runs in the DAG's bookkeeping tail, so
// `ContainerRemoved` arrives *after* this 200 rather than before it — the
// row disappears when the event lands, same as a rebuild's does.
actions::destroy(&state.coord, &name, purge);
(StatusCode::OK, "ok").into_response()
}