An agent name that is not a valid Ident made `Coordinator::agent_paths` panic. Job payloads carry names as plain strings (the swarm's published wanted state is one source), and a panic inside a job-queue node never reaches `complete_growing`, so the node's resources (the deploy window included) were held until hive-c0re restarted. `agent_paths` now returns an error; the job-queue nodes, the admin-socket spawn and set-limits paths, the root-agent spawn and the dashboard set-limits handler propagate it. `lifecycle::list().await.unwrap_or_default()` turned a failed container list into "no agents": - meta-update cascade: the lock bump committed and zero rebuilds fanned out, reported as success. The cascade is now resolved before the lock bump and a list failure fails the node. - dashboard update-all: queued nothing and returned 200 "ok". Now 500 with the error. - container rescan: every row was emitted as removed and the cache emptied. Now the last snapshot stands; `hivectl status` gets an error. - dashboard journal: answered 404 "no managed container". Now 500. - spawn/rebuild port-collision check: silently skipped. Now fails. - startup migration: the per-agent phases ran over nothing, and phase 3 handed an empty agent list to `meta::sync_agents`, which renders the meta flake with exactly the agents it is given. Both now log the list failure and skip. The hive-jobq scheduler still leaks a node's resources on any executor panic; that root is not addressed here. Refs #4723
472 lines
17 KiB
Rust
472 lines
17 KiB
Rust
//! Container lifecycle endpoints for the dashboard.
|
|
//!
|
|
//! Rebuild / restart / start / stop (hard + graceful) / update-all all
|
|
//! insert DAGs into the job queue — the power ops via
|
|
//! [`crate::job_queue::power`], the static shapes straight through
|
|
//! `JobQueue::insert` — so each shows a visible queued→running transient on
|
|
//! the dashboard; a direct sub-second start/stop only flashed the badge.
|
|
//! Start/stop also persist the agent's `wanted` power intent before
|
|
//! inserting; the DAG's `Reconcile` converges to it. Destroy delegates to
|
|
//! `actions::destroy` (optionally purging).
|
|
|
|
use axum::{
|
|
extract::{Form, Path as AxumPath, Query, State},
|
|
http::StatusCode,
|
|
response::{IntoResponse, Response},
|
|
};
|
|
use serde::Deserialize;
|
|
use utoipa::{IntoParams, ToSchema};
|
|
|
|
/// Query params for `post_kill` / `post_restart`. `?graceful=1` routes to
|
|
/// the graceful-stop/-restart orchestration (quiesce the harness, flush
|
|
/// `/state`, then container stop/restart) instead of an immediate hard
|
|
/// action. Defaults false → today's hard kill/restart.
|
|
#[derive(Deserialize, IntoParams)]
|
|
pub(super) struct GracefulParams {
|
|
#[serde(default)]
|
|
graceful: bool,
|
|
}
|
|
|
|
use super::{AppState, Ident, error_response, guard_agent_name, strip_container_prefix};
|
|
use crate::{actions, lifecycle};
|
|
|
|
/// Queue a rebuild DAG for `name`.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/rebuild/{name}",
|
|
params(("name" = String, Path, description = "agent name")),
|
|
responses(
|
|
(status = 200, description = "rebuild queued", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_rebuild(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if let Err(e) = state.coord.job_queue.insert_job(|b| {
|
|
crate::job_queue::templates::rebuild(b, &logical, true);
|
|
Vec::new()
|
|
}) {
|
|
tracing::error!(agent = %logical, error = ?e, "rebuild: insert failed");
|
|
}
|
|
state.coord.emit_rebuild_queue_snapshot();
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Stop `name`, hard by default or
|
|
/// gracefully when `graceful=1`.
|
|
///
|
|
/// Graceful mode: quiesce → drain → stop.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/kill/{name}",
|
|
params(
|
|
("name" = String, Path, description = "agent name"),
|
|
GracefulParams,
|
|
),
|
|
responses(
|
|
(status = 200, description = "stop queued/performed", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_kill(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
Query(params): Query<GracefulParams>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if params.graceful {
|
|
// Graceful stop: submit the quiesce DAG (signal the harness →
|
|
// one stop-checkpoint turn → drain → container stop, with a
|
|
// timeout fallback to a hard stop). The agent's lifecycle
|
|
// lease keeps it from racing an in-flight rebuild for the same
|
|
// agent, and per-node progress surfaces on the queue snapshot.
|
|
if let Err(e) =
|
|
crate::job_queue::power::stop_many(&state.coord, std::slice::from_ref(&logical), true)
|
|
.await
|
|
{
|
|
tracing::error!(agent = %logical, error = ?e, "graceful stop: insert failed");
|
|
}
|
|
return (StatusCode::OK, "ok").into_response();
|
|
}
|
|
// Manager is stoppable from the dashboard like any other
|
|
// agent. The host's dashboard server keeps running (it's
|
|
// hive-c0re, not the manager container), per-agent approvals
|
|
// submitted by other sub-agents still process through the
|
|
// host-side approval queue without the manager up, and
|
|
// operator-driven meta-input updates work from the dashboard
|
|
// either way.
|
|
if let Err(e) =
|
|
crate::job_queue::power::stop_many(&state.coord, std::slice::from_ref(&logical), false)
|
|
.await
|
|
{
|
|
tracing::error!(agent = %logical, error = ?e, "stop: insert failed");
|
|
}
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Restart `name`, hard by default
|
|
/// or gracefully when `graceful=1`.
|
|
///
|
|
/// Graceful mode: quiesce → drain → restart.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/restart/{name}",
|
|
params(
|
|
("name" = String, Path, description = "agent name"),
|
|
GracefulParams,
|
|
),
|
|
responses(
|
|
(status = 200, description = "restart queued/performed", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_restart(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
Query(params): Query<GracefulParams>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if params.graceful {
|
|
if let Err(e) = crate::job_queue::power::restart_many(
|
|
&state.coord,
|
|
std::slice::from_ref(&logical),
|
|
true,
|
|
)
|
|
.await
|
|
{
|
|
tracing::error!(agent = %logical, error = ?e, "graceful restart: insert failed");
|
|
}
|
|
return (StatusCode::OK, "ok").into_response();
|
|
}
|
|
if let Err(e) =
|
|
crate::job_queue::power::restart_many(&state.coord, std::slice::from_ref(&logical), false)
|
|
.await
|
|
{
|
|
tracing::error!(agent = %logical, error = ?e, "restart: insert failed");
|
|
}
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Query params for `post_start`. `?paused=1` writes the pause marker
|
|
/// before (or instead of) starting — see `post_start`'s doc.
|
|
#[derive(Deserialize, IntoParams)]
|
|
pub(super) struct StartParams {
|
|
#[serde(default)]
|
|
paused: bool,
|
|
}
|
|
|
|
/// Start `name`, optionally paused.
|
|
///
|
|
/// Plain `?paused=1` mirrors `hivectl agent <name> start --paused`: if
|
|
/// `name` is already running, this just writes the pause marker in place
|
|
/// and returns without submitting a start DAG (nothing to start). If it's
|
|
/// down, the marker is written *before* the start DAG is submitted, so
|
|
/// the container comes up paused rather than racing the harness's own
|
|
/// pause-gate poll against an already-in-flight start.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/start/{name}",
|
|
params(
|
|
("name" = String, Path, description = "agent name"),
|
|
StartParams,
|
|
),
|
|
responses(
|
|
(status = 200, description = "start queued (or paused in place)", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
(status = 500, description = "pause marker write failed"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_start(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
Query(params): Query<StartParams>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if params.paused {
|
|
let ident = match Ident::parse(&logical) {
|
|
Ok(i) => i,
|
|
Err(e) => {
|
|
return (StatusCode::BAD_REQUEST, format!("bad agent name: {e}")).into_response();
|
|
}
|
|
};
|
|
let already_running = lifecycle::is_running(&logical).await;
|
|
if let Err(e) = crate::coordinator::Coordinator::set_paused(&ident, true).await {
|
|
return error_response(&format!("pause {logical}: {e}"));
|
|
}
|
|
state.coord.rescan_containers_and_emit().await;
|
|
if already_running {
|
|
// Already up — pausing in place is the whole request, no DAG
|
|
// to submit.
|
|
return (StatusCode::OK, "ok").into_response();
|
|
}
|
|
}
|
|
if let Err(e) =
|
|
crate::job_queue::power::start_many(&state.coord, std::slice::from_ref(&logical)).await
|
|
{
|
|
tracing::error!(agent = %logical, error = ?e, "start: insert failed");
|
|
}
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Write the pause marker for `name`.
|
|
///
|
|
/// Unlike the lifecycle ops above this is not a DAG: it writes a single
|
|
/// marker file, which the harness stats at the top of its serve loop.
|
|
/// Works on stopped containers too (the marker is sticky and takes effect
|
|
/// when the container next boots). Triggers an immediate rescan so the
|
|
/// `paused` badge flips on the dashboard without waiting for the next
|
|
/// periodic sweep.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/pause/{name}",
|
|
params(("name" = String, Path, description = "agent name")),
|
|
responses(
|
|
(status = 200, description = "pause marker written", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
(status = 500, description = "marker write failed"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_pause(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if let Err(e) = crate::coordinator::Coordinator::set_paused_by_name(&logical, true).await {
|
|
return set_paused_error_response(&logical, &e);
|
|
}
|
|
state.coord.rescan_containers_and_emit().await;
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Remove the pause marker for `name`.
|
|
///
|
|
/// The inverse of `post_pause`. Removing a non-existent marker is a no-op
|
|
/// (idempotent). Triggers an immediate rescan so the paused badge clears.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/resume/{name}",
|
|
params(("name" = String, Path, description = "agent name")),
|
|
responses(
|
|
(status = 200, description = "pause marker removed", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
(status = 500, description = "marker removal failed"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_resume(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
if let Err(e) = crate::coordinator::Coordinator::set_paused_by_name(&logical, false).await {
|
|
return set_paused_error_response(&logical, &e);
|
|
}
|
|
state.coord.rescan_containers_and_emit().await;
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Render a [`crate::coordinator::SetPausedByNameError`] as the response
|
|
/// `post_pause`/`post_resume` both need: 400 for a bad name (the caller's
|
|
/// mistake), 500 for a write failure (this host's).
|
|
fn set_paused_error_response(
|
|
logical: &str,
|
|
e: &crate::coordinator::SetPausedByNameError,
|
|
) -> Response {
|
|
match e {
|
|
crate::coordinator::SetPausedByNameError::BadName(_) => {
|
|
(StatusCode::BAD_REQUEST, e.to_string()).into_response()
|
|
}
|
|
crate::coordinator::SetPausedByNameError::Write(_) => {
|
|
error_response(&format!("{logical}: {e}"))
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Form fields for `post_resource_limits`. Both fields are optional strings;
|
|
/// an empty value clears the per-agent override for that field, falling back
|
|
/// to the hive-wide default.
|
|
#[derive(Deserialize, Default, ToSchema)]
|
|
pub(super) struct ResourceLimitsForm {
|
|
#[serde(default)]
|
|
cpu_quota: String,
|
|
#[serde(default)]
|
|
memory_max: String,
|
|
}
|
|
|
|
/// Write per-agent CPU/memory limit
|
|
/// overrides for `name`.
|
|
///
|
|
/// An empty `cpu_quota` or `memory_max` field clears that field's override,
|
|
/// falling back to the hive-wide default. Both empty together removes the
|
|
/// agent's entry entirely. The new drop-in is written immediately — the
|
|
/// limits take effect on the next container start or restart. Triggers an
|
|
/// immediate rescan so `ContainerView.cpu_quota`/`memory_max` update on
|
|
/// the dashboard via SSE without waiting for the next periodic sweep.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/resource-limits/{name}",
|
|
params(("name" = String, Path, description = "agent name")),
|
|
request_body(content = ResourceLimitsForm, content_type = "application/x-www-form-urlencoded"),
|
|
responses(
|
|
(status = 200, description = "limits written", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
(status = 422, description = "invalid cpu_quota/memory_max value"),
|
|
(status = 500, description = "commit or drop-in write failed"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_resource_limits(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
Form(form): Form<ResourceLimitsForm>,
|
|
) -> Response {
|
|
let logical = strip_container_prefix(&name);
|
|
if let Some(reject) = guard_agent_name(&state, &logical).await {
|
|
return reject;
|
|
}
|
|
let ident = match Ident::parse(&logical) {
|
|
Ok(i) => i,
|
|
Err(e) => return (StatusCode::BAD_REQUEST, format!("bad agent name: {e}")).into_response(),
|
|
};
|
|
let cpu_quota = if form.cpu_quota.is_empty() {
|
|
None
|
|
} else {
|
|
Some(form.cpu_quota.as_str())
|
|
};
|
|
let memory_max = if form.memory_max.is_empty() {
|
|
None
|
|
} else {
|
|
Some(form.memory_max.as_str())
|
|
};
|
|
if let Some(v) = cpu_quota
|
|
&& let Err(e) = crate::resource_limits::validate_cpu_quota(v)
|
|
{
|
|
return (StatusCode::UNPROCESSABLE_ENTITY, e).into_response();
|
|
}
|
|
// Returns the value to store, not just an ok/err verdict — see
|
|
// `validate_memory_max`'s own doc for why that isn't always `v`.
|
|
let memory_max = match memory_max.map(crate::resource_limits::validate_memory_max) {
|
|
Some(Err(e)) => return (StatusCode::UNPROCESSABLE_ENTITY, e).into_response(),
|
|
Some(Ok(v)) => Some(v),
|
|
None => None,
|
|
};
|
|
let limits = crate::resource_limits::AgentLimits {
|
|
cpu_quota: cpu_quota.map(str::to_owned),
|
|
memory_max,
|
|
};
|
|
if let Err(e) = crate::meta::commit_resource_limits(ident.as_str(), &limits).await {
|
|
return error_response(&format!("set limits {logical}: {e:#}"));
|
|
}
|
|
let agent_dir = crate::paths::agent_runtime_dir(ident.as_str());
|
|
let hive = state.coord.hive_env();
|
|
let paths = match crate::coordinator::Coordinator::agent_paths(ident.as_str(), agent_dir) {
|
|
Ok(p) => p,
|
|
Err(e) => return error_response(&format!("write_dropins {logical}: {e:#}")),
|
|
};
|
|
if let Err(e) = crate::lifecycle::write_dropins(ident.as_str(), &hive, &paths).await {
|
|
return error_response(&format!("write_dropins {logical}: {e:#}"));
|
|
}
|
|
state.coord.rescan_containers_and_emit().await;
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
/// Queue a rebuild DAG for every live agent
|
|
/// container.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/update-all",
|
|
responses(
|
|
(status = 200, description = "rebuilds queued", body = String),
|
|
(status = 500, description = "container list unreadable, nothing queued"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_update_all(State(state): State<AppState>) -> Response {
|
|
let agents = match lifecycle::agent_names(lifecycle::list().await) {
|
|
Ok(agents) => agents,
|
|
Err(e) => return error_response(&format!("update-all: listing containers: {e:#}")),
|
|
};
|
|
for logical in agents {
|
|
if let Err(e) = state.coord.job_queue.insert_job(|b| {
|
|
crate::job_queue::templates::rebuild(b, &logical, true);
|
|
Vec::new()
|
|
}) {
|
|
tracing::error!(agent = %logical, error = ?e, "update-all: insert failed");
|
|
}
|
|
}
|
|
state.coord.emit_rebuild_queue_snapshot();
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|
|
|
|
#[derive(Deserialize, Default, ToSchema)]
|
|
pub(super) struct DestroyForm {
|
|
#[serde(default)]
|
|
purge: Option<String>,
|
|
}
|
|
|
|
/// Destroy `name`'s container.
|
|
///
|
|
/// Form field `purge` (any non-empty value, e.g. `"on"`) also wipes the
|
|
/// retained state dir instead of leaving a tombstone.
|
|
#[utoipa::path(
|
|
post,
|
|
path = "/api/destroy/{name}",
|
|
params(("name" = String, Path, description = "agent name")),
|
|
request_body(content = DestroyForm, content_type = "application/x-www-form-urlencoded"),
|
|
responses(
|
|
(status = 200, description = "destroy queued", body = String),
|
|
(status = 400, description = "bad agent name"),
|
|
(status = 404, description = "no such agent"),
|
|
),
|
|
tag = "lifecycle_ops"
|
|
)]
|
|
pub(super) async fn post_destroy(
|
|
State(state): State<AppState>,
|
|
AxumPath(name): AxumPath<String>,
|
|
Form(form): Form<DestroyForm>,
|
|
) -> Response {
|
|
if let Some(reject) = guard_agent_name(&state, &name).await {
|
|
return reject;
|
|
}
|
|
// Checkbox semantics: any non-empty value (axum sends "on") = purge.
|
|
let purge = form.purge.as_deref().is_some_and(|v| !v.is_empty());
|
|
// Submit-and-return, like every other lifecycle endpoint here. The
|
|
// container rescan now runs in the DAG's bookkeeping tail, so
|
|
// `ContainerRemoved` arrives *after* this 200 rather than before it — the
|
|
// row disappears when the event lands, same as a rebuild's does.
|
|
actions::destroy(&state.coord, &name, purge);
|
|
(StatusCode::OK, "ok").into_response()
|
|
}
|