feat(#3139): tell a container that gave up from one stopped on purpose
is_running collapsed every non-active state into false, so a container that exhausted its bounded restarts read as plain "down" -- indistinguishable from one an operator stopped deliberately. Bounding the restarts made that gap sharper: a slow-failing agent used to grind on visibly, now it can stop quietly. Adds UnitState + unit_state() beside is_running rather than widening it. is_running has ~8 call sites and nearly all are reconcile/power logic asking "is it up? if not, start it" -- a question with two answers. Only the view builder needs more, and it gets both facts from one systemctl call, since is-active prints the state when not passed --quiet. Surfaces as a flat failed flag on ContainerView and AgentStatusRow, matching the shape those types already document: independent, orthogonally-observed facts rather than a state machine. serde(default) keeps it order-independent with the frontend half. No behaviour change: nothing acts on the flag, per the ruling.
This commit is contained in:
parent
4fd25b7f45
commit
cae2cf8df6
6 changed files with 128 additions and 2 deletions
|
|
@ -11,7 +11,7 @@ use std::path::Path;
|
||||||
use serde::Serialize;
|
use serde::Serialize;
|
||||||
|
|
||||||
use crate::coordinator::Coordinator;
|
use crate::coordinator::Coordinator;
|
||||||
use crate::lifecycle::{self, AGENT_PREFIX};
|
use crate::lifecycle::{self, AGENT_PREFIX, UnitState};
|
||||||
|
|
||||||
// Independent per-agent flags, each its own badge on the dashboard card
|
// Independent per-agent flags, each its own badge on the dashboard card
|
||||||
// and each diffed separately by `rescan_containers_and_emit`. Grouping
|
// and each diffed separately by `rescan_containers_and_emit`. Grouping
|
||||||
|
|
@ -32,6 +32,16 @@ pub struct ContainerView {
|
||||||
pub container: String,
|
pub container: String,
|
||||||
pub port: u16,
|
pub port: u16,
|
||||||
pub running: bool,
|
pub running: bool,
|
||||||
|
/// The container's unit is in systemd's `failed` state — it exhausted
|
||||||
|
/// its bounded restarts and gave up, rather than being stopped
|
||||||
|
/// deliberately.
|
||||||
|
///
|
||||||
|
/// Orthogonal to `running` rather than a variant of it: a failed unit
|
||||||
|
/// is not running, but a not-running unit is usually just *off*. That
|
||||||
|
/// distinction is the whole point — without it a container that gave
|
||||||
|
/// up is indistinguishable from one an operator stopped on purpose.
|
||||||
|
#[serde(default)]
|
||||||
|
pub failed: bool,
|
||||||
pub needs_update: bool,
|
pub needs_update: bool,
|
||||||
pub needs_login: bool,
|
pub needs_login: bool,
|
||||||
/// First 12 chars of the sha the meta flake currently has locked
|
/// First 12 chars of the sha the meta flake currently has locked
|
||||||
|
|
@ -110,7 +120,12 @@ pub async fn build_all(hive: &crate::coordinator::HiveEnv) -> Vec<ContainerView>
|
||||||
crate::auto_update::agent_config_pending(logical.as_str(), deployed_full).await;
|
crate::auto_update::agent_config_pending(logical.as_str(), deployed_full).await;
|
||||||
let deployed_sha = deployed_full.map(|s| s[..s.len().min(12)].to_owned());
|
let deployed_sha = deployed_full.map(|s| s[..s.len().min(12)].to_owned());
|
||||||
let parent = topology.get(logical.as_str()).cloned().flatten();
|
let parent = topology.get(logical.as_str()).cloned().flatten();
|
||||||
let running = lifecycle::is_running(logical.as_str()).await;
|
// One `systemctl` call for both facts: `unit_state` is the same
|
||||||
|
// shell-out `is_running` makes, minus `--quiet`. Asking twice would
|
||||||
|
// double the per-agent subprocess count on every SSE scan.
|
||||||
|
let state = lifecycle::unit_state(logical.as_str()).await;
|
||||||
|
let running = state == UnitState::Active;
|
||||||
|
let failed = state == UnitState::Failed;
|
||||||
// needs_login fires when EITHER the claude session dir is missing
|
// needs_login fires when EITHER the claude session dir is missing
|
||||||
// (boot-time / fresh container) OR the harness wrote the auth-failed
|
// (boot-time / fresh container) OR the harness wrote the auth-failed
|
||||||
// sentinel because a turn hit 401. Cleared for stopped containers —
|
// sentinel because a turn hit 401. Cleared for stopped containers —
|
||||||
|
|
@ -136,6 +151,7 @@ pub async fn build_all(hive: &crate::coordinator::HiveEnv) -> Vec<ContainerView>
|
||||||
out.push(ContainerView {
|
out.push(ContainerView {
|
||||||
port: lifecycle::agent_web_port(logical.as_str()),
|
port: lifecycle::agent_web_port(logical.as_str()),
|
||||||
running,
|
running,
|
||||||
|
failed,
|
||||||
container: c.clone(),
|
container: c.clone(),
|
||||||
name: logical.into_string(),
|
name: logical.into_string(),
|
||||||
needs_update,
|
needs_update,
|
||||||
|
|
|
||||||
|
|
@ -530,8 +530,70 @@ async fn start_with_fallback_inner(name: &str) -> Result<()> {
|
||||||
.with_context(|| format!("cold-start fallback also failed for {name}"))
|
.with_context(|| format!("cold-start fallback also failed for {name}"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The part of a container unit's systemd state we act on.
|
||||||
|
///
|
||||||
|
/// Deliberately not the full set: everything outside `Active`/`Failed` is
|
||||||
|
/// one bucket, because the only distinction anything downstream makes is
|
||||||
|
/// *gave up* versus *anything else*.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum UnitState {
|
||||||
|
/// Running.
|
||||||
|
Active,
|
||||||
|
/// systemd gave up on it — most often by exhausting the bounded start
|
||||||
|
/// limit. Its own variant precisely because it is the difference
|
||||||
|
/// between **gave up** and **stopped on purpose**, which every other
|
||||||
|
/// non-running state collapses into.
|
||||||
|
Failed,
|
||||||
|
/// `inactive` (the deliberate stop), `activating`, `deactivating`, an
|
||||||
|
/// unknown unit, or a `systemctl` we could not run at all.
|
||||||
|
Other,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl UnitState {
|
||||||
|
/// Parse `systemctl is-active`'s stdout.
|
||||||
|
///
|
||||||
|
/// Split out from the shell-out so the mapping is testable without a
|
||||||
|
/// systemd to ask — the states we care about are the two we name, and
|
||||||
|
/// everything else must land in `Other` rather than being guessed at.
|
||||||
|
fn from_is_active(stdout: &str) -> Self {
|
||||||
|
match stdout.trim() {
|
||||||
|
"active" => Self::Active,
|
||||||
|
"failed" => Self::Failed,
|
||||||
|
_ => Self::Other,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The container unit's state, in one `systemctl` call.
|
||||||
|
///
|
||||||
|
/// `is-active` **prints** the state and exits 0 only for `active`, so
|
||||||
|
/// dropping `--quiet` yields both facts from the call [`is_running`]
|
||||||
|
/// already makes. That matters: the status path runs this per agent (see
|
||||||
|
/// the spawn-count note in `socket_server::lifecycle_handlers`), so a
|
||||||
|
/// caller wanting *running* and *failed* must not pay for two subprocesses.
|
||||||
|
pub async fn unit_state(name: &str) -> UnitState {
|
||||||
|
let container = container_name(name);
|
||||||
|
let unit = format!("container@{container}.service");
|
||||||
|
match Command::new("systemctl")
|
||||||
|
.args(["is-active", &unit])
|
||||||
|
.output()
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(out) => UnitState::from_is_active(&String::from_utf8_lossy(&out.stdout)),
|
||||||
|
// Not "the unit is fine" and not "the unit failed" — we simply do
|
||||||
|
// not know, and saying `Failed` here would report a gave-up agent
|
||||||
|
// every time `systemctl` itself was unavailable.
|
||||||
|
Err(_) => UnitState::Other,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// True when the container's systemd unit is active. Used by the dashboard
|
/// True when the container's systemd unit is active. Used by the dashboard
|
||||||
/// to gate stop/restart buttons.
|
/// to gate stop/restart buttons.
|
||||||
|
///
|
||||||
|
/// Kept boolean on purpose: nearly every caller is reconcile/power logic
|
||||||
|
/// asking "is it up? if not, start it", and that question has two answers.
|
||||||
|
/// A caller that needs to tell *failed* from *stopped* wants
|
||||||
|
/// [`unit_state`] instead.
|
||||||
pub async fn is_running(name: &str) -> bool {
|
pub async fn is_running(name: &str) -> bool {
|
||||||
let container = container_name(name);
|
let container = container_name(name);
|
||||||
let unit = format!("container@{container}.service");
|
let unit = format!("container@{container}.service");
|
||||||
|
|
|
||||||
|
|
@ -180,3 +180,42 @@ async fn update_ref_cas_refuses_stale_expectation() {
|
||||||
"ref moved even though the CAS failed"
|
"ref moved even though the CAS failed"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `systemctl is-active` prints the state with a trailing newline, which is
|
||||||
|
/// the whole reason this parser trims rather than comparing raw.
|
||||||
|
#[test]
|
||||||
|
fn unit_state_reads_what_is_active_prints() {
|
||||||
|
assert_eq!(UnitState::from_is_active("active\n"), UnitState::Active);
|
||||||
|
assert_eq!(UnitState::from_is_active("failed\n"), UnitState::Failed);
|
||||||
|
assert_eq!(UnitState::from_is_active("inactive\n"), UnitState::Other);
|
||||||
|
// No trailing newline, in case the capture ever changes shape.
|
||||||
|
assert_eq!(UnitState::from_is_active("failed"), UnitState::Failed);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The asymmetry that matters: an unrecognised state must fall to `Other`,
|
||||||
|
/// never to `Failed`.
|
||||||
|
///
|
||||||
|
/// `Failed` is what the dashboard will render as *this agent gave up*, so a
|
||||||
|
/// wrong `Failed` invents an incident, while a wrong `Other` merely fails to
|
||||||
|
/// distinguish one from a deliberate stop — the behaviour we have today.
|
||||||
|
/// Guessing in the safe direction is the property, not an implementation
|
||||||
|
/// detail of the `match`.
|
||||||
|
#[test]
|
||||||
|
fn an_unknown_state_is_never_reported_as_failed() {
|
||||||
|
for s in [
|
||||||
|
"activating",
|
||||||
|
"deactivating",
|
||||||
|
"reloading",
|
||||||
|
"maintenance",
|
||||||
|
"unknown",
|
||||||
|
"",
|
||||||
|
"Failed", // capitalised: systemd prints lowercase, so this is not a state
|
||||||
|
"failed*", // `is-active` never emits this; a substring match would take it
|
||||||
|
] {
|
||||||
|
assert_eq!(
|
||||||
|
UnitState::from_is_active(s),
|
||||||
|
UnitState::Other,
|
||||||
|
"{s:?} must not be read as a give-up"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
|
||||||
|
|
@ -357,6 +357,7 @@ async fn handle_agent_status(coord: &Arc<Coordinator>) -> HostResponse {
|
||||||
.map(|v| hive_sh4re::container::AgentStatusRow {
|
.map(|v| hive_sh4re::container::AgentStatusRow {
|
||||||
name: v.name,
|
name: v.name,
|
||||||
running: v.running,
|
running: v.running,
|
||||||
|
failed: v.failed,
|
||||||
needs_update: v.needs_update,
|
needs_update: v.needs_update,
|
||||||
needs_login: v.needs_login,
|
needs_login: v.needs_login,
|
||||||
deployed_sha: v.deployed_sha,
|
deployed_sha: v.deployed_sha,
|
||||||
|
|
|
||||||
|
|
@ -248,6 +248,7 @@ mod tests {
|
||||||
container: format!("h-{name}"),
|
container: format!("h-{name}"),
|
||||||
port: 0,
|
port: 0,
|
||||||
running: true,
|
running: true,
|
||||||
|
failed: false,
|
||||||
needs_update: false,
|
needs_update: false,
|
||||||
needs_login,
|
needs_login,
|
||||||
deployed_sha: None,
|
deployed_sha: None,
|
||||||
|
|
|
||||||
|
|
@ -35,6 +35,13 @@ pub struct AgentStatusRow {
|
||||||
pub name: String,
|
pub name: String,
|
||||||
/// Whether the container is currently running.
|
/// Whether the container is currently running.
|
||||||
pub running: bool,
|
pub running: bool,
|
||||||
|
/// The container's unit is in systemd's `failed` state — it exhausted
|
||||||
|
/// its bounded restarts and gave up, as opposed to being stopped
|
||||||
|
/// deliberately. Orthogonal to `running`, like every other flag here:
|
||||||
|
/// a failed unit is not running, but a not-running unit is usually
|
||||||
|
/// just off.
|
||||||
|
#[serde(default)]
|
||||||
|
pub failed: bool,
|
||||||
/// Config commit is pending — the locked rev differs from the
|
/// Config commit is pending — the locked rev differs from the
|
||||||
/// agent's proposed/applied config (a rebuild would change it).
|
/// agent's proposed/applied config (a rebuild would change it).
|
||||||
pub needs_update: bool,
|
pub needs_update: bool,
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue