From 7b4917b2563e3d03354706fe9df3ffbc3980777e Mon Sep 17 00:00:00 2001 From: iris Date: Mon, 25 May 2026 23:35:03 +0200 Subject: [PATCH] container_view: clear live-only fields when stopped (#432) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit per mara's review on #433, move the gating from the dashboard into the host so a stopped container's stale on-disk state (rate_limited sentinel, hyperhive-needs-login, last-turn-stats row, status blob) never reaches the wire in the first place. when build_all sees is_running == false: - needs_login → false - ctx_tokens / context_window_tokens → None - rate_limited → false - status_text / status_set_at → None static / declared fields (extra_links, deployed_sha, pending_reminders, needs_update, parent) stay populated regardless of run state. extend AgentMeta (both AgentResponse + ManagerResponse) with a `running: bool` field so get_agent_meta callers can tell whether the target is up — answers the second half of #432 ("agent meta should probably show the info that it is not running as well"). read_agent_status_live wraps the existing read_agent_status with the same is_running gate so the manager/agent socket handlers don't have to know about sentinel semantics. format_agent_meta now prints a `running: yes|no` line so claude sees the run state in plain text alongside hyperhive_rev. frontend follow-up in the same commit: drop the redundant `c.running &&` guards on ctx_tokens / status_text in renderContainers — the backend now guarantees those fields are absent when the container is stopped, so the existing truthy-check is sufficient. the `■ not running` badge + icon / links fetch short-circuits stay (those are pure presentation / network-noise wins the backend can't address). --- frontend/packages/dashboard/src/app.js | 25 ++++---- hive-ag3nt/src/mcp.rs | 20 +++++-- hive-c0re/src/agent_server.rs | 9 ++- hive-c0re/src/container_view.rs | 79 ++++++++++++++++++++------ hive-c0re/src/manager_server.rs | 9 ++- hive-sh4re/src/lib.rs | 31 ++++++++-- 6 files changed, 131 insertions(+), 42 deletions(-) diff --git a/frontend/packages/dashboard/src/app.js b/frontend/packages/dashboard/src/app.js index 96275567..04468966 100644 --- a/frontend/packages/dashboard/src/app.js +++ b/frontend/packages/dashboard/src/app.js @@ -559,13 +559,13 @@ window.marked = marked; }) .catch(() => { /* graceful: agent down → no strip */ }); } - // Status / runtime badges (#432). Pending transients always win - // (start / stop / restart / rebuild is in progress). Otherwise: - // when the container is stopped, surface a single `■ stopped` - // badge and skip everything that depends on a live harness - // (alive / rate-limited / needs-login / ctx / status text); - // those badges go stale the moment the harness goes away and - // just confuse the operator if we keep showing them. + // Status / runtime badges. Pending transients always win + // (start / stop / restart / rebuild is in progress). Otherwise, + // when the container is stopped, surface a single `■ not + // running` badge; the backend has already cleared rate_limited / + // needs_login / ctx_tokens / status_text in that case (#432) so + // the rest of the chain is a no-op for stopped containers — but + // we still want SOME badge there so the row doesn't look empty. if (pending) { head.append(el('span', { class: 'pending-state' }, el('span', { class: 'spinner' }, '◐'), ' ', pending + '…')); @@ -603,7 +603,7 @@ window.marked = marked; }, `⏰ ${c.pending_reminders}`)); } - if (c.running && c.ctx_tokens != null) { + if (c.ctx_tokens != null) { const k = Math.round(c.ctx_tokens / 1000); // Thresholds track the model's real context window when the // backend supplies it; otherwise fall back to fixed constants. @@ -625,10 +625,11 @@ window.marked = marked; // ── agent status text ───────────────────────────────────────── // Self-reported status (via set_status MCP tool) — only fresh - // while the harness is up. Skip on stopped containers (#432); - // the text is from the last time the harness was running and - // just misleads now. - if (c.running && c.status_text) { + // while the harness is up. The backend already clears + // `status_text` on stopped containers (#432) so we can render + // unconditionally here: a stopped container simply has no + // `status_text` and skips this block naturally. + if (c.status_text) { const nowUnix = Math.floor(Date.now() / 1000); const ageStr = c.status_set_at != null ? ` (set ${fmtAgeSecs(nowUnix - c.status_set_at)} ago)` : ''; diff --git a/hive-ag3nt/src/mcp.rs b/hive-ag3nt/src/mcp.rs index ee0c3cdb..1f3e3474 100644 --- a/hive-ag3nt/src/mcp.rs +++ b/hive-ag3nt/src/mcp.rs @@ -52,6 +52,7 @@ pub enum SocketReply { AgentMeta { name: String, role: String, + running: bool, hyperhive_rev: Option, status_text: Option, status_set_at: Option, @@ -75,12 +76,14 @@ impl From for SocketReply { hive_sh4re::AgentResponse::AgentMeta { name, role, + running, hyperhive_rev, status_text, status_set_at, } => Self::AgentMeta { name, role, + running, hyperhive_rev, status_text, status_set_at, @@ -107,12 +110,14 @@ impl From for SocketReply { hive_sh4re::ManagerResponse::AgentMeta { name, role, + running, hyperhive_rev, status_text, status_set_at, } => Self::AgentMeta { name, role, + running, hyperhive_rev, status_text, status_set_at, @@ -270,21 +275,28 @@ fn loose_end_kind_label(kind: hive_sh4re::CancelLooseEndKind) -> &'static str { } /// Format helper for `get_agent_meta`: renders an agent's identity + -/// current status as a short human-readable block. `name`, `role`, and -/// `hyperhive_rev` are always shown; `status` only appears when one is -/// set, otherwise the line reads `status: `. +/// current status as a short human-readable block. `name`, `role`, +/// `hyperhive_rev`, and `running` are always shown; `status` only +/// appears when one is set, otherwise the line reads `status: `. +/// When `running` is false the host has already cleared `status_text` +/// (it would be stale from before the stop, #432) so the status line +/// is implicitly `` in that case — but the explicit `running: +/// no` line tells the caller WHY. #[must_use] pub fn format_agent_meta(resp: Result) -> String { match resp { Ok(SocketReply::AgentMeta { name, role, + running, hyperhive_rev, status_text, status_set_at, }) => { let rev = hyperhive_rev.as_deref().unwrap_or(""); - let mut out = format!("name: {name}\nrole: {role}\nhyperhive_rev: {rev}"); + let run = if running { "yes" } else { "no" }; + let mut out = + format!("name: {name}\nrole: {role}\nhyperhive_rev: {rev}\nrunning: {run}"); match status_text { None => out.push_str("\nstatus: "), Some(s) => { diff --git a/hive-c0re/src/agent_server.rs b/hive-c0re/src/agent_server.rs index 24facbc8..cbf763ad 100644 --- a/hive-c0re/src/agent_server.rs +++ b/hive-c0re/src/agent_server.rs @@ -246,8 +246,12 @@ async fn dispatch(req: &AgentRequest, agent: &str, coord: &Arc) -> } AgentRequest::GetAgentMeta { name } => { let target = name.as_deref().unwrap_or(agent); - let (status_text, status_set_at) = - crate::container_view::read_agent_status(target); + // #432: gate status on the target's running state so a + // stopped container's stale on-disk status doesn't leak + // through. Also surface `running` itself so callers can + // tell (e.g. "iris is down" vs "iris has no status set"). + let (status_text, status_set_at, running) = + crate::container_view::read_agent_status_live(target).await; let role = if target == hive_sh4re::MANAGER_AGENT { "manager" } else { @@ -257,6 +261,7 @@ async fn dispatch(req: &AgentRequest, agent: &str, coord: &Arc) -> AgentResponse::AgentMeta { name: target.to_owned(), role, + running, hyperhive_rev: crate::auto_update::current_flake_rev(&coord.hyperhive_flake), status_text, status_set_at, diff --git a/hive-c0re/src/container_view.rs b/hive-c0re/src/container_view.rs index f0e56e91..68c0458f 100644 --- a/hive-c0re/src/container_view.rs +++ b/hive-c0re/src/container_view.rs @@ -117,14 +117,6 @@ pub async fn build_all(coord: &Coordinator) -> Vec { }; let deployed_full = locked.get(&format!("agent-{logical}")).map(std::string::String::as_str); let needs_update = crate::auto_update::agent_config_pending(&logical, deployed_full); - // needs_login fires when EITHER the claude session dir is - // missing (boot-time / fresh container) OR the harness wrote - // the auth-failed sentinel because a turn hit 401 (#419). The - // manager has its own session lifecycle and never participates - // in needs_login. - let needs_login = !is_manager - && (!claude_has_session(&Coordinator::agent_claude_dir(&logical)) - || auth_failed_sentinel(&logical)); let deployed_sha = deployed_full.map(|s| s[..s.len().min(12)].to_owned()); // Recipient name the broker uses for this agent — sub-agents // are addressed by logical name, the manager by the @@ -140,18 +132,41 @@ pub async fn build_all(coord: &Coordinator) -> Vec { .broker .count_pending_reminders_for(reminder_recipient) .unwrap_or(0); - let last_turn = read_last_turn(&logical); - let ctx_tokens = last_turn.as_ref().map(|(toks, _)| *toks); - let context_window_tokens = last_turn - .as_ref() - .and_then(|(_, model)| resolve_ctx_window(model, &coord.context_window_tokens)); - let rate_limited = is_rate_limited(&logical); let extra_links = read_dashboard_links(&logical); - let (status_text, status_set_at) = read_status(&logical); let parent = topology.get(&logical).cloned().flatten(); + let running = lifecycle::is_running(&logical).await; + // Live-only fields (#432) — only meaningful while the harness + // is up. When the container is stopped, sentinel files + + // turn-stats rows + the on-disk status blob are all stale + // snapshots from before the stop, so we clear them here + // rather than letting the dashboard / `get_agent_meta` surface + // misleading values. Static / declared fields (extra_links, + // deployed_sha, pending_reminders, needs_update, parent) stay + // populated regardless of run state. + let (needs_login, ctx_tokens, context_window_tokens, rate_limited, status_text, status_set_at) = + if running { + // needs_login fires when EITHER the claude session dir is + // missing (boot-time / fresh container) OR the harness wrote + // the auth-failed sentinel because a turn hit 401 (#419). The + // manager has its own session lifecycle and never participates + // in needs_login. + let needs_login = !is_manager + && (!claude_has_session(&Coordinator::agent_claude_dir(&logical)) + || auth_failed_sentinel(&logical)); + let last_turn = read_last_turn(&logical); + let ctx_tokens = last_turn.as_ref().map(|(toks, _)| *toks); + let context_window_tokens = last_turn + .as_ref() + .and_then(|(_, model)| resolve_ctx_window(model, &coord.context_window_tokens)); + let rate_limited = is_rate_limited(&logical); + let (status_text, status_set_at) = read_status(&logical); + (needs_login, ctx_tokens, context_window_tokens, rate_limited, status_text, status_set_at) + } else { + (false, None, None, false, None, None) + }; out.push(ContainerView { port: lifecycle::agent_web_port(&logical), - running: lifecycle::is_running(&logical).await, + running, container: c.clone(), name: logical, is_manager, @@ -219,6 +234,10 @@ fn auth_failed_sentinel(name: &str) -> bool { /// Read the agent's free-text status and the Unix timestamp when it was last set /// (derived from the file's mtime). Returns `(None, None)` when the file is absent /// or empty. `pub` so `agent_server` and `manager_server` can populate `AgentMeta`. +/// +/// NB: callers building `AgentMeta` for a *stopped* container should +/// clear the result — the on-disk status is a stale snapshot from +/// before the stop (#432). Use `read_agent_status_live` for that. pub fn read_agent_status(name: &str) -> (Option, Option) { let path = Coordinator::agent_notes_dir(name).join("hyperhive-status"); let meta = std::fs::metadata(&path).ok(); @@ -237,6 +256,34 @@ fn read_status(name: &str) -> (Option, Option) { read_agent_status(name) } +/// Wraps `read_agent_status` with the same "stopped containers have +/// stale state" gate `build_all` uses (#432). Returns +/// `(None, None, false)` when the container isn't running so callers +/// don't have to know about the sentinel rules — they just hand back +/// what we give them. +/// +/// Returned tuple is `(status_text, status_set_at, running)`. The +/// `name` argument is the broker-side recipient — `MANAGER_AGENT` for +/// the manager, the logical agent name otherwise — so callers can +/// reuse the same string they used to look the agent up. +pub async fn read_agent_status_live(name: &str) -> (Option, Option, bool) { + // The lifecycle helper wants the on-disk name (`hm1nd` for the + // manager, the bare logical name for sub-agents) and internally + // adds the `h-` prefix. Map the broker-side `MANAGER_AGENT` + // sentinel back to the lifecycle name here so callers don't have + // to bother. + let lifecycle_name = if name == hive_sh4re::MANAGER_AGENT { + lifecycle::MANAGER_NAME + } else { + name + }; + if !lifecycle::is_running(lifecycle_name).await { + return (None, None, false); + } + let (text, set_at) = read_agent_status(name); + (text, set_at, true) +} + /// Read the agent's most recent completed turn from its turn-stats /// `SQLite`: the context-window size (prompt tokens) and the model name. /// Returns `None` when the file is absent or has no rows. Best-effort diff --git a/hive-c0re/src/manager_server.rs b/hive-c0re/src/manager_server.rs index fb0ef127..a89ae68d 100644 --- a/hive-c0re/src/manager_server.rs +++ b/hive-c0re/src/manager_server.rs @@ -494,12 +494,17 @@ async fn dispatch(req: &ManagerRequest, coord: &Arc) -> ManagerResp } ManagerRequest::GetAgentMeta { name } => { let target = name.as_deref().unwrap_or(MANAGER_AGENT); - let (status_text, status_set_at) = - crate::container_view::read_agent_status(target); + // #432: gate status on the target's running state so a + // stopped container's stale on-disk status doesn't leak + // through. Also surface `running` itself so callers can + // tell (e.g. "iris is down" vs "iris has no status set"). + let (status_text, status_set_at, running) = + crate::container_view::read_agent_status_live(target).await; let role = if target == MANAGER_AGENT { "manager" } else { "agent" }.to_owned(); ManagerResponse::AgentMeta { name: target.to_owned(), role, + running, hyperhive_rev: crate::auto_update::current_flake_rev(&coord.hyperhive_flake), status_text, status_set_at, diff --git a/hive-sh4re/src/lib.rs b/hive-sh4re/src/lib.rs index 55c9b031..32170968 100644 --- a/hive-sh4re/src/lib.rs +++ b/hive-sh4re/src/lib.rs @@ -517,14 +517,20 @@ pub enum AgentResponse { /// `GetAgentMeta` result: identity + status metadata for an agent. /// `role` is `"agent"` for sub-agents and `"manager"` for the /// manager. `hyperhive_rev` is `None` only when the configured - /// flake URL has no canonical path. `status_text` is the last value - /// written via `SetStatus`, or `None` when none has been set or the - /// agent name is unknown. `status_set_at` is a Unix timestamp - /// (seconds since epoch) of when the status was last written; - /// `None` when no status is set. + /// flake URL has no canonical path. `running` reflects whether the + /// target's container is currently up (#432); when it's false, + /// `status_text` / `status_set_at` are intentionally cleared by the + /// host because the on-disk values are stale snapshots from before + /// the stop. `status_text` is the last value written via + /// `SetStatus`, or `None` when none has been set or the agent name + /// is unknown. `status_set_at` is a Unix timestamp (seconds since + /// epoch) of when the status was last written; `None` when no + /// status is set. AgentMeta { name: String, role: String, + #[serde(default = "default_true")] + running: bool, #[serde(default, skip_serializing_if = "Option::is_none")] hyperhive_rev: Option, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -534,6 +540,14 @@ pub enum AgentResponse { }, } +/// Serde default for the `running` field on legacy wire payloads that +/// predate #432 — older harnesses never serialised it, and `true` +/// matches the historical assumption (the host only knew how to ask +/// about live containers). +fn default_true() -> bool { + true +} + // ----------------------------------------------------------------------------- // Manager socket — /run/hyperhive/manager/mcp.sock on the host, bind-mounted // into the manager container at /run/hive/mcp.sock. @@ -944,10 +958,15 @@ pub enum ManagerResponse { ReminderRollup(ReminderStats), /// Mirror of `AgentResponse::AgentMeta` on the manager surface. /// `role` is `"manager"` for the manager and `"agent"` for any - /// sub-agent looked up by name. + /// sub-agent looked up by name. `running` is false when the + /// target's container is stopped (#432) — in that case + /// `status_text` / `status_set_at` are cleared by the host so + /// stale pre-stop values don't leak through. AgentMeta { name: String, role: String, + #[serde(default = "default_true")] + running: bool, #[serde(default, skip_serializing_if = "Option::is_none")] hyperhive_rev: Option, #[serde(default, skip_serializing_if = "Option::is_none")]