workers: skip a cycle instead of spawning/alerting on an unreadable container list

crash_watch's 10s poll and auto_update's ensure_root_agent both read
lifecycle::list().await.unwrap_or_default(), which turned a failed read
into 'zero containers'. In crash_watch that made every previously-running
agent look like it crashed simultaneously (prev.difference(current) over
an empty current), and left prev empty for the next cycle too, so a
second wave of false 'agent logged in' / 'agent needs login' events fired
against the next successful read. In ensure_root_agent it read as
'manager container missing' and called lifecycle::spawn on a manager
that might already exist.

Both sites now treat a list error as its own outcome: log it at warn and
skip the cycle's decision entirely. crash_watch leaves prev exactly as
the last good read produced it. ensure_root_agent attempts no spawn.

Factors each site's decision into a pure helper (plan_cycle /
plan_root_agent) matching the check_not_live / confirm_gone_after_failed_destroy
pattern, with unit tests for the error case, a control for the readable
case, and (for crash_watch) an invert-proof run locally against the old
unwrap_or_default logic before reverting.
This commit is contained in:
atlas 2026-09-24 12:50:07 +02:00 • committed by mara
commit f62ec349d5
2 changed files with 172 additions and 43 deletions

View file

@ -81,6 +81,36 @@ fn ruthless() -> bool {
}
}
/// What `ensure_root_agent` does with a `lifecycle::list()` result, before
/// it even gets to check whether the manager is present.
#[derive(Debug, PartialEq, Eq)]
enum RootAgentPlan {
/// The manager is in this readable list.
Present,
/// The manager is absent from this readable list: spawn it.
Absent,
/// The list could not be read. That is not "manager absent" — it
/// proves nothing either way, so the auto-spawn decision is skipped
/// for this attempt rather than risking a spawn over a manager that's
/// actually there.
Unreadable,
}
/// Turn a `lifecycle::list()` result into `ensure_root_agent`'s plan.
fn plan_root_agent(list_result: anyhow::Result<Vec<String>>) -> RootAgentPlan {
match list_result {
Ok(names)
if names
.iter()
.any(|c| c.strip_prefix(AGENT_PREFIX) == Some(MANAGER_NAME)) =>
{
RootAgentPlan::Present
}
Ok(_) => RootAgentPlan::Absent,
Err(_) => RootAgentPlan::Unreadable,
}
}
/// Auto-create the manager container on startup if it isn't already there.
/// hive-c0re manages the manager end-to-end: operators no longer declare
/// `containers.h-ruth` in their host NixOS config. Bypasses the approval
@ -98,12 +128,25 @@ pub async fn ensure_root_agent(coord: &Arc<Coordinator>) -> Result<()> {
// container already exists this is the only run that can still seed the
// grant, and it has to land before her next rebuild bakes the binds.
seed_manager_capabilities();
let existing = lifecycle::list().await.unwrap_or_default();
let list_result = lifecycle::list().await;
if let Err(e) = &list_result {
tracing::warn!(
error = ?e,
"manager container list unreadable — skipping auto-spawn check for this attempt"
);
}
let plan = plan_root_agent(list_result);
let current_rev = current_flake_rev(&coord.hyperhive_flake);
if existing
.iter()
.any(|c| c.strip_prefix(AGENT_PREFIX) == Some(MANAGER_NAME))
{
if plan == RootAgentPlan::Unreadable {
// An unreadable list is not "manager absent" — spawning on it would
// both waste `provision_container`'s work and hit `nixos-container
// create`'s "already exists" failure if the manager is actually
// there. Skip the decision this attempt; there is no periodic
// retry for this boot-time call, so the next chance is the next
// `hive-c0re` restart.
return Ok(());
}
if plan == RootAgentPlan::Present {
// Container exists already. If it predates the unified lifecycle
// (no applied flake on disk) we must rebuild — otherwise it's
// running whatever the host-declarative config was at create
@ -503,9 +546,38 @@ fn submit_boot_tree(
#[cfg(test)]
mod tests {
use super::{BootAction, boot_action, manager_seed_caps, should_seed_manager_caps};
use super::{
BootAction, MANAGER_NAME, RootAgentPlan, boot_action, manager_seed_caps, plan_root_agent,
should_seed_manager_caps,
};
use crate::lifecycle::AGENT_PREFIX;
use crate::power::Wanted;
// -----------------------------------------------------------------------
// `ensure_root_agent`'s spawn-or-not decision. An unreadable list must
// never be read as "manager absent" — see `plan_root_agent`'s doc.
/// The regression this fix exists for: a read failure must not spawn.
#[test]
fn unreadable_list_skips_without_spawning() {
let result: anyhow::Result<Vec<String>> =
Err(anyhow::anyhow!("connect to hive-priv socket"));
assert_eq!(plan_root_agent(result), RootAgentPlan::Unreadable);
}
/// Control: a genuinely empty, readable list still plans a spawn.
#[test]
fn readable_list_missing_manager_spawns() {
let result: anyhow::Result<Vec<String>> = Ok(Vec::new());
assert_eq!(plan_root_agent(result), RootAgentPlan::Absent);
}
#[test]
fn readable_list_with_manager_present_is_a_noop() {
let result: anyhow::Result<Vec<String>> = Ok(vec![format!("{AGENT_PREFIX}{MANAGER_NAME}")]);
assert_eq!(plan_root_agent(result), RootAgentPlan::Present);
}
// -----------------------------------------------------------------------
// Root-agent capability seed. `capabilities_path()` resolves under
// `paths::meta_root()`, which is hardcoded to `/var/lib/hyperhive` with no