feat(#2290): split ensure_runtime — dirs to lifecycle, listeners to mcp_sockets supervisor

- lifecycle::ensure_agent_runtime_dir(name): pure filesystem op, no
  Coordinator dep. Creates /run/hyperhive/agents/<name> without touching
  the MCP listener map.

- workers/mcp_sockets::spawn_poll(coord): 10 s reconcile loop (same shape
  as agent_sockets::spawn_poll). Converges 'agent running => MCP listener
  bound'. First tick is immediate so hive-c0re restarts re-register all
  running agents without waiting a full interval. Fixes the dead-listener-
  after-daemon-restart gap.

- All ensure_runtime() call sites updated:
  - Prebuild/Swap/WriteDropin: Coordinator::agent_dir() (pure, no IO)
  - Reconcile-Start: ensure_agent_runtime_dir + agent_dir (dir may be
    missing after reboot; listener deferred to supervisor)
  - run_create / handle_spawn: ensure_agent_runtime_dir + register_agent
    (eager on first spawn so socket ready before harness first turn)
  - apply_commit / merge_config_pr: ensure_agent_runtime_dir + agent_dir
  - Manager (auto_update): ensure_agent_runtime_dir + agent_dir
    (manager has no MCP listener; socket_server::start_manager owns it)

- ensure_runtime() retained in Coordinator with updated doc pointing at
  the preferred split form. No callers remain outside tests.
This commit is contained in:
atlas 2026-07-08 23:27:48 +02:00
commit 3d919b596f
10 changed files with 147 additions and 30 deletions

View file

@ -159,7 +159,11 @@ pub async fn run_approval_apply_commit(
approval_id: i64,
) -> Result<()> {
let approval = fetch_approval_for_worker(coord, approval_id, ApprovalKind::ApplyCommit)?;
let agent_dir = coord.ensure_runtime(&approval.agent)?;
// Create the bind-mount source dir (first-spawn may not have it yet).
// MCP listener registration is deferred to mcp_sockets::spawn_poll
// which fires within 10 s of the container coming up.
lifecycle::ensure_agent_runtime_dir(&approval.agent)?;
let agent_dir = Coordinator::agent_dir(&approval.agent);
let applied_dir = Coordinator::agent_applied_dir(&approval.agent);
coord.set_queue_step(queue_entry_id, "apply commit");
let (result, terminal_tag, is_first_spawn) =
@ -192,7 +196,8 @@ pub async fn run_approval_merge_config_pr(
approval_id: i64,
) -> Result<()> {
let approval = fetch_approval_for_worker(coord, approval_id, ApprovalKind::MergeConfigPr)?;
let agent_dir = coord.ensure_runtime(&approval.agent)?;
lifecycle::ensure_agent_runtime_dir(&approval.agent)?;
let agent_dir = Coordinator::agent_dir(&approval.agent);
let applied_dir = Coordinator::agent_applied_dir(&approval.agent);
coord.set_queue_step(queue_entry_id, "merge config pr");
let (result, terminal_tag) =

View file

@ -545,14 +545,11 @@ impl Coordinator {
}
}
/// Assemble the per-agent filesystem paths for `name`. The caller
/// must supply `agent_dir` (from `ensure_runtime`) since that
/// creates the tmpfs entry on first call. All other paths are
/// derived statically from `name`.
///
/// ```ignore
/// let paths = Coordinator::agent_paths(name, coord.ensure_runtime(name)?);
/// ```
/// Assemble the per-agent filesystem paths for `name`. `agent_dir`
/// is the runtime directory (`/run/hyperhive/agents/<name>`), obtained
/// from `Coordinator::agent_dir(name)` (pure path) or from
/// `lifecycle::ensure_agent_runtime_dir` + `agent_dir` when the dir
/// must be created. All other paths are derived statically from `name`.
#[must_use]
pub fn agent_paths(name: &str, agent_dir: PathBuf) -> AgentPaths {
AgentPaths {
@ -1453,11 +1450,17 @@ impl Coordinator {
Self::agent_dir(name).join("mcp.sock")
}
/// Ensure a runtime dir + (for sub-agents) per-agent socket exists. For
/// the manager, `socket_server::start_manager` owns the socket — just return
/// the dir. For sub-agents this is `register_agent` (creates a fresh
/// listener bound to `socket_path(name)`). Source directory of the
/// `/run/hive/mcp.sock` bind that ends up in `set_nspawn_flags`.
/// Ensure a runtime dir + (for sub-agents) per-agent socket exists.
///
/// **Prefer the split form:**
/// - dir creation → `lifecycle::ensure_agent_runtime_dir(name)`
/// - listener registration → `register_agent(name)` (eagerly, on first
/// spawn) or leave it to `mcp_sockets::spawn_poll` (reconcile within
/// 10 s, safe for the start + rebuild paths where the container
/// takes longer than that to boot).
///
/// This method is kept as a convenience shim for any remaining callers
/// that need the combined semantics in one call.
pub fn ensure_runtime(self: &Arc<Self>, name: &str) -> Result<PathBuf> {
if name == crate::lifecycle::MANAGER_NAME {
let dir = Self::agent_dir(name);

View file

@ -97,9 +97,10 @@ async fn run_prebuild(
relock: bool,
) -> Result<NodeOutput> {
let name = &claim.agent;
let agent_dir = coord
.ensure_runtime(name)
.with_context(|| format!("ensure_runtime {name}"))?;
// Prebuild runs while the agent is still up — the runtime dir and
// MCP listener already exist. Use the pure path accessor; no need
// to re-register the listener (the mcp_sockets supervisor owns that).
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
crate::lifecycle::prepare_rebuild_dirs(name, &paths).await?;
@ -131,7 +132,9 @@ async fn run_prebuild(
/// `Reconcile` runs after this node terminal ok *or* fail.
async fn run_swap(coord: &Arc<Coordinator>, claim: &Claim, ctx: &Ctx<'_>) -> Result<NodeOutput> {
let name = &claim.agent;
let agent_dir = coord.ensure_runtime(name)?;
// Swap runs on an already-existing (stopped) container — runtime dir
// and listener were created earlier. Pure path accessor suffices.
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
let result =
@ -178,7 +181,12 @@ async fn run_swap(coord: &Arc<Coordinator>, claim: &Claim, ctx: &Ctx<'_>) -> Res
/// build+create — no prebuild needed).
async fn run_create(coord: &Arc<Coordinator>, claim: &Claim, ctx: &Ctx<'_>) -> Result<NodeOutput> {
let name = &claim.agent;
let agent_dir = coord.ensure_runtime(name)?;
// First-spawn: create the bind-mount source dir (tmpfs — empty after
// reboot). Register the MCP listener eagerly so it's ready when the
// tail Reconcile starts the container and the harness connects.
crate::lifecycle::ensure_agent_runtime_dir(name)?;
coord.register_agent(name)?;
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
ctx.step("nixos-container create");
@ -254,7 +262,13 @@ async fn run_reconcile(
// Prebuild/Swap nodes; the bare-Reconcile templates (boot
// reconcile, plain start/restart) otherwise start with
// nothing under /run and fail.
let agent_dir = coord.ensure_runtime(name)?;
//
// Dir creation is the pure-filesystem part (no Coordinator
// dep). The MCP listener is reconciled by mcp_sockets::spawn_poll
// whose first tick fires immediately on daemon start — the
// container boot takes longer than the 10 s interval anyway.
crate::lifecycle::ensure_agent_runtime_dir(name)?;
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
crate::lifecycle::write_dropins(name, &hive, &paths).await?;
@ -336,7 +350,10 @@ async fn run_drain(coord: &Arc<Coordinator>, claim: &Claim, ctx: &Ctx<'_>) -> Re
/// `set_nspawn_flags` + `set_resource_limits` + daemon-reload.
async fn run_write_dropin(coord: &Arc<Coordinator>, claim: &Claim) -> Result<NodeOutput> {
let name = &claim.agent;
let agent_dir = coord.ensure_runtime(name)?;
// write_dropins only needs the path value to build AgentPaths; the
// dir doesn't need to exist at this point (created by ensure_runtime
// on the upstream Prebuild/Start node).
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
crate::lifecycle::write_dropins(name, &hive, &paths).await?;

View file

@ -50,6 +50,6 @@ pub use stores::{
approvals, audit_log, broker, build_logs, db, operator_questions, power, scheduled_prompts,
};
pub use workers::{
agent_sockets, auto_update, crash_watch, knowledge, reminder_scheduler,
agent_sockets, auto_update, crash_watch, knowledge, mcp_sockets, reminder_scheduler,
scheduled_prompts_worker,
};

View file

@ -749,6 +749,23 @@ pub async fn sync_tmpfiles() {
}
}
/// Ensure the per-agent runtime directory `/run/hyperhive/agents/<name>`
/// exists. The directory is also written by `SyncAgentTmpfiles` (run at
/// boot + spawn/destroy), but explicit creation in start/spawn paths guards
/// against races where hive-c0re starts a container before tmpfiles.d has
/// applied the new entry.
///
/// Pure filesystem op — no `Coordinator` dependency — so callers that only
/// need the dir do not have to hold an `Arc<Coordinator>`.
///
/// # Errors
/// Returns an error if `create_dir_all` fails.
pub fn ensure_agent_runtime_dir(name: &str) -> Result<()> {
let dir = std::path::PathBuf::from(format!("/run/hyperhive/agents/{name}"));
std::fs::create_dir_all(&dir)
.with_context(|| format!("create agent runtime dir {}", dir.display()))
}
/// Build the per-line callback for `create_container_streaming` /
/// `update_container_streaming`. Both ops share identical dispatch logic
/// (stdout → info + `append_stdout`, stderr → warn + `append_stderr`); this

View file

@ -13,8 +13,8 @@ use hive_sh4re::{HostRequest, HostResponse};
use hive_c0re::coordinator::{Coordinator, HiveEnv, ServeConfig};
use hive_c0re::{
agent_sockets, auto_update, broker, client, crash_watch, dashboard, dashboard_events, forge,
job_queue, knowledge, matrix, migrate, reminder_scheduler, scheduled_prompts_worker, server,
socket_server,
job_queue, knowledge, matrix, migrate, mcp_sockets, reminder_scheduler,
scheduled_prompts_worker, server, socket_server,
};
#[derive(Parser)]
@ -420,6 +420,13 @@ async fn cmd_serve(
// is one stat per agent per tick.
// See `docs/gateway.md::Per-agent unix-socket upstream`.
agent_sockets::spawn_poll();
// MCP socket listener reconcile loop: every 10s re-registers any
// running agent that lost its host-side MCP listener (e.g. after a
// hive-c0re restart cleared /run/hyperhive/agents/). First tick fires
// immediately so restarts re-register all running agents without delay.
// Decouples listener registration from the start path — start only needs
// lifecycle::ensure_agent_runtime_dir; the supervisor converges the rest.
mcp_sockets::spawn_poll(coord.clone());
// Reminder scheduler: drains due reminders + handles
// file_path payload persistence. See reminder_scheduler.rs.
reminder_scheduler::spawn(coord.clone());

View file

@ -202,7 +202,12 @@ async fn dispatch(req: &HostRequest, coord: Arc<Coordinator>) -> HostResponse {
/// registration and notifying the manager on failure.
async fn handle_spawn(coord: &Arc<Coordinator>, name: &str) -> Result<HostResponse> {
tracing::info!(%name, "spawn");
let agent_dir = coord.ensure_runtime(name)?;
// Create the bind-mount source dir (pure filesystem, no Coordinator dep).
// MCP listener registration happens eagerly here (not deferred to the
// supervisor) so the socket is ready before the harness's first turn.
lifecycle::ensure_agent_runtime_dir(name)?;
coord.register_agent(name)?;
let agent_dir = Coordinator::agent_dir(name);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(name, agent_dir);
match lifecycle::spawn(name, &hive, &paths).await {

View file

@ -153,7 +153,10 @@ pub async fn ensure_root_agent(coord: &Arc<Coordinator>) -> Result<()> {
return Ok(());
}
tracing::info!("manager container missing — spawning");
let runtime = coord.ensure_runtime(MANAGER_NAME)?;
lifecycle::ensure_agent_runtime_dir(MANAGER_NAME)?;
// Manager has no MCP listener (socket_server::start_manager owns its
// socket); just need the dir + path value.
let runtime = Coordinator::agent_dir(MANAGER_NAME);
let hive = coord.hive_env();
let paths = Coordinator::agent_paths(MANAGER_NAME, runtime);
lifecycle::spawn(MANAGER_NAME, &hive, &paths).await?;

View file

@ -0,0 +1,59 @@
//! MCP socket listener reconcile loop.
//!
//! Periodically checks that every running agent container has a bound
//! MCP listener registered in the `Coordinator`. Any agent that is running
//! but whose listener has gone (e.g. after a hive-c0re restart that cleared
//! `/run/hyperhive/agents/`) gets re-registered automatically.
//!
//! Same shape as `agent_sockets::spawn_poll` — a simple 10 s tick loop that
//! converges "agent container running ⇒ MCP listener bound". The self-healing
//! guarantee means the start path no longer needs to call `register_agent`
//! directly (though spawn still does for eagerness); callers only need
//! `lifecycle::ensure_agent_runtime_dir` to create the bind-mount source.
use std::sync::Arc;
use crate::coordinator::Coordinator;
/// Spawn the MCP socket listener reconcile loop.
///
/// Every 10 s the loop lists running agent containers and calls
/// `register_agent` for any that lack a bound listener. The first tick fires
/// immediately so hive-c0re restarts re-register all running agents without
/// waiting a full interval.
pub fn spawn_poll(coord: Arc<Coordinator>) {
tokio::spawn(async move {
let mut interval = tokio::time::interval(std::time::Duration::from_secs(10));
loop {
interval.tick().await;
reconcile_once(&coord).await;
}
});
}
async fn reconcile_once(coord: &Arc<Coordinator>) {
let running = match crate::lifecycle::list().await {
Ok(names) => names,
Err(e) => {
tracing::debug!(error = ?e, "mcp_sockets poll: failed to list agents");
return;
}
};
let registered = coord.list_agents();
for container in running {
let Some(name) = container.strip_prefix(crate::lifecycle::AGENT_PREFIX) else {
continue;
};
if !registered.contains(&name.to_owned()) {
if let Err(e) = coord.register_agent(name) {
tracing::warn!(
agent = %name,
error = ?e,
"mcp_sockets poll: register_agent failed"
);
} else {
tracing::debug!(agent = %name, "mcp_sockets poll: registered missing listener");
}
}
}
}

View file

@ -1,12 +1,13 @@
//! Background tasks and periodic sweeps: crash/login watcher, the
//! reminder and scheduled-prompt delivery loops, boot-time auto-update
//! reconcile, the agent-sockets.json writer loop, and knowledge-repo
//! sync. Each submodule is re-exported at the crate root, so
//! `crate::crash_watch::…` etc. keep working unchanged.
//! reconcile, the agent-sockets.json writer loop, the MCP socket listener
//! reconcile loop, and knowledge-repo sync. Each submodule is re-exported
//! at the crate root, so `crate::crash_watch::…` etc. keep working unchanged.
pub mod agent_sockets;
pub mod auto_update;
pub mod crash_watch;
pub mod knowledge;
pub mod mcp_sockets;
pub mod reminder_scheduler;
pub mod scheduled_prompts_worker;