use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex}; use std::time::Duration; use hive_ag3nt::web_ui::TurnLock; use anyhow::Result; use clap::{Parser, Subcommand}; use hive_ag3nt::events::{Bus, LiveEvent, TurnState}; use hive_ag3nt::login::{self, LoginState}; use hive_ag3nt::turn_stats::TurnStats; use hive_ag3nt::{DEFAULT_SOCKET, DEFAULT_WEB_PORT, client, mcp, plugins, serve_common, turn, web_ui}; use hive_sh4re::{AgentRequest, AgentResponse}; #[derive(Parser)] #[command(name = "hive-ag3nt", about = "hyperhive sub-agent harness")] struct Cli { /// Path to the per-agent MCP socket (bind-mounted from the host). #[arg(long, global = true, default_value = DEFAULT_SOCKET)] socket: PathBuf, #[command(subcommand)] cmd: Cmd, } #[derive(Subcommand)] enum Cmd { /// Run the long-lived harness loop. Polls inbox; replies via `claude --print` /// when available, falling back to a simple echo otherwise. Serve { /// Inbox poll interval in milliseconds. #[arg(long, default_value_t = 1000)] poll_ms: u64, }, /// Run the agent's MCP server on stdio. Spawned by `claude` via /// `--mcp-config`; tools dispatch through `/run/hive/mcp.sock` back into /// the hyperhive broker. Mcp, /// Inject a wake-up event into this agent's inbox so the next turn /// fires with the given body. Intended for extra MCP servers / /// helpers running inside the container (matrix bridge, scraper, /// webhook listener) that need to nudge claude on external events. /// `from` is the sender label that appears in the wake prompt /// (claude sees "from: matrix" etc.). Wake { #[arg(long)] from: String, /// Body of the wake message. Pass `-` to read from stdin. #[arg(long)] body: String, }, } #[tokio::main] async fn main() -> Result<()> { tracing_subscriber::fmt() .with_env_filter( tracing_subscriber::EnvFilter::try_from_default_env() .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info")), ) .init(); let cli = Cli::parse(); match cli.cmd { Cmd::Serve { poll_ms } => { let port = std::env::var("HIVE_PORT") .ok() .and_then(|s| s.parse::().ok()) .unwrap_or(DEFAULT_WEB_PORT); let label = std::env::var("HIVE_LABEL").unwrap_or_else(|_| "hive-ag3nt".into()); let claude_dir = login::default_dir(); let initial = LoginState::from_dir(&claude_dir); tracing::info!(state = ?initial, claude_dir = %claude_dir.display(), "harness boot"); let login_state = Arc::new(Mutex::new(initial)); let bus = Bus::new(); let stats = TurnStats::open_default(); if let Some(s) = &stats { let (ctx, cost) = s.last_usage(); if ctx.is_some() || cost.is_some() { bus.seed_usage(ctx, cost); } } let files = turn::TurnFiles::prepare(&cli.socket, &label, mcp::Flavor::Agent).await?; let turn_lock: TurnLock = Arc::new(tokio::sync::Mutex::new(())); plugins::install_configured(&cli.socket, Some("manager")).await; tokio::spawn(hive_ag3nt::forge_notify::run(cli.socket.clone(), false)); tokio::spawn(web_ui::serve( label.clone(), port, login_state.clone(), bus.clone(), cli.socket.clone(), files.clone(), turn_lock.clone(), )); match initial { LoginState::Online => { serve( &cli.socket, Duration::from_millis(poll_ms), login_state, claude_dir, bus, stats, &files, turn_lock, &label, ) .await } LoginState::NeedsLogin => { // Partial-run mode: keep the harness alive (so the web UI // stays bound) but don't drive the turn loop. Poll the // claude dir; once a session lands we enter `serve`. turn::wait_for_login(&claude_dir, login_state.clone(), &bus, poll_ms).await; serve( &cli.socket, Duration::from_millis(poll_ms), login_state, claude_dir, bus, stats, &files, turn_lock, &label, ) .await } } } Cmd::Mcp => mcp::serve_agent_stdio(cli.socket).await, Cmd::Wake { from, body } => { // Read body from stdin if caller passed `-`. Same convention // many CLI tools use; keeps multi-line / shell-quoting // friction out of the body content. let body = if body == "-" { let mut buf = String::new(); std::io::Read::read_to_string(&mut std::io::stdin(), &mut buf)?; buf } else { body }; let resp: AgentResponse = client::request(&cli.socket, &AgentRequest::Wake { from, body }).await?; match resp { AgentResponse::Ok => Ok(()), AgentResponse::Err { message } => anyhow::bail!("wake: {message}"), other => anyhow::bail!("wake: unexpected response {other:?}"), } } } } #[allow(clippy::too_many_arguments)] async fn serve( socket: &Path, interval: Duration, login_state: Arc>, claude_dir: std::path::PathBuf, bus: Bus, stats: Option, files: &turn::TurnFiles, turn_lock: TurnLock, label: &str, ) -> Result<()> { tracing::info!(socket = %socket.display(), "hive-ag3nt serve"); requeue_inflight(socket).await; loop { let recv: Result = // Explicit long-poll: park until a message arrives (180s cap). // `max: None` (= 1) — one turn per wake; claude calls // recv(max: N) in-turn to drain bursts. client::request( socket, &AgentRequest::Recv { wait_seconds: Some(180), max: None, }, ) .await; match recv { Ok(AgentResponse::Messages { messages }) if !messages.is_empty() => { let first = messages.into_iter().next().expect("checked non-empty"); let auth_failed = handle_agent_turn(socket, &bus, stats.as_ref(), files, &turn_lock, label, first) .await; if auth_failed { // Park: flip LoginState + wait for the operator's // re-auth to repopulate claude_dir. wait_for_login // emits `online` on resume, which clears the // needs_login sentinel. *login_state.lock().unwrap() = LoginState::NeedsLogin; turn::wait_for_login( &claude_dir, login_state.clone(), &bus, u64::try_from(interval.as_millis()).unwrap_or(2000), ) .await; } } Ok(AgentResponse::Messages { .. }) => { // Idle: empty list = nothing pending. Brief sleep // before next poll so a stretch of empty long-poll // returns doesn't tight-loop. tokio::time::sleep(interval).await; } Ok( AgentResponse::Ok | AgentResponse::Status { .. } | AgentResponse::Recent { .. } | AgentResponse::QuestionQueued { .. } | AgentResponse::LooseEnds { .. } | AgentResponse::PendingRemindersCount { .. } | AgentResponse::ReminderRollup { .. } | AgentResponse::AgentMeta { .. }, ) => { tracing::warn!("recv produced unexpected response kind"); } Ok(AgentResponse::Err { message }) => { tracing::warn!(%message, "recv error"); } Err(e) => { tracing::warn!(error = ?e, "recv failed; retrying"); } } } } /// Drive one turn for a received agent-inbox message. Returns `true` /// when the turn ended with `AuthFailed` so the caller knows to park /// in `wait_for_login`. async fn handle_agent_turn( socket: &Path, bus: &Bus, stats: Option<&TurnStats>, files: &turn::TurnFiles, turn_lock: &TurnLock, label: &str, first: hive_sh4re::DeliveredMessage, ) -> bool { let from = first.from; let body = first.body; let redelivered = first.redelivered; tracing::info!(%from, %body, %redelivered, "inbox"); let unread = inbox_unread(socket).await; bus.emit(LiveEvent::TurnStart { from: from.clone(), body: body.clone(), unread }); bus.set_state(TurnState::Thinking); let started_at = serve_common::now_unix(); let started_instant = std::time::Instant::now(); let model_at_start = bus.model(); let prompt = serve_common::format_wake_prompt(&from, &body, unread, redelivered); let outcome = { let _guard = turn_lock.lock().await; turn::drive_turn(&prompt, files, bus).await }; turn::emit_turn_end(bus, &outcome); bus.set_state(TurnState::Idle); // Ack only on a clean turn-end. `Failed` leaves every message popped // during the turn in the unacked list; next harness boot requeues them. if matches!(outcome, turn::TurnOutcome::Ok | turn::TurnOutcome::Compacted) { ack_turn(socket).await; } if matches!(outcome, turn::TurnOutcome::RateLimited) { let secs = turn::rate_limit_sleep_secs(); bus.emit_status("rate_limited"); bus.emit(LiveEvent::Note { text: format!("API rate-limited — sleeping {secs}s before retry"), }); tracing::warn!(sleep_secs = secs, "rate-limited; parking"); tokio::time::sleep(Duration::from_secs(secs)).await; requeue_inflight(socket).await; bus.emit_status("online"); } // 401: flip into needs_login + requeue the message that triggered // the turn so it survives the re-auth. The serve loop's outer // login-state watcher parks until the operator's `/login` flow // completes; once it does, the requeued message replays the turn // (closes #419). if matches!(outcome, turn::TurnOutcome::AuthFailed) { bus.emit_status("needs_login_idle"); bus.emit(LiveEvent::Note { text: "API 401 — waiting for re-login via web UI".into(), }); tracing::warn!("auth-failed; parking until re-login"); requeue_inflight(socket).await; } // Real crash: PromptTooLong is absorbed by compaction inside drive_turn. if let turn::TurnOutcome::Failed(e) = &outcome { notify_manager_of_failure(socket, label, e).await; } if let Some(stats) = stats { let ended_at = serve_common::now_unix(); let duration_ms = i64::try_from(started_instant.elapsed().as_millis()).unwrap_or(i64::MAX); let (open_threads, open_reminders) = fetch_agent_post_turn_counts(socket).await; let row = serve_common::build_row( started_at, ended_at, duration_ms, model_at_start, from.clone(), &outcome, bus, open_threads, open_reminders, ); stats.record(&row); } let pending = inbox_unread(socket).await; if pending > 0 { tracing::info!(%pending, "pending messages after turn; fetching next"); } // `request_next_turn` MCP tool: agent wrote a sentinel requesting // an immediate self-continuation. Clear and inject synthetic wake. check_and_inject_continue(socket, label).await; matches!(outcome, turn::TurnOutcome::AuthFailed) } // Per-turn user prompt: the role/tools/etc. is in the system prompt // (`prompts/system.md` filtered to this agent's role-block via // `hive_ag3nt::prompt::render` → `claude --system-prompt-file`); this // is just the wake signal claude reacts to. `unread` is the count of *other* // messages in the inbox right after this one was popped. // `redelivered` flags messages that were popped in a prior harness // session, never acked, and resurfaced after a restart — a banner // at the top of the wake prompt warns that any side-effects of // previous handling may already have happened. /// Best-effort: tell the broker every message we popped during the /// turn is now fully handled (turn-end-OK). Swallows transport /// errors — the worst case is a redundant requeue on next boot. async fn ack_turn(socket: &Path) { match client::request::<_, AgentResponse>(socket, &AgentRequest::AckTurn).await { Ok(AgentResponse::Ok) => {} Ok(AgentResponse::Err { message }) => { tracing::warn!(%message, "ack_turn rejected by broker"); } Ok(other) => { tracing::warn!(?other, "ack_turn unexpected response"); } Err(e) => tracing::warn!(error = ?e, "ack_turn transport error"), } } /// Boot-time recovery: ask the broker to resurface anything we /// popped in a previous harness session but never acked. The broker /// resets `delivered_at = NULL` on those rows and remembers their /// ids so the next `Recv` carries `redelivered: true`. Swallows /// transport errors — they degrade to "no recovery this boot", /// which is no worse than the pre-feature behaviour (silent drop). async fn requeue_inflight(socket: &Path) { match client::request::<_, AgentResponse>(socket, &AgentRequest::RequeueInflight).await { Ok(AgentResponse::Ok) => {} Ok(AgentResponse::Err { message }) => { tracing::warn!(%message, "requeue_inflight rejected by broker"); } Ok(other) => { tracing::warn!(?other, "requeue_inflight unexpected response"); } Err(e) => tracing::warn!(error = ?e, "requeue_inflight transport error"), } } /// Best-effort: tell the manager that this agent's last turn crashed /// (claude exited non-zero, compaction didn't help, etc.). Routed /// through the normal send path so the manager's inbox surfaces it /// as a system-style event; `label` is included explicitly in the /// body so the manager can identify the failing agent without having /// to look at the `from` field (which is broker-stamped and may /// differ from what the operator sees in the dashboard). Swallows /// transport errors — we just logged the failure, the worst case is /// the manager learns about the crash from the dashboard instead of /// inbox. async fn notify_manager_of_failure(socket: &Path, label: &str, err: &anyhow::Error) { let body = format!("[system] agent `{label}` claude turn failed:\n{err:#}"); let res = client::request::<_, AgentResponse>( socket, &AgentRequest::Send { to: "manager".into(), body, in_reply_to: None, }, ) .await; if let Err(e) = res { tracing::warn!(error = ?e, "failed to notify manager of turn failure"); } } /// Best-effort: ask our own per-agent socket how many messages are still /// pending after the wake-up Recv. Returns 0 if anything goes wrong. async fn inbox_unread(socket: &Path) -> u64 { match client::request::<_, AgentResponse>(socket, &AgentRequest::Status).await { Ok(AgentResponse::Status { unread }) => unread, _ => 0, } } /// Best-effort: ask hive-c0re for this agent's open thread count + pending /// reminder count, after the turn finishes. Either roundtrip can fail /// (transport hiccup, race with hive-c0re restart) — in those cases we /// just drop a `None` into the stats row rather than blocking the loop. async fn fetch_agent_post_turn_counts(socket: &Path) -> (Option, Option) { let threads = match client::request::<_, AgentResponse>( socket, &AgentRequest::GetLooseEnds, ) .await { Ok(AgentResponse::LooseEnds { loose_ends }) => u64::try_from(loose_ends.len()).ok(), _ => None, }; let reminders = match client::request::<_, AgentResponse>( socket, &AgentRequest::CountPendingReminders, ) .await { Ok(AgentResponse::PendingRemindersCount { count }) => Some(count), _ => None, }; (threads, reminders) } /// Check for the `request_next_turn` sentinel file. If present, remove it /// and inject a synthetic `from: "self", body: "continue"` message so the /// serve loop fires an immediate follow-up turn even when the inbox is empty. /// Best-effort: any I/O error is logged and ignored (the agent just waits /// for a real message as normal). async fn check_and_inject_continue(socket: &Path, label: &str) { let sentinel = hive_ag3nt::paths::state_dir().join("hyperhive-continue"); if !sentinel.exists() { return; } if let Err(e) = std::fs::remove_file(&sentinel) { tracing::warn!(error = %e, "check_and_inject_continue: remove sentinel failed"); return; } // Sentinel was present: inject a wake so the outer loop fires immediately. // Route through the `Wake` request which is already wired in agent_server. let res = client::request::<_, AgentResponse>( socket, &AgentRequest::Wake { from: "self".into(), body: "continue".into(), }, ) .await; match res { Ok(AgentResponse::Ok) => { tracing::info!(%label, "request_next_turn: injected self-continue wake"); } Ok(AgentResponse::Err { message }) => { tracing::warn!(%message, "check_and_inject_continue: wake rejected"); } Err(e) => { tracing::warn!(error = ?e, "check_and_inject_continue: wake transport error"); } _ => {} } }