use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex}; use std::time::Duration; use hive_ag3nt::web_ui::TurnLock; use anyhow::Result; use clap::{Parser, Subcommand}; use hive_ag3nt::events::{Bus, LiveEvent, TurnState}; use hive_ag3nt::login::{self, LoginState}; use hive_ag3nt::turn_stats::{TurnStatRow, TurnStats}; use hive_ag3nt::{DEFAULT_SOCKET, DEFAULT_WEB_PORT, client, mcp, plugins, turn, web_ui}; use hive_sh4re::{AgentRequest, AgentResponse}; #[derive(Parser)] #[command(name = "hive-ag3nt", about = "hyperhive sub-agent harness")] struct Cli { /// Path to the per-agent MCP socket (bind-mounted from the host). #[arg(long, global = true, default_value = DEFAULT_SOCKET)] socket: PathBuf, #[command(subcommand)] cmd: Cmd, } #[derive(Subcommand)] enum Cmd { /// Run the long-lived harness loop. Polls inbox; replies via `claude --print` /// when available, falling back to a simple echo otherwise. Serve { /// Inbox poll interval in milliseconds. #[arg(long, default_value_t = 1000)] poll_ms: u64, }, /// Run the agent's MCP server on stdio. Spawned by `claude` via /// `--mcp-config`; tools dispatch through `/run/hive/mcp.sock` back into /// the hyperhive broker. Mcp, /// Inject a wake-up event into this agent's inbox so the next turn /// fires with the given body. Intended for extra MCP servers / /// helpers running inside the container (matrix bridge, scraper, /// webhook listener) that need to nudge claude on external events. /// `from` is the sender label that appears in the wake prompt /// (claude sees "from: matrix" etc.). Wake { #[arg(long)] from: String, /// Body of the wake message. Pass `-` to read from stdin. #[arg(long)] body: String, }, } #[tokio::main] async fn main() -> Result<()> { tracing_subscriber::fmt() .with_env_filter( tracing_subscriber::EnvFilter::try_from_default_env() .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info")), ) .init(); let cli = Cli::parse(); match cli.cmd { Cmd::Serve { poll_ms } => { let port = std::env::var("HIVE_PORT") .ok() .and_then(|s| s.parse::().ok()) .unwrap_or(DEFAULT_WEB_PORT); let label = std::env::var("HIVE_LABEL").unwrap_or_else(|_| "hive-ag3nt".into()); let claude_dir = login::default_dir(); let initial = LoginState::from_dir(&claude_dir); tracing::info!(state = ?initial, claude_dir = %claude_dir.display(), "harness boot"); let login_state = Arc::new(Mutex::new(initial)); let bus = Bus::new(); let stats = TurnStats::open_default(); let files = turn::TurnFiles::prepare(&cli.socket, &label, mcp::Flavor::Agent).await?; let turn_lock: TurnLock = Arc::new(tokio::sync::Mutex::new(())); plugins::install_configured(&cli.socket, Some("manager")).await; tokio::spawn(web_ui::serve( label.clone(), port, login_state.clone(), bus.clone(), cli.socket.clone(), files.clone(), turn_lock.clone(), )); match initial { LoginState::Online => { serve( &cli.socket, Duration::from_millis(poll_ms), login_state, bus, stats, &files, turn_lock, &label, ) .await } LoginState::NeedsLogin => { // Partial-run mode: keep the harness alive (so the web UI // stays bound) but don't drive the turn loop. Poll the // claude dir; once a session lands we enter `serve`. turn::wait_for_login(&claude_dir, login_state.clone(), &bus, poll_ms).await; serve( &cli.socket, Duration::from_millis(poll_ms), login_state, bus, stats, &files, turn_lock, &label, ) .await } } } Cmd::Mcp => mcp::serve_agent_stdio(cli.socket).await, Cmd::Wake { from, body } => { // Read body from stdin if caller passed `-`. Same convention // many CLI tools use; keeps multi-line / shell-quoting // friction out of the body content. let body = if body == "-" { let mut buf = String::new(); std::io::Read::read_to_string(&mut std::io::stdin(), &mut buf)?; buf } else { body }; let resp: AgentResponse = client::request(&cli.socket, &AgentRequest::Wake { from, body }).await?; match resp { AgentResponse::Ok => Ok(()), AgentResponse::Err { message } => anyhow::bail!("wake: {message}"), other => anyhow::bail!("wake: unexpected response {other:?}"), } } } } async fn serve( socket: &Path, interval: Duration, state: Arc>, bus: Bus, stats: Option, files: &turn::TurnFiles, turn_lock: TurnLock, label: &str, ) -> Result<()> { tracing::info!(socket = %socket.display(), "hive-ag3nt serve"); let _ = state; // reserved for future state transitions (turn-loop -> needs-login) loop { let recv: Result = // Explicit long-poll: the new agent_server semantics treat // `None` as "peek, don't wait", which would tight-loop on // sleep(interval). The harness wants to park until a // message arrives, so opt into the full 180s cap. client::request( socket, &AgentRequest::Recv { wait_seconds: Some(180), }, ) .await; match recv { Ok(AgentResponse::Message { from, body }) => { tracing::info!(%from, %body, "inbox"); let unread = inbox_unread(socket).await; bus.emit(LiveEvent::TurnStart { from: from.clone(), body: body.clone(), unread, }); bus.set_state(TurnState::Thinking); let started_at = now_unix(); let started_instant = std::time::Instant::now(); let model_at_start = bus.model(); let prompt = format_wake_prompt(&from, &body, unread); let outcome = { let _guard = turn_lock.lock().await; turn::drive_turn(&prompt, files, &bus).await }; turn::emit_turn_end(&bus, &outcome); bus.set_state(TurnState::Idle); // Failures are unhandled by definition — PromptTooLong is // absorbed inside drive_turn via compaction, so anything // that reaches Failed here is a real crash. Notify the // manager so it can investigate / restart / page the // operator; best-effort, swallow the send error. if let turn::TurnOutcome::Failed(e) = &outcome { notify_manager_of_failure(socket, label, e).await; } if let Some(s) = &stats { let ended_at = now_unix(); let duration_ms = i64::try_from(started_instant.elapsed().as_millis()).unwrap_or(i64::MAX); let (open_threads, open_reminders) = fetch_agent_post_turn_counts(socket).await; let row = build_row( started_at, ended_at, duration_ms, model_at_start, from.clone(), &outcome, &bus, open_threads, open_reminders, ); s.record(&row); } // After turn completes, check if there are pending messages waiting. // If so, immediately process them instead of blocking on recv(). // This ensures messages queued during the turn are processed ASAP. let pending = inbox_unread(socket).await; if pending > 0 { tracing::info!(%pending, "pending messages after turn; fetching next"); continue; // Loop back to recv() immediately instead of sleeping } } Ok(AgentResponse::Empty) => { // Idle: brief sleep before next poll to avoid busy-looping // on consecutive Empty responses. The recv() call already // waits up to 180s for messages, so this is just for // responsiveness if recv() times out. tokio::time::sleep(interval).await; } Ok( AgentResponse::Ok | AgentResponse::Status { .. } | AgentResponse::Recent { .. } | AgentResponse::QuestionQueued { .. } | AgentResponse::OpenThreads { .. } | AgentResponse::PendingRemindersCount { .. } | AgentResponse::Whoami { .. }, ) => { tracing::warn!("recv produced unexpected response kind"); } Ok(AgentResponse::Err { message }) => { tracing::warn!(%message, "recv error"); } Err(e) => { tracing::warn!(error = ?e, "recv failed; retrying"); } } } } /// Per-turn user prompt. The role/tools/etc. is in the system prompt /// (`prompts/agent.md` → `claude --system-prompt-file`); this is just the /// wake signal claude reacts to. `unread` is the count of *other* /// messages in the inbox right after this one was popped. fn format_wake_prompt(from: &str, body: &str, unread: u64) -> String { let pending = if unread == 0 { String::new() } else { format!( "\n\n({unread} more message(s) pending in your inbox — drain via `mcp__hyperhive__recv` if relevant.)" ) }; format!("Incoming message from `{from}`:\n---\n{body}\n---{pending}") } /// Best-effort: tell the manager that this agent's last turn crashed /// (claude exited non-zero, compaction didn't help, etc.). Routed /// through the normal send path so the manager's inbox surfaces it /// as a system-style event; `label` is included explicitly in the /// body so the manager can identify the failing agent without having /// to look at the `from` field (which is broker-stamped and may /// differ from what the operator sees in the dashboard). Swallows /// transport errors — we just logged the failure, the worst case is /// the manager learns about the crash from the dashboard instead of /// inbox. async fn notify_manager_of_failure(socket: &Path, label: &str, err: &anyhow::Error) { let body = format!("[system] agent `{label}` claude turn failed:\n{err:#}"); let res = client::request::<_, AgentResponse>( socket, &AgentRequest::Send { to: "manager".into(), body, }, ) .await; if let Err(e) = res { tracing::warn!(error = ?e, "failed to notify manager of turn failure"); } } /// Best-effort: ask our own per-agent socket how many messages are still /// pending after the wake-up Recv. Returns 0 if anything goes wrong. async fn inbox_unread(socket: &Path) -> u64 { match client::request::<_, AgentResponse>(socket, &AgentRequest::Status).await { Ok(AgentResponse::Status { unread }) => unread, _ => 0, } } fn now_unix() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .ok() .and_then(|d| i64::try_from(d.as_secs()).ok()) .unwrap_or(0) } /// Best-effort: ask hive-c0re for this agent's open thread count + pending /// reminder count, after the turn finishes. Either roundtrip can fail /// (transport hiccup, race with hive-c0re restart) — in those cases we /// just drop a `None` into the stats row rather than blocking the loop. async fn fetch_agent_post_turn_counts(socket: &Path) -> (Option, Option) { let threads = match client::request::<_, AgentResponse>( socket, &AgentRequest::GetOpenThreads, ) .await { Ok(AgentResponse::OpenThreads { threads }) => u64::try_from(threads.len()).ok(), _ => None, }; let reminders = match client::request::<_, AgentResponse>( socket, &AgentRequest::CountPendingReminders, ) .await { Ok(AgentResponse::PendingRemindersCount { count }) => Some(count), _ => None, }; (threads, reminders) } /// Assemble a TurnStatRow from the harness's per-turn state. Shared /// shape between the agent + manager bin loops (each lives in its own /// crate root so this helper is duplicated; the savings of a shared /// module aren't worth the cross-crate ceremony at this size). fn build_row( started_at: i64, ended_at: i64, duration_ms: i64, model: String, wake_from: String, outcome: &turn::TurnOutcome, bus: &Bus, open_threads_count: Option, open_reminders_count: Option, ) -> TurnStatRow { let usage = bus.last_usage().unwrap_or_default(); let tool_calls = bus.take_tool_calls(); let tool_call_count: u64 = tool_calls.values().copied().sum(); let tool_call_breakdown_json = if tool_calls.is_empty() { None } else { serde_json::to_string(&tool_calls).ok() }; let (result_kind, note) = match outcome { turn::TurnOutcome::Ok => ("ok", None), turn::TurnOutcome::PromptTooLong => ("prompt_too_long", None), turn::TurnOutcome::Failed(e) => ("failed", Some(format!("{e:#}"))), }; TurnStatRow { started_at, ended_at, duration_ms, model, wake_from, input_tokens: usage.input_tokens, output_tokens: usage.output_tokens, cache_read_input_tokens: usage.cache_read_input_tokens, cache_creation_input_tokens: usage.cache_creation_input_tokens, tool_call_count, tool_call_breakdown_json, open_threads_count, open_reminders_count, result_kind, note, } }