409 lines
16 KiB
Rust
409 lines
16 KiB
Rust
//! Startup auto-migration. Six idempotent phases: applied repo,
|
|
//! proposed repo, meta repo, container repoint, root→h-root rename,
|
|
//! and manager tool-groups backfill.
|
|
//! Kill-switch: `HIVE_SKIP_META_MIGRATION=1`. Full migration sequence
|
|
//! and phase details: `docs/approvals.md::Migration from the pre-tag`.
|
|
|
|
use std::path::Path;
|
|
use std::sync::Arc;
|
|
use std::time::Duration;
|
|
|
|
use anyhow::{Context, Result};
|
|
use tokio::process::Command;
|
|
|
|
use crate::coordinator::Coordinator;
|
|
use crate::lifecycle::{self, AGENT_PREFIX, MANAGER_CONTAINER, MANAGER_NAME};
|
|
use crate::meta;
|
|
use crate::tool_groups;
|
|
|
|
const KILL_SWITCH: &str = "HIVE_SKIP_META_MIGRATION";
|
|
|
|
/// Per-shellout timeouts for the blocking startup migration. `run` is
|
|
/// awaited *before* the daemon starts serving (main.rs), so any child
|
|
/// process that wedges here freezes the whole daemon — admin socket +
|
|
/// dashboard included — with no diagnostics: a git/container shellout was
|
|
/// observed blocked for 86min under a concurrent `nixos-rebuild`. Every
|
|
/// shellout now runs under a timeout that kills the child on elapse, so a
|
|
/// stuck migration degrades to a logged warning instead of a hung boot.
|
|
/// Git ops are quick; `nixos-container update` can legitimately trigger a
|
|
/// nix build, so it gets a much longer budget.
|
|
const GIT_TIMEOUT: Duration = Duration::from_mins(2);
|
|
const CONTAINER_TIMEOUT: Duration = Duration::from_mins(10);
|
|
|
|
/// Substring that identifies the *current* agent flake boilerplate.
|
|
/// Bumped whenever the template changes so the startup migration
|
|
/// re-renders existing agents onto the new shape. Today the marker
|
|
/// is the `flakeInputs` module-arg forwarding line — older templates
|
|
/// (raw `import ./agent.nix`) get rewritten on next hive-c0re start.
|
|
const MODULE_FLAKE_MARKER: &str = "_module.args.flakeInputs";
|
|
|
|
pub async fn run(coord: &Arc<Coordinator>) -> Result<()> {
|
|
if std::env::var(KILL_SWITCH).is_ok() {
|
|
tracing::info!("migration: {KILL_SWITCH} set — skipping");
|
|
return Ok(());
|
|
}
|
|
// Stale meta index lock: a previous hive-c0re crash mid-`git add`
|
|
// can leave `.git/index.lock` behind, which blocks every
|
|
// subsequent meta op until somebody `rm`s it manually. We just
|
|
// booted so nothing of ours is holding it; safe to clear.
|
|
let meta_lock = crate::paths::meta_git_index_lock();
|
|
if meta_lock.exists() {
|
|
match std::fs::remove_file(&meta_lock) {
|
|
Ok(()) => tracing::warn!("cleared stale meta/.git/index.lock"),
|
|
Err(e) => tracing::warn!(error = ?e, "clear stale meta lock failed"),
|
|
}
|
|
}
|
|
let names = enumerate_agents().await;
|
|
tracing::info!(count = names.len(), "migration: scanning");
|
|
|
|
// Phase 0: move harness-owned files out of state/ into harness/.
|
|
// Idempotent — rename is a no-op if the source doesn't exist and
|
|
// the destination already does.
|
|
tracing::debug!("migration: phase 0 (harness files)");
|
|
for name in &names {
|
|
migrate_harness_files(name);
|
|
}
|
|
|
|
// Phase 1 + 2: per-agent applied + proposed.
|
|
tracing::debug!("migration: phase 1+2 (applied + proposed repos)");
|
|
for name in &names {
|
|
tracing::debug!(%name, "migration: applied+proposed");
|
|
if let Err(e) = migrate_applied_repo(name.as_str()).await {
|
|
tracing::warn!(%name, error = ?e, "migration: applied repo rewrite failed");
|
|
}
|
|
let proposed_dir = Coordinator::agent_proposed_dir(name);
|
|
let proposed = lifecycle::setup_proposed(&proposed_dir, name.as_str());
|
|
match tokio::time::timeout(GIT_TIMEOUT, proposed).await {
|
|
Ok(Err(e)) => tracing::warn!(%name, error = ?e, "migration: setup_proposed failed"),
|
|
Err(_) => {
|
|
tracing::warn!(%name, timeout = ?GIT_TIMEOUT, "migration: setup_proposed timed out — skipping");
|
|
}
|
|
Ok(Ok(())) => {}
|
|
}
|
|
}
|
|
|
|
// Phase 3: meta repo.
|
|
tracing::debug!("migration: phase 3 (meta sync_agents)");
|
|
let agents = lifecycle::agents_for_meta_listing()
|
|
.await
|
|
.unwrap_or_default();
|
|
match tokio::time::timeout(GIT_TIMEOUT, meta::sync_agents(&coord.hive_env(), &agents)).await {
|
|
Ok(Err(e)) => tracing::warn!(error = ?e, "migration: meta sync_agents failed"),
|
|
Err(_) => {
|
|
tracing::warn!(timeout = ?GIT_TIMEOUT, "migration: meta sync_agents timed out — skipping");
|
|
}
|
|
Ok(Ok(())) => {}
|
|
}
|
|
|
|
// Phase 4: container repoint, guarded by marker.
|
|
if crate::paths::meta_migration_marker().exists() {
|
|
tracing::debug!("migration: phase 4 marker present, skipping repoint");
|
|
return Ok(());
|
|
}
|
|
tracing::debug!("migration: phase 4 (container repoint)");
|
|
let mut all_ok = true;
|
|
for name in &names {
|
|
// Mark Rebuilding so the crash watcher skips this container
|
|
// during the brief stop+start window the nixos-container
|
|
// update activation triggers. Without this, crash_watch
|
|
// would fire ContainerCrash for every agent here and the
|
|
// manager would spuriously try to recover them.
|
|
let guard =
|
|
coord.transient_guard(name.as_str(), crate::coordinator::TransientKind::Rebuilding);
|
|
let result = repoint_container(name.as_str()).await;
|
|
drop(guard);
|
|
if let Err(e) = result {
|
|
tracing::warn!(%name, error = ?e, "migration: container repoint failed");
|
|
all_ok = false;
|
|
}
|
|
}
|
|
if all_ok
|
|
&& !names.is_empty()
|
|
&& let Err(e) = std::fs::write(crate::paths::meta_migration_marker(), b"done\n")
|
|
{
|
|
tracing::warn!(error = ?e, "migration: write repoint marker failed");
|
|
}
|
|
|
|
// Phase 5: rename `root` nixos-container to `h-root` for naming
|
|
// consistency with sub-agents. Guarded by marker; skipped on
|
|
// fresh installs (conf file absent) and after first successful run.
|
|
rename_manager_container(coord).await;
|
|
|
|
// Phase 6: ensure ruth has explicit tool groups so removing the
|
|
// role-based fallback (Role::Manager → MANAGER_DEFAULT) doesn't
|
|
// silently strip her privileged tools on next rebuild.
|
|
backfill_manager_tool_groups(&names);
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Move harness-owned sqlite/config files out of the agent-visible state dir
|
|
/// and into the sibling harness dir. Best-effort: logs warnings but never
|
|
/// fails. Idempotent — each file is only moved if present at the old path
|
|
/// and absent at the new path.
|
|
fn migrate_harness_files(name: &hive_types::Ident) {
|
|
const HARNESS_FILES: &[&str] = &[
|
|
"hyperhive-events.sqlite",
|
|
"hyperhive-turn-stats.sqlite",
|
|
"hyperhive-model",
|
|
];
|
|
let state_dir = Coordinator::agent_notes_dir(name);
|
|
let harness_dir = Coordinator::agent_harness_dir(name);
|
|
if let Err(e) = std::fs::create_dir_all(&harness_dir) {
|
|
tracing::warn!(%name, error = ?e, "migration: create harness dir failed");
|
|
return;
|
|
}
|
|
for file in HARNESS_FILES {
|
|
let src = state_dir.join(file);
|
|
let dst = harness_dir.join(file);
|
|
if !src.exists() || dst.exists() {
|
|
continue;
|
|
}
|
|
match std::fs::rename(&src, &dst) {
|
|
Ok(()) => tracing::info!(%name, %file, "migration: moved to harness dir"),
|
|
Err(e) => {
|
|
tracing::warn!(%name, %file, error = ?e, "migration: move to harness dir failed");
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Phase 5: rename the `root` nixos-container to `h-root` so the
|
|
/// manager container name is consistent with the `h-` prefix used by
|
|
/// all sub-agents. Idempotent and marker-guarded. Steps:
|
|
///
|
|
/// 1. Check `/etc/nixos-containers/root.conf` exists (old name present).
|
|
/// 2. Stop the `root` container.
|
|
/// 3. Copy `root.conf` → `h-root.conf`.
|
|
/// 4. Move `/var/lib/nixos-containers/root/` → `h-root/` (if present).
|
|
/// 5. `systemctl daemon-reload` so systemd sees the new unit name.
|
|
/// 6. `nixos-container start h-root`.
|
|
/// 7. Write the done marker.
|
|
///
|
|
/// Best-effort: logs warnings on failure. A failed rename leaves both
|
|
/// conf files present; on the next hive-c0re start the marker is
|
|
/// absent so the phase retries.
|
|
async fn rename_manager_container(coord: &Arc<Coordinator>) {
|
|
if crate::paths::hroot_rename_marker().exists() {
|
|
return;
|
|
}
|
|
let old_conf = std::path::PathBuf::from("/etc/nixos-containers/root.conf");
|
|
let new_conf = std::path::PathBuf::from("/etc/nixos-containers/h-root.conf");
|
|
if !old_conf.exists() {
|
|
// Fresh install — root container was never created under the old name.
|
|
let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n");
|
|
return;
|
|
}
|
|
if new_conf.exists() {
|
|
// Already renamed (but marker was lost — write it and return).
|
|
tracing::info!("migration phase 5: h-root.conf already present, marking done");
|
|
let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n");
|
|
return;
|
|
}
|
|
tracing::info!("migration phase 5: renaming root container to h-root");
|
|
let _guard = coord.transient_guard(MANAGER_NAME, crate::coordinator::TransientKind::Rebuilding);
|
|
|
|
// Stop the old container. Abort if stop fails — continuing with a
|
|
// running `root` and then starting `h-root` risks two manager
|
|
// instances racing for the same broker / state files.
|
|
match Command::new("nixos-container")
|
|
.args(["stop", "root"])
|
|
.status()
|
|
.await
|
|
{
|
|
Ok(s) if s.success() => {}
|
|
Ok(s) => {
|
|
tracing::warn!(status = %s, "migration phase 5: nixos-container stop root failed — aborting");
|
|
return;
|
|
}
|
|
Err(e) => {
|
|
tracing::warn!(error = ?e, "migration phase 5: nixos-container stop root failed — aborting");
|
|
return;
|
|
}
|
|
}
|
|
|
|
// Copy conf file.
|
|
if let Err(e) = std::fs::copy(&old_conf, &new_conf) {
|
|
tracing::warn!(error = ?e, "migration phase 5: copy root.conf failed — aborting");
|
|
return;
|
|
}
|
|
|
|
// Move rootfs if it exists (may be absent for ephemeral containers).
|
|
let old_rootfs = std::path::PathBuf::from("/var/lib/nixos-containers/root");
|
|
let new_rootfs = std::path::PathBuf::from("/var/lib/nixos-containers/h-root");
|
|
if old_rootfs.exists()
|
|
&& !new_rootfs.exists()
|
|
&& let Err(e) = std::fs::rename(&old_rootfs, &new_rootfs)
|
|
{
|
|
tracing::warn!(error = ?e, "migration phase 5: rename rootfs failed (non-fatal)");
|
|
}
|
|
|
|
// Daemon reload so systemd picks up the new container@h-root unit.
|
|
if let Err(e) = Command::new("systemctl")
|
|
.args(["daemon-reload"])
|
|
.status()
|
|
.await
|
|
{
|
|
tracing::warn!(error = ?e, "migration phase 5: systemctl daemon-reload failed");
|
|
}
|
|
|
|
// Start the renamed container.
|
|
if let Err(e) = Command::new("nixos-container")
|
|
.args(["start", "h-root"])
|
|
.status()
|
|
.await
|
|
{
|
|
tracing::warn!(error = ?e, "migration phase 5: nixos-container start h-root failed");
|
|
return;
|
|
}
|
|
|
|
tracing::info!("migration phase 5: root container renamed to h-root");
|
|
let _ = std::fs::write(crate::paths::hroot_rename_marker(), b"done\n");
|
|
// Clean up the old conf file so `nixos-container list` doesn't show
|
|
// a stale stopped `root` entry. Best-effort; a failure here is
|
|
// harmless — h-root is already running and the marker is written.
|
|
if let Err(e) = std::fs::remove_file(&old_conf) {
|
|
tracing::warn!(error = ?e, "migration phase 5: remove old root.conf failed (non-fatal)");
|
|
}
|
|
}
|
|
|
|
async fn enumerate_agents() -> Vec<hive_types::Ident> {
|
|
let containers = lifecycle::list().await.unwrap_or_default();
|
|
containers
|
|
.into_iter()
|
|
.filter_map(|c| {
|
|
let name = if c == MANAGER_CONTAINER {
|
|
MANAGER_NAME
|
|
} else {
|
|
c.strip_prefix(AGENT_PREFIX)?
|
|
};
|
|
hive_types::Ident::parse(name).ok()
|
|
})
|
|
.collect()
|
|
}
|
|
|
|
async fn migrate_applied_repo(name: &str) -> Result<()> {
|
|
let dir = crate::paths::applied_dir(name);
|
|
if !dir.join(".git").exists() {
|
|
return Ok(());
|
|
}
|
|
let flake_path = dir.join("flake.nix");
|
|
let cur = std::fs::read_to_string(&flake_path).unwrap_or_default();
|
|
if cur.contains(MODULE_FLAKE_MARKER) {
|
|
return Ok(());
|
|
}
|
|
let want = lifecycle::initial_flake_nix();
|
|
std::fs::write(&flake_path, want).with_context(|| format!("write {}", flake_path.display()))?;
|
|
raw_git(
|
|
&dir,
|
|
&[
|
|
"-c",
|
|
"user.name=c0re",
|
|
"-c",
|
|
"user.email=c0re@hyperhive.local",
|
|
"add",
|
|
"flake.nix",
|
|
],
|
|
)
|
|
.await?;
|
|
raw_git(
|
|
&dir,
|
|
&[
|
|
"-c",
|
|
"user.name=c0re",
|
|
"-c",
|
|
"user.email=c0re@hyperhive.local",
|
|
"commit",
|
|
"-m",
|
|
"migration: module-only flake",
|
|
],
|
|
)
|
|
.await?;
|
|
// Relocate deployed/0 to the migration commit so
|
|
// setup_applied's existence check passes.
|
|
raw_git(&dir, &["tag", "-f", "deployed/0", "HEAD"]).await?;
|
|
tracing::info!(%name, "migration: applied repo migrated to module-only flake");
|
|
Ok(())
|
|
}
|
|
|
|
async fn repoint_container(name: &str) -> Result<()> {
|
|
let container = lifecycle::container_name(name);
|
|
let flake_ref = format!("{}#{name}", crate::paths::meta_root().display());
|
|
let mut cmd = Command::new("nixos-container");
|
|
cmd.args(["update", &container, "--flake", &flake_ref]);
|
|
let out = output_with_timeout(
|
|
cmd,
|
|
CONTAINER_TIMEOUT,
|
|
&format!("nixos-container update {container}"),
|
|
)
|
|
.await?;
|
|
if !out.status.success() {
|
|
anyhow::bail!(
|
|
"nixos-container update {container} exited {}: {}",
|
|
out.status,
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
tracing::info!(%name, %container, "migration: container repointed at meta");
|
|
Ok(())
|
|
}
|
|
|
|
/// Phase 6: if ruth is a deployed agent and has no explicit entry in
|
|
/// `tool-groups.json`, set her groups to `MANAGER_DEFAULT` (all groups).
|
|
/// Idempotent — skips when entry already present. Prevents a silent tool
|
|
/// downgrade when upgrading from a build that relied on the manager-flavor
|
|
/// fallback in `effective_tool_groups()`.
|
|
fn backfill_manager_tool_groups(names: &[hive_types::Ident]) {
|
|
if !names.iter().any(|n| n.as_str() == MANAGER_NAME) {
|
|
return; // ruth not deployed — nothing to backfill
|
|
}
|
|
let existing = tool_groups::groups_for(MANAGER_NAME);
|
|
if !existing.is_empty() {
|
|
tracing::debug!("migration: ruth already has explicit tool groups — skipping backfill");
|
|
return;
|
|
}
|
|
let all_groups: Vec<String> = hive_sh4re::ToolGroup::MANAGER_DEFAULT
|
|
.iter()
|
|
.map(|g| g.as_str().to_owned())
|
|
.collect();
|
|
match tool_groups::set_groups(MANAGER_NAME, &all_groups) {
|
|
Ok(()) => tracing::info!(
|
|
"migration: backfilled ruth's tool groups to MANAGER_DEFAULT (all groups)"
|
|
),
|
|
Err(e) => tracing::warn!(
|
|
error = ?e,
|
|
"migration: failed to backfill ruth's tool groups — she may lose privileged tools on next rebuild"
|
|
),
|
|
}
|
|
}
|
|
|
|
/// Run a command to completion under a timeout, capturing its output. On
|
|
/// timeout the child is killed (`kill_on_drop`) and an error is returned,
|
|
/// so a wedged shellout can never freeze startup migration. `what` is a
|
|
/// human label surfaced in the timeout error + `with_context`.
|
|
async fn output_with_timeout(
|
|
mut cmd: Command,
|
|
timeout: Duration,
|
|
what: &str,
|
|
) -> Result<std::process::Output> {
|
|
cmd.kill_on_drop(true);
|
|
match tokio::time::timeout(timeout, cmd.output()).await {
|
|
Ok(r) => r.with_context(|| format!("run {what}")),
|
|
Err(_) => anyhow::bail!("{what} timed out after {timeout:?} (child killed)"),
|
|
}
|
|
}
|
|
|
|
async fn raw_git(dir: &Path, args: &[&str]) -> Result<()> {
|
|
let mut cmd = lifecycle::git_command();
|
|
cmd.current_dir(dir).args(args);
|
|
let label = format!("git {} in {}", args.join(" "), dir.display());
|
|
let out = output_with_timeout(cmd, GIT_TIMEOUT, &label).await?;
|
|
if !out.status.success() {
|
|
anyhow::bail!(
|
|
"git {} failed: {}",
|
|
args.join(" "),
|
|
String::from_utf8_lossy(&out.stderr).trim()
|
|
);
|
|
}
|
|
Ok(())
|
|
}
|