Watch
0
0
Fork
You've already forked hyperhive
0

hive-c0re: bound prebuild_toplevel's nix build with a timeout

prebuild_toplevel awaited nix build with a bare child.wait().await, so
a wedged nix-daemon (unreachable remote builder, stuck build slot)
hung the rebuild job forever with no way for the job queue to recover
short of restarting hive-c0re.

Mirrors #4741's hive-priv fix: the child now leads its own process
group, and after PREBUILD_TIMEOUT (1h, four times CI's observed
cold-cache flake check) the whole group is SIGKILLed and the call
fails with a named timeout error instead of hanging.

Refs #4723, #4741
This commit is contained in:
atlas 2026-09-28 23:22:10 +02:00 • committed by mara
commit 5409f8160b
2 changed files with 113 additions and 3 deletions

View file

@ -694,6 +694,39 @@ fn confirm_gone_after_failed_destroy(
}
}
/// Wall-clock bound on [`prebuild_toplevel`]'s `nix build`. Four times CI's
/// observed cold-cache `nix flake check` (~15 min, `.forgejo/workflows/ci.yml`),
/// which compiles this workspace's crates, the part of an agent toplevel no
/// binary cache holds. A stalled download is bounded by nix's own
/// `stalled-download-timeout`; this covers the rest, such as a wedged
/// nix-daemon holding the rebuild's build slot forever.
const PREBUILD_TIMEOUT: std::time::Duration = std::time::Duration::from_hours(1);
/// Wait for `child` until `limit`. `None` means it was still running, and its
/// whole process group was `SIGKILL`ed and the child reaped, so a grandchild
/// holding the daemon connection dies with it. The child must have been
/// spawned with `process_group(0)`.
async fn wait_bounded(
child: &mut tokio::process::Child,
limit: std::time::Duration,
) -> std::io::Result<Option<std::process::ExitStatus>> {
// Group leader, so its pid is the pgid. Read before waiting: a reaped
// child has no id.
let pgid = child.id().and_then(|pid| i32::try_from(pid).ok());
if let Ok(status) = tokio::time::timeout(limit, child.wait()).await {
return status.map(Some);
}
if let Some(pgid) = pgid {
// SAFETY: plain `kill(2)`; `-pgid` targets the child's own process
// group, which `process_group(0)` made it lead.
unsafe {
libc::kill(-pgid, libc::SIGKILL);
}
}
child.wait().await?;
Ok(None)
}
/// Pre-build `system.build.toplevel` against `meta#<name>` so the
/// subsequent `nixos-container update` finds the result cached and
/// skips straight to the profile-swap. Store-warming only — container
@ -750,6 +783,7 @@ pub async fn prebuild_toplevel(name: &str, flake_ref: &str, node_id: Option<u64>
.args(&args)
.stdout(std::process::Stdio::piped())
.stderr(std::process::Stdio::piped())
.process_group(0)
.spawn()
.with_context(|| format!("spawn {cmdline}"))?;
@ -780,10 +814,24 @@ pub async fn prebuild_toplevel(name: &str, flake_ref: &str, node_id: Option<u64>
}
});
let status = child
.wait()
let Some(status) = wait_bounded(&mut child, PREBUILD_TIMEOUT)
.await
.with_context(|| format!("wait {cmdline}"))?;
.with_context(|| format!("wait {cmdline}"))?
else {
// A process that left the group can still hold the pipes open.
pump_stdout.abort();
pump_stderr.abort();
if let (Some(h), Some(id)) = (&logs, log_id) {
h.finish(id, crate::build_logs::BuildStatus::Fail);
}
let secs = PREBUILD_TIMEOUT.as_secs();
match log_id {
Some(id) => {
bail!("prebuild {cmdline} timed out after {secs}s; killed it; see build log #{id}")
}
None => bail!("prebuild {cmdline} timed out after {secs}s; killed it"),
}
};
let _ = pump_stdout.await;
let _ = pump_stderr.await;

View file

@ -348,3 +348,65 @@ fn agent_names_propagates_an_unreadable_list() {
.expect_err("an unreadable list must not read as zero agents");
assert!(format!("{err:#}").contains("hive-priv"), "{err:#}");
}
fn is_dead(pid: &str) -> bool {
// A killed process that is not our child stays a zombie until its own
// parent reaps it, so a zombie counts as dead.
match std::fs::read_to_string(format!("/proc/{pid}/stat")) {
Err(_) => true,
Ok(stat) => stat
.rsplit_once(") ")
.is_some_and(|(_, rest)| rest.starts_with('Z')),
}
}
/// A child still running at the limit comes back `None`, and the kill reaches
/// its whole process group: the backgrounded grandchild dies too, not just the
/// shell that started it.
#[tokio::test]
async fn a_prebuild_past_its_limit_is_killed_with_its_group() {
use std::time::{Duration, Instant};
let dir = tempfile::tempdir().expect("tempdir");
let pidfile = dir.path().join("grandchild");
let mut child = Command::new("sh")
.arg("-c")
.arg(format!("sleep 600 & echo $! > {}; wait", pidfile.display()))
.process_group(0)
.spawn()
.expect("spawn sh");
let started = Instant::now();
let status = wait_bounded(&mut child, Duration::from_secs(2))
.await
.expect("wait_bounded");
assert!(status.is_none(), "must time out, got {status:?}");
assert!(
started.elapsed() < Duration::from_mins(1),
"the limit did not bound the wait"
);
let pid = std::fs::read_to_string(&pidfile).expect("grandchild pid");
let pid = pid.trim();
let deadline = Instant::now() + Duration::from_secs(5);
while !is_dead(pid) && Instant::now() < deadline {
tokio::time::sleep(Duration::from_millis(50)).await;
}
assert!(is_dead(pid), "grandchild {pid} outlived the timeout");
}
/// The control: a child that exits inside its limit comes back with its
/// status.
#[tokio::test]
async fn a_prebuild_inside_its_limit_returns_its_status() {
let mut child = Command::new("sh")
.arg("-c")
.arg("exit 3")
.process_group(0)
.spawn()
.expect("spawn sh");
let status = wait_bounded(&mut child, std::time::Duration::from_mins(1))
.await
.expect("wait_bounded")
.expect("a fast child must not time out");
assert_eq!(status.code(), Some(3));
}