hive-c0re: fail a nixos-container destroy that leaves the container in place
lifecycle::destroy logged a failed `nixos-container destroy` and returned Ok, so the DestroyContainer node went green, the agent was unregistered, and with purge the after_ok PurgeState deleted its state while the container config and root still existed. Propagate the error unless the container list, read after the failure, no longer names the container. An unreadable list fails too.
This commit is contained in:
parent
5b54b6ddc8
commit
18bd8dd2c7
3 changed files with 85 additions and 3 deletions
|
|
@ -721,13 +721,18 @@ pub async fn is_running(name: &str) -> bool {
|
|||
/// Fully tear down a sub-agent's container: stop + remove via `nixos-container
|
||||
/// destroy`, then clean our own systemd drop-in. Leaves it to the caller to
|
||||
/// wipe `/var/lib/hyperhive/...` state and the per-agent runtime dir.
|
||||
///
|
||||
/// Fails when the container survives the destroy. The destroy DAG purges the
|
||||
/// agent's state only after this returns `Ok`, so a swallowed failure here
|
||||
/// deletes the state of a container that still exists.
|
||||
pub async fn destroy(name: &str) -> Result<()> {
|
||||
validate(name)?;
|
||||
let container = container_name(name);
|
||||
// nixos-container destroy handles stop + removal of /var/lib/nixos-containers/<C>
|
||||
// and /etc/nixos-containers/<C>.conf. Tolerate "no such container".
|
||||
// and /etc/nixos-containers/<C>.conf. When it errors, the container list
|
||||
// decides: a container that is already gone is still a successful destroy.
|
||||
if let Err(e) = priv_run("destroy", name).await {
|
||||
tracing::warn!(error = ?e, "nixos-container destroy returned an error; continuing cleanup");
|
||||
confirm_gone_after_failed_destroy(e, &container, list().await)?;
|
||||
}
|
||||
// Remove the systemd resource-limits drop-in via hive-priv.
|
||||
if let Err(e) = crate::priv_client::remove_service_dropin(&container).await {
|
||||
|
|
@ -736,6 +741,30 @@ pub async fn destroy(name: &str) -> Result<()> {
|
|||
Ok(())
|
||||
}
|
||||
|
||||
/// Settle a failed `nixos-container destroy` against the container list taken
|
||||
/// after it. `Ok` only when the list is readable and no longer names the
|
||||
/// container; a list that cannot be read proves nothing, so it fails too.
|
||||
fn confirm_gone_after_failed_destroy(
|
||||
destroy_err: anyhow::Error,
|
||||
container: &str,
|
||||
listed: Result<Vec<String>>,
|
||||
) -> Result<()> {
|
||||
match listed {
|
||||
Ok(names) if !names.iter().any(|n| n == container) => {
|
||||
tracing::warn!(
|
||||
error = ?destroy_err,
|
||||
%container,
|
||||
"nixos-container destroy returned an error, but the container is gone"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
Ok(_) => Err(destroy_err.context(format!("container {container} still exists"))),
|
||||
Err(list_err) => Err(destroy_err.context(format!(
|
||||
"cannot confirm whether container {container} still exists: {list_err:#}"
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
/// Pre-build `system.build.toplevel` against `meta#<name>` so the
|
||||
/// subsequent `nixos-container update` finds the result cached and
|
||||
/// skips straight to the profile-swap. Store-warming only — container
|
||||
|
|
|
|||
Loading…
Reference in a new issue