hive-priv/hive-c0re: drop stop's SIGKILL escalation, surface a crit dashboard warning instead
This commit is contained in:
parent
da3fc9bf95
commit
b118b12520
2 changed files with 73 additions and 25 deletions
|
|
@ -1927,12 +1927,19 @@ async fn wait_name_released(machine: &str) -> bool {
|
|||
/// there is no cleaning it up after the fact. The only place to catch it is
|
||||
/// here, in the stop.
|
||||
///
|
||||
/// So: ask for the stop, wait for the name, and escalate to SIGKILL if it is
|
||||
/// still held. If even that doesn't free it, fail loudly — a caller that is
|
||||
/// told the stop worked will go on to start the container and hit the
|
||||
/// confusing registration error instead of this one.
|
||||
/// So: ask for the stop, wait for the name, and fail loudly if it is still
|
||||
/// held — a caller that is told the stop worked will go on to start the
|
||||
/// container and hit the confusing registration error instead of this one.
|
||||
///
|
||||
/// The verify-then-escalate lives in the helper rather than at a call site so
|
||||
/// This used to escalate to `machinectl kill --signal=SIGKILL` here. Dropped
|
||||
/// per real incident data: the case this guards is a container genuinely
|
||||
/// wedged (e.g. a root-login process survived the stop), and SIGKILL
|
||||
/// doesn't recover that in practice — only a host-level reboot has.
|
||||
/// Pretending a kill attempt handled it hides a condition that needs a human
|
||||
/// to look at the host, so this now just reports the failure instead of
|
||||
/// quietly (and ineffectually) trying to force it.
|
||||
///
|
||||
/// The verify-then-fail lives in the helper rather than at a call site so
|
||||
/// that every stop gets it: dashboard, reconcile, destroy, cold-start
|
||||
/// fallback. The start path already distrusts its own exit code the same way;
|
||||
/// this is the missing half of that pair.
|
||||
|
|
@ -1945,28 +1952,19 @@ async fn stop_and_release(machine: &str) -> Result<(String, String)> {
|
|||
return stop;
|
||||
}
|
||||
|
||||
tracing::warn!(
|
||||
tracing::error!(
|
||||
%machine,
|
||||
"stop finished but machined still holds the registration; escalating to SIGKILL"
|
||||
);
|
||||
let kill = machinectl_run(&["kill", machine, "--signal=SIGKILL"]).await;
|
||||
if wait_name_released(machine).await {
|
||||
if let Err(e) = &kill {
|
||||
// The name is free, which is what was asked for. machinectl
|
||||
// complaining on the way there is worth a line, not a failure.
|
||||
tracing::warn!(%machine, error = %e, "SIGKILL reported an error but the name was released");
|
||||
}
|
||||
return Ok(kill.unwrap_or_default());
|
||||
}
|
||||
|
||||
let detail = kill.err().map_or_else(
|
||||
|| "SIGKILL was delivered".to_owned(),
|
||||
|e| format!("SIGKILL also failed: {e}"),
|
||||
"stop finished but machined still holds the registration — the \
|
||||
container is likely wedged (a process outside its init tree \
|
||||
survived the stop); this needs a host-level look, not another \
|
||||
stop attempt"
|
||||
);
|
||||
bail!(
|
||||
"stop {machine}: machined still holds the machine name after {}s ({detail}). \
|
||||
Starting this container again will fail to register.",
|
||||
NAME_RELEASE_TIMEOUT.as_secs() * 2
|
||||
"stop {machine}: machined still holds the machine name after {}s. \
|
||||
This container is likely wedged and starting it again will fail to \
|
||||
register — needs host-level intervention (a reboot has been the \
|
||||
only reliable fix in practice).",
|
||||
NAME_RELEASE_TIMEOUT.as_secs()
|
||||
)
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue