swarm-controller: answer 503 on a disconnected queue, not 500 or a hang

put_matrix_account writes the credential to bao before checking that
the queue it must notify is actually connected — only that a queue is
configured, via state.status.as_ref(). While the queue is Pending or
Disconnected, client.flush().await hangs (async-nats does not process
commands during the initial connect retry), so the request hangs until
nginx times out and the credential is already stored. Add the
ensure_connected check every sibling queue route already makes
(wanted.rs, term_stream.rs, agent_state_stream.rs), placed immediately
before store::connect() so a request that cannot be delivered never
reaches the store.

Same shape, lower impact, in webhook::announce_knowledge_change: it
awaits publish/flush inline in the webhook handler, so a disconnected
queue can run past forgejo's short delivery timeout. Add the same
ensure_connected guard, warn and return.

set_agent_state and get_hive_wanted map every writer error to 500,
including swarm_queue_client::Error::NotConnected surfaced through
WantedWriter::view/set. Add wanted_error_status, which downcasts the
anyhow::Error back to the concrete type and maps NotConnected to 503
(retryable) while leaving every other failure at 500. Update both
routes' OpenAPI descriptions to say so.

Closes #4688
This commit is contained in:
atlas 2026-09-24 14:05:18 +02:00 • committed by mara
commit 6a87b25488
3 changed files with 327 additions and 14 deletions

View file

@ -795,6 +795,26 @@ fn error_problem(status: axum::http::StatusCode, detail: &str) -> problem_detail
problem_details::ProblemDetails::from_status_code(status).with_detail(detail)
}
/// The status a [`wanted::WantedWriter`] failure should answer with.
///
/// `WantedWriter::view`/`set` call `ensure_connected` internally and
/// propagate through `anyhow`, so the concrete
/// [`swarm_queue_client::Error`] survives underneath but is not the type in
/// hand — `swarm_queue_client`'s own doc says a caller "gets a type it can
/// match on", which means downcasting back to it rather than string-matching
/// the rendered message. `NotConnected` is a retryable condition (the queue
/// is reconnecting, or hasn't yet) and every other queue-backed route here
/// already answers 503 for it; anything else is a genuine failure to publish
/// or read, which stays 500.
fn wanted_error_status(e: &anyhow::Error) -> axum::http::StatusCode {
match e.downcast_ref::<swarm_queue_client::Error>() {
Some(swarm_queue_client::Error::NotConnected(_)) => {
axum::http::StatusCode::SERVICE_UNAVAILABLE
}
_ => axum::http::StatusCode::INTERNAL_SERVER_ERROR,
}
}
/// The state to declare for one agent.
#[derive(Debug, Deserialize, ToSchema)]
struct SetAgentStateRequest {
@ -905,7 +925,7 @@ fn declaration_target(
responses(
(status = 200, description = "the declaration as now published", body = Vec<AgentDeclaration>),
(status = 400, description = "a name is not an identifier, the hive is not in this swarm, or the state is unknown (problem+json)", body = String),
(status = 503, description = "no swarm queue is wired up (problem+json)", body = String),
(status = 503, description = "no swarm queue is wired up, or it is not connected (problem+json)", body = String),
(status = 500, description = "the declaration could not be published (problem+json)", body = String),
),
tag = "agents"
@ -923,10 +943,7 @@ async fn set_agent_state(
let declaration = writer.set(&hive, &agent, req.state).await.map_err(|e| {
tracing::warn!(hive = %hive, agent = %agent, error = %format!("{e:#}"), "declaring agent state failed");
error_problem(
axum::http::StatusCode::INTERNAL_SERVER_ERROR,
&format!("{e:#}"),
)
error_problem(wanted_error_status(&e), &format!("{e:#}"))
})?;
Ok(Json(render(&declaration)))
}
@ -942,7 +959,7 @@ async fn set_agent_state(
responses(
(status = 200, description = "the declaration, empty when nothing is published yet", body = Vec<AgentDeclaration>),
(status = 400, description = "not an identifier, or not a hive in this swarm (problem+json)", body = String),
(status = 503, description = "no swarm queue is wired up (problem+json)", body = String),
(status = 503, description = "no swarm queue is wired up, or it is not connected (problem+json)", body = String),
(status = 500, description = "the declaration could not be read (problem+json)", body = String),
),
tag = "agents"
@ -955,10 +972,7 @@ async fn get_hive_wanted(
declaration_target(&state, &hive).map_err(|(s, d)| error_problem(s, &d))?;
let declaration = writer.view(&hive).await.map_err(|e| {
tracing::warn!(hive = %hive, error = %format!("{e:#}"), "reading the declaration failed");
error_problem(
axum::http::StatusCode::INTERNAL_SERVER_ERROR,
&format!("{e:#}"),
)
error_problem(wanted_error_status(&e), &format!("{e:#}"))
})?;
Ok(Json(declaration.as_ref().map(render).unwrap_or_default()))
}
@ -1970,8 +1984,9 @@ fn build_app(state: AppState) -> axum::Router {
#[cfg(test)]
mod tests {
use super::{
DEFAULT_SOCKET, HIVES_ENV, HiveEntry, LINKS_ENV, NAME_ENV, ServiceLink, StatusUnavailable,
SwarmNodeKind, WorkerDeps, load_hives, load_links, load_swarm_name, run_swarm_node,
DEFAULT_SOCKET, HIVES_ENV, HiveEntry, LINKS_ENV, NAME_ENV, ServiceLink,
SetAgentStateRequest, StatusUnavailable, SwarmNodeKind, WorkerDeps, load_hives, load_links,
load_swarm_name, run_swarm_node, wanted,
};
use std::path::Path;
@ -2782,6 +2797,150 @@ mod tests {
std::env::remove_var(NAME_ENV);
}
}
// ── NotConnected maps to 503, everything else stays 500 ─────────────
/// `wanted_error_status` as a pure function first: build the exact
/// `swarm_queue_client::Error::NotConnected` `ensure_connected` raises,
/// wrap it the same way `?` inside `WantedWriter::view`/`set` does, and
/// confirm it downcasts back to the status those routes are meant to
/// answer.
#[test]
fn not_connected_downcasts_to_503() {
let e: anyhow::Error =
swarm_queue_client::Error::NotConnected(async_nats::connection::State::Pending).into();
assert_eq!(
super::wanted_error_status(&e),
axum::http::StatusCode::SERVICE_UNAVAILABLE
);
}
/// Invert-proof for the test above: an `anyhow::Error` that does not
/// wrap a `swarm_queue_client::Error` at all (the shape a genuine write
/// or decode failure takes — `apply`'s "declaration is not decodable"
/// context, for instance) must NOT downcast to `NotConnected`, and must
/// stay 500. Without this, a `wanted_error_status` that answered 503
/// unconditionally would still pass the test above.
#[test]
fn an_unrelated_error_stays_500() {
let e = anyhow::anyhow!("the hive's current declaration is not decodable");
assert_eq!(
super::wanted_error_status(&e),
axum::http::StatusCode::INTERNAL_SERVER_ERROR
);
}
/// Second invert-proof, closer to the actual bug this closes: a
/// *different* `swarm_queue_client::Error` variant (a real connect
/// failure, not `NotConnected`) also stays 500 — proving the match
/// looks at the variant, not just at "was this crate's error type
/// involved at all".
#[test]
fn a_different_queue_client_error_variant_stays_500() {
let e: anyhow::Error = swarm_queue_client::Error::Connect {
url: "nats://queue.example:4222".to_owned(),
source: async_nats::ConnectErrorKind::TimedOut.into(),
}
.into();
assert_eq!(
super::wanted_error_status(&e),
axum::http::StatusCode::INTERNAL_SERVER_ERROR
);
}
/// A client that exists but has never connected — same shape
/// `matrix_account`'s tests build, duplicated here rather than shared:
/// both are a handful of lines and neither crate has a test-support
/// module to put a shared helper in.
async fn disconnected_client() -> async_nats::Client {
async_nats::ConnectOptions::new()
.retry_on_initial_connect()
.connect("127.0.0.1:1")
.await
.expect("retry_on_initial_connect returns without waiting for a real connection")
}
/// `state_with_roster` plus a `wanted` writer over a client that never
/// connected — the shape `set_agent_state`/`get_hive_wanted` see when
/// the queue is mid-reconnect.
async fn state_with_disconnected_wanted() -> super::AppState {
let (state, _sched) = state_with_roster();
super::AppState {
wanted: Some(std::sync::Arc::new(wanted::WantedWriter::new(
disconnected_client().await,
))),
..state
}
}
/// End to end through the real handler: `PUT .../state` against a
/// disconnected queue answers 503, not the 500 every other writer
/// error still gets.
#[tokio::test]
async fn set_agent_state_answers_503_on_a_disconnected_queue() {
let state = state_with_disconnected_wanted().await;
let result = super::set_agent_state(
axum::extract::State(state),
axum::extract::Path(("pr1ma".to_owned(), "atlas".to_owned())),
axum::Json(SetAgentStateRequest {
state: swarm_queue_client::wanted::AgentState::Paused,
}),
)
.await;
let problem = result.expect_err("a disconnected queue must not read as success");
assert_eq!(
problem.status,
Some(axum::http::StatusCode::SERVICE_UNAVAILABLE),
"{problem:?}"
);
}
/// Same fix, the read side: `GET .../wanted` against a disconnected
/// queue also answers 503.
#[tokio::test]
async fn get_hive_wanted_answers_503_on_a_disconnected_queue() {
let state = state_with_disconnected_wanted().await;
let result = super::get_hive_wanted(
axum::extract::State(state),
axum::extract::Path("pr1ma".to_owned()),
)
.await;
let problem = result.expect_err("a disconnected queue must not read as success");
assert_eq!(
problem.status,
Some(axum::http::StatusCode::SERVICE_UNAVAILABLE),
"{problem:?}"
);
}
/// Control for both tests above: a hive that is not in the roster is
/// still a plain 400, even with the same disconnected writer wired up —
/// so the 503 above is `ensure_connected` firing, not every error this
/// handler can produce collapsing to 503.
#[tokio::test]
async fn set_agent_state_still_answers_400_for_an_unknown_hive() {
let state = state_with_disconnected_wanted().await;
let result = super::set_agent_state(
axum::extract::State(state),
axum::extract::Path(("not-a-hive".to_owned(), "atlas".to_owned())),
axum::Json(SetAgentStateRequest {
state: swarm_queue_client::wanted::AgentState::Paused,
}),
)
.await;
let problem = result.expect_err("an unknown hive is still a 400");
assert_eq!(
problem.status,
Some(axum::http::StatusCode::BAD_REQUEST),
"{problem:?}"
);
}
}
#[cfg(test)]