hive-agent: present this agent's own queue credential, then fall back
When the per-agent secret `queue-identity.nix` fetched is present, the harness connects with `swarm-agent.<agent>.<secret>` as a static token and publishes on `$SWARM.term.<agent>` and `$SWARM.agent-state.<agent>`. When it is absent, or that first connect fails for any reason, a refusal from a responder that does not verify agent tokens included, it connects with the hive's shared OIDC client and publishes on the hive-scoped subjects as before. Which one it took is logged once per connect. `swarm_queue_client::connect_with_token` is the static-token connect: no retry on the initial attempt, so the caller sees the refusal and can fall back. Reconnects share the existing backoff, now a named function. Closes #4630
This commit is contained in:
parent
0c1fb44a4f
commit
727065c960
5 changed files with 287 additions and 76 deletions
|
|
@ -320,6 +320,15 @@ const TOKEN_REFRESH_SKEW: std::time::Duration = std::time::Duration::from_mins(2
|
|||
/// degrades every other client of it.
|
||||
const MAX_RECONNECT_DELAY: std::time::Duration = std::time::Duration::from_mins(1);
|
||||
|
||||
/// Exponential from 500ms, capped at [`MAX_RECONNECT_DELAY`].
|
||||
fn reconnect_delay(attempts: usize) -> std::time::Duration {
|
||||
let exp = u32::try_from(attempts.saturating_sub(1)).unwrap_or(u32::MAX);
|
||||
std::cmp::min(
|
||||
std::time::Duration::from_millis(500u64.saturating_mul(2u64.saturating_pow(exp.min(8)))),
|
||||
MAX_RECONNECT_DELAY,
|
||||
)
|
||||
}
|
||||
|
||||
/// Where the controller finds the queue and what it authenticates with.
|
||||
///
|
||||
/// Every field comes from an environment variable the NixOS module sets, the
|
||||
|
|
@ -742,15 +751,7 @@ pub async fn connect(cfg: QueueConfig) -> Result<async_nats::Client, Error> {
|
|||
// which turns an unreachable queue into a permanent 4s poll — and, before
|
||||
// the cache above, a permanent 4s token-request loop against authelia.
|
||||
// Exponential from 500ms so a momentary blip still reconnects promptly.
|
||||
.reconnect_delay_callback(|attempts| {
|
||||
let exp = u32::try_from(attempts.saturating_sub(1)).unwrap_or(u32::MAX);
|
||||
std::cmp::min(
|
||||
std::time::Duration::from_millis(
|
||||
500u64.saturating_mul(2u64.saturating_pow(exp.min(8))),
|
||||
),
|
||||
MAX_RECONNECT_DELAY,
|
||||
)
|
||||
})
|
||||
.reconnect_delay_callback(reconnect_delay)
|
||||
// The controller and the queue are separate units on (possibly)
|
||||
// separate hosts, and nothing orders them. Without this, a queue that
|
||||
// comes up one second later leaves the controller permanently
|
||||
|
|
@ -784,6 +785,29 @@ pub async fn connect(cfg: QueueConfig) -> Result<async_nats::Client, Error> {
|
|||
Ok(client)
|
||||
}
|
||||
|
||||
/// Connect presenting a fixed `token`, as an agent presents its own queue
|
||||
/// credential. Reconnects present the same token.
|
||||
///
|
||||
/// Unlike [`connect`], the first attempt must succeed: a refused or failed
|
||||
/// connect is returned rather than retried in the background, so the caller
|
||||
/// can fall back to another credential. `ca_file` is the queue's trust anchor,
|
||||
/// as [`QueueConfig::ca_file`].
|
||||
pub async fn connect_with_token(
|
||||
url: &str,
|
||||
ca_file: Option<&std::path::Path>,
|
||||
token: String,
|
||||
) -> Result<async_nats::Client, Error> {
|
||||
let mut options =
|
||||
async_nats::ConnectOptions::with_token(token).reconnect_delay_callback(reconnect_delay);
|
||||
if let Some(path) = ca_file {
|
||||
options = options.add_root_certificates(path.to_path_buf());
|
||||
}
|
||||
options.connect(url).await.map_err(|source| Error::Connect {
|
||||
url: url.to_owned(),
|
||||
source,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
|
|
|||
Loading…
Reference in a new issue