refactor(swarm-queue-client): extract the queue connect into a shared crate
A hive publishing its own status needs the same connect the controller already has - mint an authelia token, present it at CONNECT for the callout responder, let async-nats re-run the callback per attempt. Only the use differs: the controller reads, a hive writes. Copying it would put credential handling in two places, and a token-refresh fix would then have to be found twice. That is the same reasoning that already put hive-sock-client in its own crate rather than in each daemon that speaks to a unix socket. `from_env` takes a prefix rather than hardcoding SWARM_CONTROLLER_*: the variables belong to the consuming unit, since a NixOS module sets them alongside its other options. What is shared is the RULE - all four together or none at all - not the spelling. The half-set case gains a test, because it is the case the rule exists for and it previously had none. No jetstream/kv feature on the crate: it ends at a connected client, and what a consumer does with it should be visible in that consumer's own Cargo.toml. Behaviour-preserving, and proven that way rather than by inspection: the full behavioural gate (real nats-server, credential rotation, mutation) is 20/0 unchanged, and the controller's own tests still pass.
This commit is contained in:
parent
bc594a36ef
commit
a9603214c2
7 changed files with 166 additions and 33 deletions
14
Cargo.lock
generated
14
Cargo.lock
generated
|
|
@ -4564,6 +4564,7 @@ dependencies = [
|
||||||
"reqwest 0.13.1",
|
"reqwest 0.13.1",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
|
"swarm-queue-client",
|
||||||
"tokio",
|
"tokio",
|
||||||
"tracing",
|
"tracing",
|
||||||
"tracing-subscriber",
|
"tracing-subscriber",
|
||||||
|
|
@ -4591,6 +4592,19 @@ dependencies = [
|
||||||
"tracing-subscriber",
|
"tracing-subscriber",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "swarm-queue-client"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"anyhow",
|
||||||
|
"async-nats",
|
||||||
|
"reqwest 0.13.1",
|
||||||
|
"serde",
|
||||||
|
"serde_json",
|
||||||
|
"tokio",
|
||||||
|
"tracing",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "swarmctl"
|
name = "swarmctl"
|
||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
|
|
|
||||||
|
|
@ -23,6 +23,7 @@ members = [
|
||||||
"hivectl",
|
"hivectl",
|
||||||
"swarm-controller",
|
"swarm-controller",
|
||||||
"swarm-nats-auth",
|
"swarm-nats-auth",
|
||||||
|
"swarm-queue-client",
|
||||||
"swarmctl",
|
"swarmctl",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
@ -84,6 +85,7 @@ hive-host-sock = { path = "hive-host-sock" }
|
||||||
hive-priv-sock = { path = "hive-priv-sock" }
|
hive-priv-sock = { path = "hive-priv-sock" }
|
||||||
hive-sock-client = { path = "hive-sock-client" }
|
hive-sock-client = { path = "hive-sock-client" }
|
||||||
hive-types = { path = "hive-types" }
|
hive-types = { path = "hive-types" }
|
||||||
|
swarm-queue-client = { path = "swarm-queue-client" }
|
||||||
thiserror = "2"
|
thiserror = "2"
|
||||||
tower-http = { version = "0.7", features = ["fs"] }
|
tower-http = { version = "0.7", features = ["fs"] }
|
||||||
uuid = { version = "1", features = ["v4"] }
|
uuid = { version = "1", features = ["v4"] }
|
||||||
|
|
|
||||||
|
|
@ -21,6 +21,11 @@ futures-util.workspace = true
|
||||||
reqwest.workspace = true
|
reqwest.workspace = true
|
||||||
serde.workspace = true
|
serde.workspace = true
|
||||||
serde_json.workspace = true
|
serde_json.workspace = true
|
||||||
|
# The queue connect (token mint + auth callback + reconnect) is shared with
|
||||||
|
# every other participant - a hive publishing its own status runs the same
|
||||||
|
# code with a different client id. Two copies of credential handling is one
|
||||||
|
# token-refresh fix that has to be found twice.
|
||||||
|
swarm-queue-client.workspace = true
|
||||||
tokio.workspace = true
|
tokio.workspace = true
|
||||||
tracing.workspace = true
|
tracing.workspace = true
|
||||||
tracing-subscriber.workspace = true
|
tracing-subscriber.workspace = true
|
||||||
|
|
|
||||||
|
|
@ -31,7 +31,6 @@ use serde::{Deserialize, Serialize};
|
||||||
use utoipa::{OpenApi, ToSchema};
|
use utoipa::{OpenApi, ToSchema};
|
||||||
use utoipa_axum::{router::OpenApiRouter, routes};
|
use utoipa_axum::{router::OpenApiRouter, routes};
|
||||||
|
|
||||||
mod queue;
|
|
||||||
mod status;
|
mod status;
|
||||||
|
|
||||||
/// Where the daemon binds, overridable via `SWARM_CONTROLLER_SOCKET`.
|
/// Where the daemon binds, overridable via `SWARM_CONTROLLER_SOCKET`.
|
||||||
|
|
@ -298,12 +297,12 @@ async fn main() -> Result<()> {
|
||||||
// a half-set environment — `QueueConfig::from_env` refuses that, because
|
// a half-set environment — `QueueConfig::from_env` refuses that, because
|
||||||
// silently behaving like an unconfigured host is how every hive ends up
|
// silently behaving like an unconfigured host is how every hive ends up
|
||||||
// reading `never_reported` with nothing to point at.
|
// reading `never_reported` with nothing to point at.
|
||||||
let status = match queue::QueueConfig::from_env()? {
|
let status = match swarm_queue_client::QueueConfig::from_env("SWARM_CONTROLLER")? {
|
||||||
None => {
|
None => {
|
||||||
tracing::info!("no swarm queue configured; status aggregation is off");
|
tracing::info!("no swarm queue configured; status aggregation is off");
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
Some(cfg) => match queue::connect(cfg).await {
|
Some(cfg) => match swarm_queue_client::connect(cfg).await {
|
||||||
Ok(client) => {
|
Ok(client) => {
|
||||||
// NOT "connected": `retry_on_initial_connect` returns a client
|
// NOT "connected": `retry_on_initial_connect` returns a client
|
||||||
// before any connection has been established, so claiming a
|
// before any connection has been established, so claiming a
|
||||||
|
|
|
||||||
21
swarm-queue-client/Cargo.toml
Normal file
21
swarm-queue-client/Cargo.toml
Normal file
|
|
@ -0,0 +1,21 @@
|
||||||
|
[package]
|
||||||
|
name = "swarm-queue-client"
|
||||||
|
version.workspace = true
|
||||||
|
readme = "README.md"
|
||||||
|
edition.workspace = true
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
anyhow.workspace = true
|
||||||
|
# No `kv`/`jetstream` feature here on purpose: this crate's job ends at a
|
||||||
|
# connected client. What a consumer does with it - KV for the controller and
|
||||||
|
# the hive, plain messaging for anything later - is the consumer's business,
|
||||||
|
# and its Cargo.toml is where that requirement should be visible.
|
||||||
|
async-nats.workspace = true
|
||||||
|
reqwest.workspace = true
|
||||||
|
serde.workspace = true
|
||||||
|
serde_json.workspace = true
|
||||||
|
tokio.workspace = true
|
||||||
|
tracing.workspace = true
|
||||||
|
|
||||||
|
[lints]
|
||||||
|
workspace = true
|
||||||
55
swarm-queue-client/README.md
Normal file
55
swarm-queue-client/README.md
Normal file
|
|
@ -0,0 +1,55 @@
|
||||||
|
# swarm-queue-client
|
||||||
|
|
||||||
|
Connecting to the swarm message queue as an authenticated client. Shared by
|
||||||
|
every process that participates: the swarm controller reads hive status out of
|
||||||
|
the queue, a hive publishes its own status into it.
|
||||||
|
|
||||||
|
## Why a crate and not a module per binary
|
||||||
|
|
||||||
|
The *connect* is identical for every participant — mint an authelia token,
|
||||||
|
present it at CONNECT for the `auth_callout` responder to introspect, let
|
||||||
|
`async-nats` re-run the callback on each connection attempt. Only the **use**
|
||||||
|
differs.
|
||||||
|
|
||||||
|
Two copies of that would be two copies of credential handling, and a
|
||||||
|
token-refresh fix would have to be found twice. The same reasoning already put
|
||||||
|
`hive-sock-client` in its own crate rather than in each daemon that speaks to a
|
||||||
|
unix socket.
|
||||||
|
|
||||||
|
## The two properties that constrain the code
|
||||||
|
|
||||||
|
**A token expires.** Authelia issues `client_credentials` access tokens with
|
||||||
|
`expires_in: 3599`. Authentication happens at CONNECT, so a long-lived
|
||||||
|
connection is fine — but a *reconnect* an hour later needs a token minted an
|
||||||
|
hour later.
|
||||||
|
|
||||||
|
**The refresh therefore lives in the auth callback, not in a timer.**
|
||||||
|
`async-nats` invokes it per connection attempt, so there is no window in which
|
||||||
|
the client holds a token it minted for a previous connection. The alternative —
|
||||||
|
mint once, own the reconnect loop — fails in the way this subsystem exists to
|
||||||
|
prevent: the process keeps serving while its data quietly stops moving, and
|
||||||
|
nothing says so until someone reads a dashboard.
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
`QueueConfig::from_env(prefix)` reads `<prefix>_NATS_URL`,
|
||||||
|
`<prefix>_OIDC_TOKEN_ENDPOINT`, `<prefix>_OIDC_CLIENT_ID` and
|
||||||
|
`<prefix>_OIDC_CLIENT_SECRET_FILE`.
|
||||||
|
|
||||||
|
The prefix is a parameter because the variables belong to the consuming unit —
|
||||||
|
a NixOS module sets them alongside its other options. What is shared is the
|
||||||
|
rule, not the spelling: **all four together or none at all.** A half-set
|
||||||
|
environment is a hard error, because the failure it would otherwise produce is
|
||||||
|
the expensive kind — the process comes up "fine", never connects, and the data
|
||||||
|
it was supposed to move silently stops.
|
||||||
|
|
||||||
|
The client secret is a **path, not a value**: putting it in the environment
|
||||||
|
would publish it to anything that can read `/proc/<pid>/environ`. It is read
|
||||||
|
per token request rather than cached, so a rotation the operator believes took
|
||||||
|
effect actually did.
|
||||||
|
|
||||||
|
## What this crate does not do
|
||||||
|
|
||||||
|
It ends at a connected client. No `jetstream`/`kv` feature is enabled here —
|
||||||
|
what a consumer does with the connection is its own business, and its
|
||||||
|
`Cargo.toml` is where that requirement should be visible.
|
||||||
|
|
@ -1,10 +1,17 @@
|
||||||
//! The controller's client end of the swarm message queue.
|
//! Connecting to the swarm message queue as an authenticated client.
|
||||||
|
//!
|
||||||
|
//! Shared by every process that needs the queue — the swarm controller reads
|
||||||
|
//! hive status out of it, a hive publishes its own status into it — because
|
||||||
|
//! the *connect* is identical for all of them and only the use differs.
|
||||||
|
//! Duplicating it per binary would put credential handling in two places, and
|
||||||
|
//! a token-refresh fix would then have to be found twice.
|
||||||
//!
|
//!
|
||||||
//! The queue admits every non-responder client through `auth_callout`: a
|
//! The queue admits every non-responder client through `auth_callout`: a
|
||||||
//! client presents a token at CONNECT, the callout responder introspects it
|
//! client presents a token at CONNECT, the callout responder introspects it
|
||||||
//! against authelia and mints a user JWT if it is good. So the controller is
|
//! against authelia and mints a user JWT if it is good. So each participant is
|
||||||
//! an ordinary client and needs an identity of its own — it is not a hive, and
|
//! an ordinary client that needs an identity of its own — the controller is
|
||||||
//! the per-hive clients issued from the roster are not its to use.
|
//! not a hive, and the per-hive clients issued from the roster are not its to
|
||||||
|
//! use.
|
||||||
//!
|
//!
|
||||||
//! Two things about that shape drive everything here:
|
//! Two things about that shape drive everything here:
|
||||||
//!
|
//!
|
||||||
|
|
@ -54,19 +61,26 @@ pub struct QueueConfig {
|
||||||
}
|
}
|
||||||
|
|
||||||
impl QueueConfig {
|
impl QueueConfig {
|
||||||
/// Read the config from the environment, or `None` when the queue was not
|
/// Read the config from `<prefix>_NATS_URL`, `<prefix>_OIDC_TOKEN_ENDPOINT`,
|
||||||
/// wired up for this deployment.
|
/// `<prefix>_OIDC_CLIENT_ID` and `<prefix>_OIDC_CLIENT_SECRET_FILE`, or
|
||||||
|
/// `None` when the queue was not wired up for this deployment.
|
||||||
///
|
///
|
||||||
/// `None` rather than an error on purpose: the controller serves its HTTP
|
/// The prefix is a parameter rather than a constant because the variables
|
||||||
/// surface on hosts where the queue is not enabled, and refusing to start
|
/// belong to the *consuming unit* — a NixOS module sets them alongside its
|
||||||
/// there would trade a missing feature for a dead daemon. What must NOT
|
/// other options, and two daemons sharing one name would be a worse
|
||||||
|
/// coupling than passing four characters. What is shared is the RULE
|
||||||
|
/// below, not the spelling.
|
||||||
|
///
|
||||||
|
/// `None` rather than an error on purpose: a daemon serves its other
|
||||||
|
/// surfaces on hosts where the queue is not enabled, and refusing to start
|
||||||
|
/// there would trade a missing feature for a dead process. What must NOT
|
||||||
/// happen is a *half* configuration silently behaving like an absent one —
|
/// happen is a *half* configuration silently behaving like an absent one —
|
||||||
/// hence the explicit partial check below.
|
/// hence the explicit partial check below.
|
||||||
pub fn from_env() -> Result<Option<Self>> {
|
pub fn from_env(prefix: &str) -> Result<Option<Self>> {
|
||||||
let url = std::env::var("SWARM_CONTROLLER_NATS_URL").ok();
|
let url = std::env::var(format!("{prefix}_NATS_URL")).ok();
|
||||||
let token_endpoint = std::env::var("SWARM_CONTROLLER_OIDC_TOKEN_ENDPOINT").ok();
|
let token_endpoint = std::env::var(format!("{prefix}_OIDC_TOKEN_ENDPOINT")).ok();
|
||||||
let client_id = std::env::var("SWARM_CONTROLLER_OIDC_CLIENT_ID").ok();
|
let client_id = std::env::var(format!("{prefix}_OIDC_CLIENT_ID")).ok();
|
||||||
let secret = std::env::var("SWARM_CONTROLLER_OIDC_CLIENT_SECRET_FILE").ok();
|
let secret = std::env::var(format!("{prefix}_OIDC_CLIENT_SECRET_FILE")).ok();
|
||||||
|
|
||||||
match (url, token_endpoint, client_id, secret) {
|
match (url, token_endpoint, client_id, secret) {
|
||||||
(None, None, None, None) => Ok(None),
|
(None, None, None, None) => Ok(None),
|
||||||
|
|
@ -77,13 +91,14 @@ impl QueueConfig {
|
||||||
client_secret_file: PathBuf::from(secret),
|
client_secret_file: PathBuf::from(secret),
|
||||||
})),
|
})),
|
||||||
// A partially-set environment is a deployment bug, and the failure
|
// A partially-set environment is a deployment bug, and the failure
|
||||||
// it would otherwise produce is the expensive kind: the controller
|
// it would otherwise produce is the expensive kind: the process
|
||||||
// comes up "fine", never connects, and every hive reads as having
|
// comes up "fine", never connects, and the data it was supposed to
|
||||||
// never reported. Naming the missing variables costs one line.
|
// move silently stops moving. Naming the variables costs one line.
|
||||||
_ => bail!(
|
_ => bail!(
|
||||||
"swarm queue is half-configured: SWARM_CONTROLLER_NATS_URL, \
|
"swarm queue is half-configured: {prefix}_NATS_URL, \
|
||||||
_OIDC_TOKEN_ENDPOINT, _OIDC_CLIENT_ID and \
|
{prefix}_OIDC_TOKEN_ENDPOINT, {prefix}_OIDC_CLIENT_ID and \
|
||||||
_OIDC_CLIENT_SECRET_FILE must be set together or not at all"
|
{prefix}_OIDC_CLIENT_SECRET_FILE must be set together or not \
|
||||||
|
at all"
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -183,24 +198,46 @@ mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
/// The all-unset case is the common one — most hosts do not run the queue.
|
/// The all-unset case is the common one — most hosts do not run the queue.
|
||||||
|
///
|
||||||
|
/// Uses a prefix no deployment sets, so it cannot pass vacuously by
|
||||||
|
/// running inside a configured environment (the guard below covers the
|
||||||
|
/// same ground, and both are cheap).
|
||||||
#[test]
|
#[test]
|
||||||
fn an_absent_environment_is_not_an_error() {
|
fn an_absent_environment_is_not_an_error() {
|
||||||
// Guard: this test would pass vacuously inside a configured
|
|
||||||
// environment, so it asserts the variables really are unset first.
|
|
||||||
for k in [
|
for k in [
|
||||||
"SWARM_CONTROLLER_NATS_URL",
|
"SWARM_QUEUE_TEST_NATS_URL",
|
||||||
"SWARM_CONTROLLER_OIDC_TOKEN_ENDPOINT",
|
"SWARM_QUEUE_TEST_OIDC_TOKEN_ENDPOINT",
|
||||||
"SWARM_CONTROLLER_OIDC_CLIENT_ID",
|
"SWARM_QUEUE_TEST_OIDC_CLIENT_ID",
|
||||||
"SWARM_CONTROLLER_OIDC_CLIENT_SECRET_FILE",
|
"SWARM_QUEUE_TEST_OIDC_CLIENT_SECRET_FILE",
|
||||||
] {
|
] {
|
||||||
if std::env::var(k).is_ok() {
|
assert!(std::env::var(k).is_err(), "{k} must be unset for this test");
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
assert!(
|
assert!(
|
||||||
QueueConfig::from_env()
|
QueueConfig::from_env("SWARM_QUEUE_TEST")
|
||||||
.expect("absent is not an error")
|
.expect("absent is not an error")
|
||||||
.is_none()
|
.is_none()
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The half-set case is the one the rule exists for: a deployment bug that
|
||||||
|
/// would otherwise look exactly like "no queue configured".
|
||||||
|
///
|
||||||
|
/// SAFETY: single-threaded mutation of a process env var under a prefix no
|
||||||
|
/// other test or deployment uses; removed before returning.
|
||||||
|
#[test]
|
||||||
|
fn a_half_set_environment_is_a_hard_error() {
|
||||||
|
unsafe {
|
||||||
|
std::env::set_var("SWARM_QUEUE_HALF_NATS_URL", "nats://127.0.0.1:4222");
|
||||||
|
}
|
||||||
|
let err = QueueConfig::from_env("SWARM_QUEUE_HALF")
|
||||||
|
.expect_err("a partial set must not read as absent");
|
||||||
|
let msg = format!("{err}");
|
||||||
|
assert!(
|
||||||
|
msg.contains("SWARM_QUEUE_HALF_OIDC_CLIENT_ID"),
|
||||||
|
"the error must name the missing variables, got: {msg}"
|
||||||
|
);
|
||||||
|
unsafe {
|
||||||
|
std::env::remove_var("SWARM_QUEUE_HALF_NATS_URL");
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
Loading…
Reference in a new issue