tidaldb/tidal-server/tests/cluster_region.rs
jx12n 225751d34d feat(m11): WAL-as-stream replication + perf floor (m11p1+m11p2)
m11p1 — decoupled ack/ship path: staged writes (seqno+WAL+relay-push,
microseconds) separate from group-commit fsync; ShipQueue batches+windows
outbound segments; receiver coalesces inbound chunks before applying.
Adds first tidaldb_cluster_* metrics.

m11p2 — leader WAL is now THE replicated log: fsynced batches feed a
bounded WalShipFeed and ship byte-identical to followers; WAL seqnos
survive restarts (relay-reset hazard gone). Item metadata and embeddings
journal kind-1/2 blob records on the same stream as signals; the m8p10
HTTP broadcast is deleted. StreamSegments catch-up is follower-pulled via
server-streaming RPC, triggered on gap detection, follower boot, and
leader heal nudge. Promote carries a stream baseline so peers skip
pre-stream history.
2026-06-11 09:10:06 -06:00

924 lines
34 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

//! m8p10 in-process multi-process-cluster tests.
//!
//! Each test builds TWO `RegionClusterState`s in ONE test process — distinct
//! topologies pointing at each other's REAL loopback gRPC addresses — and drives
//! them over real HTTP (axum on loopback) + real `GrpcTransport` replication.
//! Unlike `cluster_e2e.rs` (tier-3, spawns OS processes), these run in the
//! default test build with no OS processes, exactly like `cluster_grpc.rs`.
//!
//! They prove the multi-process region node: convergence over real loopback
//! gRPC with decay parity, the typed `NotLeader` rejection, partition→heal with
//! idempotent re-heal, and that promote flips roles with the ALWAYS-ON receiver
//! (the demoted node applies the new leader's ships).
#![allow(
clippy::unwrap_used,
clippy::missing_panics_doc,
clippy::too_many_lines,
clippy::doc_markdown
)]
use std::{
net::{SocketAddr, TcpListener},
sync::Arc,
time::{Duration, Instant},
};
use tidal_server::cluster::{
RegionClusterState, RegionSpec, ReplicationSpec, TimeoutsSpec, TopologySpec, WalSpec,
build_region_router,
};
use tidaldb::schema::{DecaySpec, EntityKind, Schema, SchemaBuilder, Window};
/// A single-`view`-signal schema with a `hide` hard-negative signal (so the
/// `/hardnegs` route's `signal_with_context("hide", …)` resolves).
fn region_schema() -> Schema {
let mut builder = SchemaBuilder::new();
let _ = builder
.signal(
"view",
EntityKind::Item,
DecaySpec::Exponential {
half_life: Duration::from_secs(7 * 24 * 3600),
},
)
.windows(&[Window::OneHour])
.velocity(false)
.add();
let _ = builder
.signal("hide", EntityKind::Item, DecaySpec::Permanent)
.velocity(false)
.add();
builder.build().unwrap()
}
/// Reserve a free loopback port and return its address.
fn free_addr() -> SocketAddr {
TcpListener::bind("127.0.0.1:0")
.unwrap()
.local_addr()
.unwrap()
}
/// A pair of fully-declared (grpc + http) region specs that point at each
/// other. `leader` is the first region's name.
struct Pair {
leader_name: String,
follower_name: String,
leader_grpc: SocketAddr,
follower_grpc: SocketAddr,
leader_http: SocketAddr,
follower_http: SocketAddr,
}
impl Pair {
fn new() -> Self {
Self {
leader_name: "us-east".into(),
follower_name: "eu-west".into(),
leader_grpc: free_addr(),
follower_grpc: free_addr(),
leader_http: free_addr(),
follower_http: free_addr(),
}
}
/// The shared topology both processes parse (same declaration order ⇒ same
/// RegionIds in both). HTTP addrs are the in-test axum binds.
fn topology(&self) -> TopologySpec {
TopologySpec {
regions: vec![
RegionSpec {
name: self.leader_name.clone(),
grpc_addr: Some(self.leader_grpc.to_string()),
http_addr: Some(self.leader_http.to_string()),
grpc_tls: None,
metrics_addr: None,
},
RegionSpec {
name: self.follower_name.clone(),
grpc_addr: Some(self.follower_grpc.to_string()),
http_addr: Some(self.follower_http.to_string()),
grpc_tls: None,
metrics_addr: None,
},
],
leader: self.leader_name.clone(),
write_workers: None,
timeouts: TimeoutsSpec::default(),
replication: ReplicationSpec::default(),
wal: WalSpec::default(),
}
}
}
/// A node's persistent data dir (m11p2: the durable WAL is the replication
/// stream, so multi-process cluster mode requires one).
///
/// DECLARE THE DIR BEFORE ANYTHING THAT CAN HOLD THE NODE (including the
/// tokio runtime): locals — and bindings within one tuple pattern — drop in
/// reverse declaration order, and a dir deleted while its node still runs
/// wedges fjall's flush worker on `NotFound` (the sealed memtable then never
/// drains and `rotate_memtable_and_wait` polls forever — observed as a
/// permanently hung test on the panic-unwind path).
fn region_dir() -> tempfile::TempDir {
tempfile::tempdir().expect("create per-region data dir")
}
/// Build one region node off the reactor (GrpcTransport::new blocks on its own
/// runtime, so it must run on a plain thread). `dir` is the node's data dir;
/// see [`region_dir`] for the declaration-order contract.
fn build_region(
topology: TopologySpec,
region: &str,
dir: &tempfile::TempDir,
) -> RegionClusterState {
let region = region.to_string();
let data_dir = dir.path().to_path_buf();
std::thread::spawn(move || {
RegionClusterState::new(
&topology,
&region,
region_schema(),
Vec::new(),
Some(data_dir),
0,
)
})
.join()
.unwrap()
.expect("region node builds with real gRPC transport")
}
/// Serve `router` on `addr` using `rt`; returns once the listener is bound.
fn serve(rt: &tokio::runtime::Runtime, router: axum::Router, addr: SocketAddr) {
let listener = rt
.block_on(tokio::net::TcpListener::bind(addr))
.unwrap_or_else(|e| panic!("bind {addr}: {e}"));
rt.spawn(async move {
let _ = axum::serve(listener, router).await;
});
}
/// Poll `GET /cluster/status/local` on `base` until `pred(applied, lag)` holds
/// or the deadline elapses.
fn poll_status(
client: &reqwest::blocking::Client,
base: &str,
pred: impl Fn(u64, u64) -> bool,
) -> serde_json::Value {
let deadline = Instant::now() + Duration::from_secs(5);
loop {
let status: serde_json::Value = client
.get(format!("{base}/cluster/status/local"))
.send()
.unwrap()
.json()
.unwrap();
let applied = status["applied_events"].as_u64().unwrap_or(0);
let lag = status["lag_events"].as_u64().unwrap_or(u64::MAX);
if pred(applied, lag) {
return status;
}
assert!(
Instant::now() <= deadline,
"status predicate not met within 5s: {status}"
);
std::thread::sleep(Duration::from_millis(20));
}
}
/// Read entity `entity`'s trending feed score on `base` (0.0 if absent).
fn feed_score(client: &reqwest::blocking::Client, base: &str, entity: u64) -> f64 {
let feed: serde_json::Value = client
.get(format!("{base}/feed?profile=trending&limit=10"))
.send()
.unwrap()
.json()
.unwrap();
feed["items"]
.as_array()
.unwrap()
.iter()
.find(|it| it["entity_id"].as_u64() == Some(entity))
.and_then(|it| it["score"].as_f64())
.unwrap_or(0.0)
}
/// Two region nodes converge over real loopback gRPC; the follower's feed scores
/// match the leader's to 1e-6 (decay parity).
#[test]
fn region_node_replicates_over_grpc() {
let pair = Pair::new();
let leader_dir = region_dir();
let leader = build_region(pair.topology(), &pair.leader_name, &leader_dir);
let follower_dir = region_dir();
let follower = build_region(pair.topology(), &pair.follower_name, &follower_dir);
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(leader), None),
pair.leader_http,
);
serve(
&rt,
build_region_router(Arc::new(follower), None),
pair.follower_http,
);
let client = reqwest::blocking::Client::new();
let leader_base = format!("http://{}", pair.leader_http);
let follower_base = format!("http://{}", pair.follower_http);
// Broadcast items to BOTH nodes (items are not WAL-replicated in this task),
// then write signals on the leader (replicated to the follower over gRPC).
for i in 1..=8u64 {
for base in [&leader_base, &follower_base] {
let resp = client
.post(format!("{base}/items"))
.json(&serde_json::json!({
"entity_id": i,
"metadata": { "title": format!("item {i}") }
}))
.send()
.unwrap();
assert!(resp.status().is_success(), "POST /items: {}", resp.status());
}
let resp = client
.post(format!("{leader_base}/signals"))
.json(&serde_json::json!({ "entity_id": i, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert!(
resp.status().is_success(),
"POST /signals on leader: {}",
resp.status()
);
}
// Follower converges: applied reaches 8 and lag returns to 0.
poll_status(&client, &follower_base, |applied, lag| {
applied >= 8 && lag == 0
});
// Decay parity: the leader and follower feeds rank the same items with
// scores equal to 1e-6 (the follower replayed the exact same WAL events).
let leader_feed: serde_json::Value = client
.get(format!("{leader_base}/feed?profile=trending&limit=8"))
.send()
.unwrap()
.json()
.unwrap();
let follower_feed: serde_json::Value = client
.get(format!("{follower_base}/feed?profile=trending&limit=8"))
.send()
.unwrap()
.json()
.unwrap();
let l_items = leader_feed["items"].as_array().unwrap();
let f_items = follower_feed["items"].as_array().unwrap();
assert!(!f_items.is_empty(), "follower must serve replicated items");
assert_eq!(l_items.len(), f_items.len(), "same number of ranked items");
let mut l_scores: std::collections::HashMap<u64, f64> = std::collections::HashMap::new();
for it in l_items {
l_scores.insert(
it["entity_id"].as_u64().unwrap(),
it["score"].as_f64().unwrap(),
);
}
for it in f_items {
let id = it["entity_id"].as_u64().unwrap();
let f = it["score"].as_f64().unwrap();
let l = *l_scores.get(&id).expect("follower item also on leader");
assert!((l - f).abs() < 1e-6, "entity {id}: leader={l} follower={f}");
}
rt.shutdown_timeout(Duration::from_secs(2));
}
/// A write to the FOLLOWER (a non-leader node) is FORWARDED to the leader; with
/// the leader process not running, the forward fails and degrades to a 503 whose
/// body names the (unreachable) leader — the task-03 leader-unreachable contract.
#[test]
fn region_node_rejects_writes_when_not_leader() {
let pair = Pair::new();
let follower_dir = region_dir();
let follower = build_region(pair.topology(), &pair.follower_name, &follower_dir);
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(follower), None),
pair.follower_http,
);
let client = reqwest::blocking::Client::new();
let follower_base = format!("http://{}", pair.follower_http);
let resp = client
.post(format!("{follower_base}/signals"))
.json(&serde_json::json!({ "entity_id": 1, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert_eq!(
resp.status().as_u16(),
503,
"a non-leader write must be 503 NotLeader"
);
let body: serde_json::Value = resp.json().unwrap();
assert_eq!(
body["leader"].as_str(),
Some(pair.leader_name.as_str()),
"the 503 body must name the leader: {body}"
);
rt.shutdown_timeout(Duration::from_secs(2));
}
/// Partition the follower → leader ships are skipped (follower lags) → heal →
/// the leader redelivers over gRPC and the follower converges. A SECOND heal is
/// a no-op (idempotent): the follower's scores are unchanged.
#[test]
fn region_node_partition_heal() {
let pair = Pair::new();
let leader_dir = region_dir();
let leader = build_region(pair.topology(), &pair.leader_name, &leader_dir);
let follower_dir = region_dir();
let follower = build_region(pair.topology(), &pair.follower_name, &follower_dir);
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(leader), None),
pair.leader_http,
);
serve(
&rt,
build_region_router(Arc::new(follower), None),
pair.follower_http,
);
let client = reqwest::blocking::Client::new();
let leader_base = format!("http://{}", pair.leader_http);
let follower_base = format!("http://{}", pair.follower_http);
let post_signal = |entity: u64| {
let resp = client
.post(format!("{leader_base}/signals"))
.json(&serde_json::json!({ "entity_id": entity, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert!(
resp.status().is_success(),
"leader signal: {}",
resp.status()
);
};
let post_item = |entity: u64| {
let resp = client
.post(format!("{leader_base}/items"))
.json(&serde_json::json!({ "entity_id": entity, "metadata": { "t": entity.to_string() } }))
.send()
.unwrap();
assert!(resp.status().is_success(), "leader item: {}", resp.status());
};
// Create items 1..=5 on the LEADER up front, so heal can backfill the ones the
// follower misses while partitioned (BUG 3). Write 2 signals, let the follower
// catch up.
for e in 1..=5u64 {
post_item(e);
}
post_signal(1);
post_signal(2);
// m11p2: items ride the WAL too — 5 item records + 2 signals = 7 seqnos.
poll_status(&client, &follower_base, |applied, _| applied >= 7);
// Partition the follower from the leader, then write 3 more.
let resp = client
.post(format!("{leader_base}/cluster/partition"))
.json(&serde_json::json!({ "region": pair.follower_name }))
.send()
.unwrap();
assert!(resp.status().is_success(), "partition: {}", resp.status());
for e in 3..=5u64 {
post_signal(e);
}
// While partitioned the follower is STUCK at applied=7: the leader's
// ships were skipped, so no segments arrive. (Its locally-reported lag stays
// 0 because a follower cannot observe the leader's progress while
// partitioned — cross-node lag aggregation is task 03; the load-bearing
// proof here is that `applied` does NOT advance past 2.)
let lagging = poll_status(&client, &follower_base, |applied, _| applied == 7);
assert_eq!(lagging["applied_events"].as_u64(), Some(7));
// Give the leader a beat to (not) ship — applied must remain 7, proving the
// partition truly skips the eager ships rather than racing convergence.
std::thread::sleep(Duration::from_millis(200));
let still: serde_json::Value = client
.get(format!("{follower_base}/cluster/status/local"))
.send()
.unwrap()
.json()
.unwrap();
assert_eq!(
still["applied_events"].as_u64(),
Some(7),
"partitioned follower must NOT receive the leader's post-partition writes"
);
// Heal: the leader redelivers the missed segments over gRPC.
let heal = |base: &str| {
let resp = client
.post(format!("{base}/cluster/heal"))
.json(&serde_json::json!({ "region": pair.follower_name }))
.send()
.unwrap();
assert!(resp.status().is_success(), "heal: {}", resp.status());
};
heal(&leader_base);
poll_status(&client, &follower_base, |applied, lag| {
applied >= 10 && lag == 0
});
// Capture the follower's converged decay for entity 5. Heal redelivered the
// signal AND backfilled the item (BUG 3), so the follower can rank entity 5
// WITHOUT the test posting anything. Poll until the score stabilizes (two equal
// consecutive non-zero reads) so the capture does not race the async item
// index; after it settles the score is stable, so re-heal idempotence is exact.
let score_of = |entity: u64| -> f64 {
let deadline = Instant::now() + Duration::from_secs(10);
let mut last = -1.0f64;
loop {
let s = feed_score(&client, &follower_base, entity);
if s > 0.0 && (s - last).abs() < 1e-12 {
return s;
}
last = s;
assert!(
Instant::now() <= deadline,
"heal must deliver signal+item for entity {entity} so the follower ranks it"
);
std::thread::sleep(Duration::from_millis(50));
}
};
let before = score_of(5);
// Second heal is idempotent: re-ships nothing new (follower already applied
// through seq 10) and re-broadcasts the same items, so the score is unchanged.
heal(&leader_base);
poll_status(&client, &follower_base, |applied, lag| {
applied == 10 && lag == 0
});
let after = score_of(5);
assert!(
(before - after).abs() < 1e-9,
"idempotent re-heal must not change scores: before={before} after={after}"
);
rt.shutdown_timeout(Duration::from_secs(2));
}
/// Promote flips the leadership view: the OLD leader now rejects writes
/// (NotLeader), the NEW leader accepts and ships, and the always-on receiver on
/// the demoted node applies the new leader's segments.
#[test]
fn region_node_promote_local() {
let pair = Pair::new();
let leader_dir = region_dir();
let leader = build_region(pair.topology(), &pair.leader_name, &leader_dir);
let follower_dir = region_dir();
let follower = build_region(pair.topology(), &pair.follower_name, &follower_dir);
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(leader), None),
pair.leader_http,
);
serve(
&rt,
build_region_router(Arc::new(follower), None),
pair.follower_http,
);
let client = reqwest::blocking::Client::new();
let leader_base = format!("http://{}", pair.leader_http);
let follower_base = format!("http://{}", pair.follower_http);
// Promote the FOLLOWER to leader on BOTH nodes (each node holds its own view).
for base in [&leader_base, &follower_base] {
let resp = client
.post(format!("{base}/cluster/promote"))
.json(&serde_json::json!({ "region": pair.follower_name }))
.send()
.unwrap();
assert!(resp.status().is_success(), "promote: {}", resp.status());
}
// The OLD leader (us-east), now demoted, transparently FORWARDS a write to the
// new leader (eu-west) and relays its 204 (task 03 replaces the standalone
// NotLeader 503 with leader forwarding). Item 1 must exist on eu-west to be
// rankable, but the forwarded signal itself proves the forward path.
let _ = client
.post(format!("{follower_base}/items"))
.json(&serde_json::json!({ "entity_id": 1, "metadata": {} }))
.send()
.unwrap();
let resp = client
.post(format!("{leader_base}/signals"))
.json(&serde_json::json!({ "entity_id": 1, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert_eq!(
resp.status().as_u16(),
204,
"demoted node must FORWARD the write to the new leader (204), not 503: {}",
resp.status()
);
// The NEW leader (eu-west) accepts and ships to the demoted node (us-east),
// whose ALWAYS-ON receiver applies the segments. Broadcast items to both.
for e in 1..=4u64 {
for base in [&leader_base, &follower_base] {
let _ = client
.post(format!("{base}/items"))
.json(&serde_json::json!({ "entity_id": e, "metadata": {} }))
.send()
.unwrap();
}
let resp = client
.post(format!("{follower_base}/signals"))
.json(&serde_json::json!({ "entity_id": e, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert!(
resp.status().is_success(),
"new-leader signal: {}",
resp.status()
);
}
// The demoted node (us-east) applies the new leader's stream — proving the
// always-on receiver runs on every node, not just the initial follower.
poll_status(&client, &leader_base, |applied, lag| {
applied >= 4 && lag == 0
});
rt.shutdown_timeout(Duration::from_secs(2));
}
/// `POST /hardnegs` records a hide on the local node (node-local by design).
#[test]
fn region_node_records_hardneg() {
let pair = Pair::new();
let leader_dir = region_dir();
let leader = build_region(pair.topology(), &pair.leader_name, &leader_dir);
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(leader), None),
pair.leader_http,
);
let client = reqwest::blocking::Client::new();
let leader_base = format!("http://{}", pair.leader_http);
let resp = client
.post(format!("{leader_base}/hardnegs"))
.json(&serde_json::json!({ "user_id": 42, "item_id": 7 }))
.send()
.unwrap();
assert_eq!(
resp.status().as_u16(),
204,
"POST /hardnegs must record a hide: {}",
resp.status()
);
rt.shutdown_timeout(Duration::from_secs(2));
}
// ── BUG 1: lag gauge across a leadership change ──────────────────────────────
/// A fully-declared THREE-region topology pointing at three loopback addresses.
struct Trio {
names: [String; 3],
grpc: [SocketAddr; 3],
http: [SocketAddr; 3],
}
impl Trio {
fn new() -> Self {
Self {
names: ["us-east".into(), "eu-west".into(), "ap-south".into()],
grpc: [free_addr(), free_addr(), free_addr()],
http: [free_addr(), free_addr(), free_addr()],
}
}
fn topology(&self) -> TopologySpec {
TopologySpec {
regions: (0..3)
.map(|i| RegionSpec {
name: self.names[i].clone(),
grpc_addr: Some(self.grpc[i].to_string()),
http_addr: Some(self.http[i].to_string()),
grpc_tls: None,
metrics_addr: None,
})
.collect(),
leader: self.names[0].clone(),
write_workers: None,
timeouts: TimeoutsSpec::default(),
replication: ReplicationSpec::default(),
wal: WalSpec::default(),
}
}
}
/// BUG 3 reproduction: items are HTTP-broadcast (not WAL-relayed), so a follower
/// that is partitioned during the item broadcast misses the items forever — and
/// heal (which only re-ships signal segments) must ALSO backfill the item
/// metadata + embeddings, so heal is the single recovery verb that leaves the
/// follower with EXACTLY the leader's data.
///
/// Before the fix, the leader's `/items` broadcast to a partitioned follower
/// landed in the `failed` list and nothing backfilled it; the follower's feed
/// could not rank items it never received, even after heal closed the signal gap.
#[test]
fn region_node_heal_backfills_missed_items() {
let pair = Pair::new();
let leader_dir = region_dir();
let leader = build_region(pair.topology(), &pair.leader_name, &leader_dir);
let follower_node_dir = region_dir();
let follower = Arc::new(build_region(
pair.topology(),
&pair.follower_name,
&follower_node_dir,
));
// Build the runtime AFTER the nodes (and their data-dir guards):
// locals drop in reverse order, so the runtime — which owns the
// nodes via the serve tasks — tears down BEFORE the dirs delete,
// on the panic-unwind path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.enable_all()
.build()
.unwrap();
serve(
&rt,
build_region_router(Arc::new(leader), None),
pair.leader_http,
);
// The follower's HTTP server is NOT serving yet — it models a region that is
// DOWN during the leader's item/signal broadcast. The leader's best-effort
// HTTP broadcast to it will fail (connection refused) and land in `failed`,
// exactly as it would for a crashed/restarting region. The follower's gRPC
// receiver IS running (started in RegionClusterState::new), so once the
// leader heals it, the relay re-ships signals — but the item broadcast that
// failed during downtime is what heal must backfill.
let client = reqwest::blocking::Client::new();
let leader_base = format!("http://{}", pair.leader_http);
let follower_base = format!("http://{}", pair.follower_http);
// Partition the follower at the RELAY level too, so the eager signal ships
// are skipped while it is down (mirrors the real partition: no gRPC either).
let resp = client
.post(format!("{leader_base}/cluster/partition"))
.json(&serde_json::json!({ "region": pair.follower_name }))
.send()
.unwrap();
assert!(resp.status().is_success(), "partition: {}", resp.status());
// Leader writes items + signals for entities 6,7,8 while the follower is DOWN.
// The item broadcast to the follower fails (connection refused); the signal
// ship is skipped (partitioned).
for e in 6..=8u64 {
let resp = client
.post(format!("{leader_base}/items"))
.json(&serde_json::json!({
"entity_id": e, "metadata": { "title": format!("item {e}") }
}))
.send()
.unwrap();
assert!(resp.status().is_success(), "leader item: {}", resp.status());
let resp = client
.post(format!("{leader_base}/signals"))
.json(&serde_json::json!({ "entity_id": e, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert!(
resp.status().is_success(),
"leader signal: {}",
resp.status()
);
}
// The follower comes back up (HTTP server starts serving). Its store is empty
// for these items — the broadcast during its downtime was lost.
serve(
&rt,
build_region_router(Arc::clone(&follower), None),
pair.follower_http,
);
// The recovered follower knows none of these items: its feed is empty.
let pre_feed: serde_json::Value = client
.get(format!("{follower_base}/feed?profile=trending&limit=10"))
.send()
.unwrap()
.json()
.unwrap();
assert!(
pre_feed["items"].as_array().unwrap().is_empty(),
"recovered follower must not know the leader's items yet (broadcast lost during \
downtime): {pre_feed}"
);
// Heal: the leader re-ships the missed signal segments AND must backfill the
// missed item metadata. After heal, the follower has EXACTLY the leader's
// data — its feed ranks items 6,7,8 WITHOUT the test re-posting them.
let resp = client
.post(format!("{leader_base}/cluster/heal"))
.json(&serde_json::json!({ "region": pair.follower_name }))
.send()
.unwrap();
assert!(resp.status().is_success(), "heal: {}", resp.status());
// Converge on the signal HWM (3 segments), then assert item parity.
poll_status(&client, &follower_base, |applied, lag| {
applied >= 3 && lag == 0
});
// Poll the follower's feed until the backfilled items appear (heal's item
// re-broadcast is async over HTTP).
let deadline = Instant::now() + Duration::from_secs(5);
loop {
let feed: serde_json::Value = client
.get(format!("{follower_base}/feed?profile=trending&limit=10"))
.send()
.unwrap()
.json()
.unwrap();
let ids: std::collections::HashSet<u64> = feed["items"]
.as_array()
.unwrap()
.iter()
.filter_map(|it| it["entity_id"].as_u64())
.collect();
if [6u64, 7, 8].iter().all(|e| ids.contains(e)) {
break;
}
assert!(
Instant::now() <= deadline,
"heal must backfill items 6,7,8 to the follower's feed; saw {ids:?}"
);
std::thread::sleep(Duration::from_millis(20));
}
rt.shutdown_timeout(Duration::from_secs(2));
}
/// BUG 1 reproduction: after the leader stream populates a follower's lag gauge
/// (HWM = N for the OLD leader's shard), promoting a DIFFERENT region to leader
/// must NOT leave the now-non-leader follower reporting a permanent stale lag.
///
/// Before the fix, `local_status` computed
/// `lag = lag_gauge.leader_seqno() applied_seqno(NEW-leader-shard)`. The gauge's
/// `leader_seqno` is a single monotonic scalar fed by EVERY source stream, so it
/// still held the OLD leader's HWM (N), while `applied_seqno(new-leader-shard)`
/// was 0 (the new leader had not shipped). A fully-converged follower then
/// reported `lag = N 0 = N` forever.
#[test]
fn region_node_lag_honest_across_promote() {
// Number of signals the old leader ships before the leadership change.
const N: u64 = 6;
let trio = Trio::new();
// Three nodes: us-east (leader, shard 0), eu-west (shard 1), ap-south
// (shard 2). Dirs FIRST (see `region_dir` for the drop-order contract).
let dirs: Vec<tempfile::TempDir> = (0..3).map(|_| region_dir()).collect();
let nodes: Vec<Arc<RegionClusterState>> = (0..3)
.map(|i| Arc::new(build_region(trio.topology(), &trio.names[i], &dirs[i])))
.collect();
// Build the runtime AFTER the nodes (and their data-dir guards): locals
// drop in reverse order, so the runtime — which owns the nodes via the
// serve tasks — tears down BEFORE the dirs delete, on the panic-unwind
// path too (m11p2: nodes are persistent now).
let rt = tokio::runtime::Builder::new_multi_thread()
.worker_threads(3)
.enable_all()
.build()
.unwrap();
for (i, node) in nodes.iter().enumerate() {
serve(
&rt,
build_region_router(Arc::clone(node), None),
trio.http[i],
);
}
let client = reqwest::blocking::Client::new();
let bases: Vec<String> = (0..3).map(|i| format!("http://{}", trio.http[i])).collect();
// us-east (shard 0) leads: write N signals, replicated to eu-west + ap-south.
// This populates each follower's lag gauge HWM for the OLD leader's shard.
for e in 1..=N {
for base in &bases {
let _ = client
.post(format!("{base}/items"))
.json(&serde_json::json!({ "entity_id": e, "metadata": {} }))
.send()
.unwrap();
}
let resp = client
.post(format!("{}/signals", bases[0]))
.json(&serde_json::json!({ "entity_id": e, "signal": "view", "weight": 1.0 }))
.send()
.unwrap();
assert!(
resp.status().is_success(),
"leader signal: {}",
resp.status()
);
}
// eu-west (the future non-leader) fully converges on the old leader's stream.
poll_status(&client, &bases[1], |applied, lag| applied >= N && lag == 0);
// Promote ap-south (shard 2) to leader on ALL nodes. ap-south has shipped
// NOTHING yet, so every node's applied-seqno for shard 2 is 0.
for base in &bases {
let resp = client
.post(format!("{base}/cluster/promote"))
.json(&serde_json::json!({ "region": trio.names[2] }))
.send()
.unwrap();
assert!(resp.status().is_success(), "promote: {}", resp.status());
}
// eu-west is now a NON-leader, fully converged (nothing new to apply). Its lag
// MUST be 0 — it is not behind the new leader, which has shipped nothing.
// Before the fix this read N (stale gauge HWM from shard 0 applied(shard2)=0).
let status: serde_json::Value = client
.get(format!("{}/cluster/status/local", bases[1]))
.send()
.unwrap()
.json()
.unwrap();
assert_eq!(
status["lag_events"].as_u64(),
Some(0),
"a converged non-leader must report lag 0 against a new leader that shipped \
nothing — stale-gauge lag is bug 1: {status}"
);
rt.shutdown_timeout(Duration::from_secs(2));
}