tidaldb/tidal-net/tests/catchup_retry.rs
jx12n d0a52e4530 feat(m11): catch-up timer retry + TSEG segment version header (m11p4)
WAL segment format: 8-byte TSEG header (magic + version byte + 3 reserved)
prepended to every new segment. Legacy headerless segments (m0-m11p3) read
as implicit v0 — no migration. Unknown magic/version surfaces as
WalError::SegmentFormatUnknown at open time; foreign files are never
repaired or truncated (fixes the silent data-loss path from the p3 rollout
incident where torn-tail repair zeroed a follower's unreadable segments).

Catch-up transport: FAILED_PRECONDITION ("snapshot required") and stream
errors that skip the shard now arm a timer retry (re-arm-on-skip is the
load-bearing liveness fix — without it a skipped pull never re-fires and
the follower stays permanently behind). Single retry pending per shard;
CatchupRunner owns the Arc'd state shared between the retry tasks and the
transport. Test: tidal-net/tests/catchup_retry.rs covers the retry path.

Stress: k8s stress-job-t2a/t2b yaml + ops/stress-test-p3-t2 runbook.
2026-06-11 17:05:20 -06:00

195 lines
6.9 KiB
Rust

//! m11p4: the catch-up retry timer.
//!
//! The event path alone (re-pull on the next pushed segment) deadlocks an
//! idle cluster: a follower whose pull failed — e.g. the leader's gRPC
//! server was not yet ready during a rolling restart — waited for a push
//! that never came and stayed lagged forever (the 2026-06-11 p3 rollout
//! incident). These tests drive the REAL transports over real sockets: a
//! pull that fails against a not-yet-listening leader must self-heal on the
//! timer with NO push ever sent, and a pull that completes cleanly must not
//! keep re-pulling.
use std::{
net::SocketAddr,
sync::{
Arc,
atomic::{AtomicU64, Ordering},
},
time::{Duration, Instant},
};
use tidal_net::{
GrpcTransport, GrpcTransportConfig,
sources::{SegmentChunk, SegmentReadError, SegmentSource, ServingSources},
};
use tidaldb::replication::{shard::ShardId, transport::Transport};
/// Bind port 0 to obtain a free, OS-assigned address (tonic cannot bind 0
/// directly, so we resolve a concrete port up front).
fn free_addr() -> SocketAddr {
let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind");
listener.local_addr().expect("local_addr")
}
/// A leader-side WAL read-back serving one synthetic chunk `[1..=5]`, plus a
/// stream-open counter (`flushed_seq` is read exactly once per
/// `StreamSegments` open, so it doubles as the open count).
struct FixedSegments {
opens: Arc<AtomicU64>,
}
impl SegmentSource for FixedSegments {
fn source_shard(&self) -> ShardId {
ShardId(0)
}
fn stream_baseline(&self) -> u64 {
0
}
fn flushed_seq(&self) -> u64 {
self.opens.fetch_add(1, Ordering::AcqRel);
5
}
fn collect_from(
&self,
from_seq: u64,
_max_events: u64,
_max_bytes: usize,
) -> Result<Vec<SegmentChunk>, SegmentReadError> {
if from_seq > 5 {
return Ok(vec![]);
}
Ok(vec![SegmentChunk {
bytes: vec![0xAB; 16],
first_seq: from_seq,
last_seq: 5,
event_count: 5 - from_seq + 1,
}])
}
}
fn follower_config(listen: SocketAddr, leader: SocketAddr) -> GrpcTransportConfig {
GrpcTransportConfig {
local_shard: ShardId(1),
listen_addr: listen,
peers: std::iter::once((ShardId(0), leader)).collect(),
insecure: true,
// Short enough to keep the test fast; the re-arm-on-skip logic walks
// it past MIN_CATCHUP_INTERVAL's 2s rate limit regardless.
catchup_retry_interval: Duration::from_millis(300),
..GrpcTransportConfig::default()
}
}
/// THE incident shape: the follower's only pull fails (leader not yet
/// listening), the cluster stays completely idle (no pushes, no further
/// `request_catchup`), and the data must still arrive — via the retry timer
/// alone.
#[test]
#[allow(clippy::significant_drop_tightening)] // transports intentionally live to test end
fn failed_pull_retries_on_timer_with_no_push() {
let leader_addr = free_addr();
let follower =
GrpcTransport::new(follower_config(free_addr(), leader_addr)).expect("follower transport");
// One pull while the leader is down: `tcp connect error`, exactly like
// the rolling-restart race.
follower.request_catchup(ShardId(0), 1);
// The leader comes up ~200ms later. NOTHING else happens: no writes, no
// pushes, no new request_catchup.
std::thread::sleep(Duration::from_millis(200));
let opens = Arc::new(AtomicU64::new(0));
let leader_sources = ServingSources {
applied: None,
segments: Some(Arc::new(FixedSegments {
opens: Arc::clone(&opens),
})),
applied_sink: Arc::default(),
};
let _leader = GrpcTransport::new_with_sources(
GrpcTransportConfig {
local_shard: ShardId(0),
listen_addr: leader_addr,
insecure: true,
..GrpcTransportConfig::default()
},
leader_sources,
)
.expect("leader transport");
// The retry cadence is 300ms, but the shared MIN_CATCHUP_INTERVAL rate
// limit (2s) defers real attempts — the re-arm-on-skip logic must carry
// the wake-up across those skips. Allow a generous deadline; typical
// arrival is ~2.5s.
let deadline = Instant::now() + Duration::from_secs(15);
let payload = loop {
if let Some(p) = follower.try_recv_segment() {
break p;
}
assert!(
Instant::now() < deadline,
"catch-up payload never arrived: the retry timer is not firing \
(followers would stay lagged forever in an idle cluster)"
);
std::thread::sleep(Duration::from_millis(25));
};
assert_eq!(payload.id.shard_id, ShardId(0));
assert_eq!(
payload.id.seqno, 1,
"the pull starts at the requested seqno"
);
assert_eq!(payload.leader_last_seq, 5);
assert_eq!(payload.event_count, 5);
}
/// A pull that completes cleanly must NOT keep the timer alive: no further
/// streams open once the follower is caught up (the retry exists for FAILED
/// pulls, not as a polling loop).
#[test]
#[allow(clippy::significant_drop_tightening)] // transports intentionally live to test end
fn clean_completion_does_not_keep_retrying() {
let leader_addr = free_addr();
let opens = Arc::new(AtomicU64::new(0));
let leader_sources = ServingSources {
applied: None,
segments: Some(Arc::new(FixedSegments {
opens: Arc::clone(&opens),
})),
applied_sink: Arc::default(),
};
let _leader = GrpcTransport::new_with_sources(
GrpcTransportConfig {
local_shard: ShardId(0),
listen_addr: leader_addr,
insecure: true,
..GrpcTransportConfig::default()
},
leader_sources,
)
.expect("leader transport");
let follower =
GrpcTransport::new(follower_config(free_addr(), leader_addr)).expect("follower transport");
// The leader is up: the one pull succeeds.
follower.request_catchup(ShardId(0), 1);
let deadline = Instant::now() + Duration::from_secs(10);
while follower.try_recv_segment().is_none() {
assert!(Instant::now() < deadline, "the healthy pull must succeed");
std::thread::sleep(Duration::from_millis(25));
}
let opens_after_success = opens.load(Ordering::Acquire);
assert_eq!(opens_after_success, 1, "exactly one stream served the pull");
// Wait past several retry intervals AND the 2s MIN_CATCHUP_INTERVAL rate
// limit: a buggy always-armed timer would only produce its real re-open
// once the rate limit allows (~2.1s), so a shorter wait would miss it.
std::thread::sleep(Duration::from_millis(2600));
assert_eq!(
opens.load(Ordering::Acquire),
opens_after_success,
"a completed pull must not keep re-opening streams on the timer"
);
}