tidaldb/tidal-server/src/cluster/topology.rs
jx12n d5d1e7d81a feat(m11): observability+ops (m11p8) + perf-sweep wave 2 T2
m11p8 closes G-O + §1.4-3:
- Cluster metrics: breaker state, forwards, self-heal on /metrics; multi-shard sibling render (shard="N")
- Grafana cluster row + 8-rule Prometheus alert group
- Request-id / TraceLayer on both cluster routers; id rides forward hop
- Truthful status: flushed leader applied_events frontier; post-promote ShardId(0) keying fix
- Self-driving heal: tick_self_heal re-arms stuck-peer backlog every ~3s
- WAL PITR: wal.archive_dir, archive-before-delete gap-free
- tidalctl backup/restore with BLAKE3 content-hash verification
- Rolling-upgrade build_version handshake (N/N+1, never rejects) + Woodpecker release gate

perf-sweep wave 2 T2: one-get-per-type pre-pass in ranking executor
- signal_values.rs pre-fetches all signal kinds before scoring loop
- Eliminates per-item repeated DashMap lookups: −18.8% for_you, −31% under writes
- Byte-identical output verified with A/B test harness
2026-06-13 09:17:49 -06:00

1275 lines
54 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

//! Cluster topology: the YAML spec, region declarations, and validation.
//!
//! `default-cluster.yaml` is the compiled-in default; an operator can override
//! it with `--topology`. The single-process cluster (m8p8) needs only region
//! names and an optional `grpc_addr`; the multi-process region node (m8p10)
//! also requires a per-region `http_addr` and a stable `grpc_addr` so sibling
//! processes can reach each other (see [`validate_multiproc`]).
use std::{
path::{Path, PathBuf},
time::Duration,
};
use serde::Deserialize;
use tidaldb::replication::shard::{RegionId, ShardId};
use crate::{
error::{Result, ServerError},
offload::ClusterWritePoolConfig,
};
/// Top-level cluster topology spec parsed from YAML.
#[derive(Debug, Deserialize)]
pub struct TopologySpec {
/// All regions in declaration order. The 0-based index of a region in this
/// list IS its `RegionId`, so every process that parses the same file agrees
/// on the region → id mapping (see `ClusterState::new` / `ShardReplica::new`).
pub regions: Vec<RegionSpec>,
/// Name of the region that initially leads (must name a declared region).
pub leader: String,
/// Number of runtime-free OS worker threads serving cluster write/heal
/// requests (gRPC segment ship). Bounds write concurrency on the hottest
/// cluster path. When omitted, defaults to
/// [`ClusterWritePoolConfig::default`] (derived from available
/// parallelism). See [`crate::offload::ClusterWritePool`]. Must be ≥ 1 when
/// given ([`load_topology`] rejects 0 instead of silently clamping).
#[serde(default)]
pub write_workers: Option<usize>,
/// Optional operation-timeout overrides. Omitted in every topology this
/// phase ships (the defaults are right for loopback/VPC); the knob exists so
/// a WAN deployment can widen budgets without recompiling.
#[serde(default)]
pub timeouts: TimeoutsSpec,
/// Optional replication ship-path tuning (the m11p1 batched/windowed
/// per-peer senders). Omitted = engine defaults (256-event batches,
/// window 4, 100ms retry backoff).
#[serde(default)]
pub replication: ReplicationSpec,
/// Optional WAL group-commit tuning. Omitted = engine defaults
/// (batch 100 events / 10ms). Tune against the measured
/// `tidaldb_cluster_wal_fsync_us` on the deployment's volume.
#[serde(default)]
pub wal: WalSpec,
/// Optional election / failure-detector tuning (m11p4). Omitted =
/// defaults (300ms heartbeats, 15003000ms election timeout, 900ms
/// leader lease, auto-election on).
#[serde(default)]
pub election: ElectionSpec,
/// Optional shard-group declaration (m11p6). When present, the keyspace is
/// hash-split into `shards.len()` groups, each a replication group at RF =
/// `replicas.len()` with its own WAL/relay/commit-index and its own elected
/// leader; a node hosts one [`ShardReplica`](super::node) per group it
/// appears in. When ABSENT, the legacy "1 shard × RF=all-regions" group is
/// synthesized (id 0, every region a replica with its address verbatim,
/// `leader` = [`Self::leader`]) — byte-for-byte today's single-log cluster.
/// See [`Self::resolve_shard_groups`].
#[serde(default)]
pub shards: Option<Vec<ShardSpec>>,
}
/// One shard group's declaration (the optional `shards:` YAML block, m11p6).
#[derive(Debug, Deserialize)]
pub struct ShardSpec {
/// Dense, unique group id in `[0, shards.len())`. The gateway hash-routes an
/// entity to `ShardId(id)` via [`tidaldb::replication::ShardRouter`]; the id
/// also names the group's data subdir and gRPC port offset.
pub id: u16,
/// Term-0 / preferred leader (must name a declared region that is also one of
/// this group's `replicas`). Defaults to the first replica when omitted.
#[serde(default)]
pub leader: Option<String>,
/// The RF replica nodes hosting this group.
pub replicas: Vec<ShardReplicaSpec>,
}
/// One replica placement within a [`ShardSpec`] (m11p6).
#[derive(Debug, Deserialize)]
pub struct ShardReplicaSpec {
/// The hosting node — must name a declared region.
pub node: String,
/// This replica's advertised gRPC address (`host:port`, may be DNS). When
/// omitted, derived as the node's `grpc_addr` with `id` added to its port
/// (so id 0 = the node's address verbatim — the legacy port).
#[serde(default)]
pub grpc_addr: Option<String>,
/// This replica's local gRPC bind (`SocketAddr`). When omitted, derived from
/// the node's `grpc_bind` (or `grpc_addr`) with `id` added to its port.
#[serde(default)]
pub grpc_bind: Option<String>,
}
/// A fully-resolved shard group: ids resolved to [`RegionId`]s, addresses
/// resolved (explicit or port-derived). Produced by
/// [`TopologySpec::resolve_shard_groups`] — the single seam that replaces the
/// 1:1 [`shard_of_region`] assumption.
#[derive(Debug, Clone)]
pub struct ResolvedShardGroup {
/// The data-shard group id (gateway hash output; map key; subdir/port name).
pub shard: ShardId,
/// The group's term-0 / preferred leader node.
pub leader: RegionId,
/// `leader`'s region name (display).
pub leader_name: String,
/// The group's RF replica placements.
pub replicas: Vec<ResolvedReplica>,
}
/// One resolved replica placement within a [`ResolvedShardGroup`] (m11p6).
#[derive(Debug, Clone)]
pub struct ResolvedReplica {
/// The hosting node's region id.
pub region: RegionId,
/// The hosting node's region name.
pub name: String,
/// The replica's advertised gRPC address (explicit or port-derived).
pub grpc_addr: String,
/// The replica's local gRPC bind, when explicit/derived (else `None` ⇒ the
/// node derives it from `grpc_addr` at boot per `resolve_grpc_bind_addr`).
pub grpc_bind: Option<String>,
}
/// Election / failure-detector tuning (the optional `election:` YAML block,
/// m11p4).
#[derive(Debug, Default, Deserialize)]
pub struct ElectionSpec {
/// Leader → follower heartbeat cadence in milliseconds (default 300).
#[serde(default)]
pub heartbeat_interval_ms: Option<u64>,
/// Election-timeout range floor in milliseconds (default 1500): a
/// follower that hears no valid leader for a randomized draw from
/// `[min, max)` starts a pre-vote.
#[serde(default)]
pub election_timeout_min_ms: Option<u64>,
/// Election-timeout range ceiling in milliseconds (default 3000).
#[serde(default)]
pub election_timeout_max_ms: Option<u64>,
/// Check-quorum window in milliseconds (default 900): a leader that
/// cannot reach a majority of peers within it steps down. Constrained:
/// `leader_lease_ms + heartbeat_interval_ms < election_timeout_min_ms`
/// — the deposed leader must stop before any successor can be elected.
#[serde(default)]
pub leader_lease_ms: Option<u64>,
/// Whether timeouts start elections automatically (default true). When
/// false the failure detector and term fencing still run; elections fire
/// only through `/cluster/promote` (the fenced transfer). After a
/// full-cluster restart in this mode the cluster is deliberately
/// leaderless until an operator promotes.
#[serde(default)]
pub auto_election: Option<bool>,
}
impl ElectionSpec {
/// Resolved heartbeat interval.
#[must_use]
pub fn heartbeat_interval(&self) -> Duration {
Duration::from_millis(self.heartbeat_interval_ms.unwrap_or(300))
}
/// Resolved election-timeout floor.
#[must_use]
pub fn election_timeout_min(&self) -> Duration {
Duration::from_millis(self.election_timeout_min_ms.unwrap_or(1_500))
}
/// Resolved election-timeout ceiling.
#[must_use]
pub fn election_timeout_max(&self) -> Duration {
Duration::from_millis(self.election_timeout_max_ms.unwrap_or(3_000))
}
/// Resolved check-quorum lease.
#[must_use]
pub fn leader_lease(&self) -> Duration {
Duration::from_millis(self.leader_lease_ms.unwrap_or(900))
}
/// Resolved auto-election flag.
#[must_use]
pub fn auto(&self) -> bool {
self.auto_election.unwrap_or(true)
}
}
/// Replication ship-path tuning (the optional `replication:` YAML block).
#[derive(Debug, Default, Deserialize)]
pub struct ReplicationSpec {
/// Max relay events coalesced into one shipped batch (1256, the WAL wire
/// format's per-batch ceiling).
#[serde(default)]
pub batch_max_events: Option<usize>,
/// In-flight batches per peer (164). 1 = strictly in-order shipping;
/// higher values pipeline across the peer RTT (out-of-order arrivals park
/// gap-aware on the receiver).
#[serde(default)]
pub window: Option<usize>,
/// Milliseconds before a transiently-failed batch ship is retried.
#[serde(default)]
pub retry_ms: Option<u64>,
/// Deployment-default write acknowledgment mode (m11p3): `leader`
/// (default — success at leader group-commit fsync) or `quorum` (success
/// once a majority of the replica set durably holds the write). Callers
/// override per request with the `x-tidal-ack` header.
#[serde(default)]
pub ack: Option<String>,
/// Milliseconds an `ack=quorum` write waits for the commit index before
/// returning a retryable 503 naming the laggards. Default 2000 (the
/// cross-region replication SLO).
#[serde(default)]
pub quorum_timeout_ms: Option<u64>,
/// Milliseconds a FAILED catch-up pull waits before the transport's
/// timer re-pulls (m11p4). The timer is what lets an IDLE cluster
/// self-heal a follower whose pull failed during a rolling restart —
/// without it, the next retry waited for a write that might never
/// arrive. Default 30000. Must be ≥ 1 when given.
#[serde(default)]
pub catchup_retry_ms: Option<u64>,
/// Milliseconds a staged snapshot artifact is REUSED before it is refreshed
/// (m11p5 §2.1). Counted from the LAST fetch's completion, not creation, so
/// a steady stream of joiners keeps one artifact warm. This governs only
/// artifact REUSE — it NEVER unpins the WAL retention of an active consumer
/// (a mid-fetch joiner holds the pin until it releases); a separate hard cap
/// (4× this) force-drops a never-releasing pin so a dead joiner cannot
/// freeze compaction forever. Default 600000 (10 min). Must be ≥ 1 when
/// given (0 would expire an artifact the instant it is staged, defeating the
/// single-flight share for concurrent joiners).
#[serde(default)]
pub snapshot_artifact_ttl_ms: Option<u64>,
/// Drain + clean-exit(0) once the durable `reseed_required` marker latches
/// (m11p5 §2.4). Default `false`; the k8s cluster manifests set it `true` so
/// a node that needs a reseed restarts itself (the PVC/pod orchestration
/// then re-runs boot-time install). Even with this `true`, the self-restart
/// is REFUSED (loudly, in status + gauge) when the node's local view shows
/// the remaining voters cannot sustain quorum without it — exiting during a
/// 2-voter window would be total write unavailability.
#[serde(default)]
pub reseed_self_restart: Option<bool>,
/// Learner promotion distance AND the readiness-convergence hysteresis
/// threshold (m11p5 §3.3 / §4), in events. A learner is promotion-eligible
/// once its durable mark is within this many events of the leader's flushed
/// frontier; an install-boot node is readiness-converged once its catch-up
/// lag first falls at or below this threshold (never `lag == 0`, which an
/// open-loop load keeps perpetually false). Default 1024.
#[serde(default)]
pub learner_promote_lag: Option<u64>,
}
/// WAL group-commit tuning (the optional `wal:` YAML block).
#[derive(Debug, Default, Deserialize)]
pub struct WalSpec {
/// Events per group-commit fsync (1256).
#[serde(default)]
pub batch_size: Option<usize>,
/// Max milliseconds a partial batch waits before flushing.
#[serde(default)]
pub batch_timeout_ms: Option<u64>,
/// Optional WAL archive directory for point-in-time recovery (m11p8). When
/// set, each hosted shard group copies its sealed segments here before
/// compaction deletes them (segment filenames encode the shard id, so
/// co-located groups can share one archive dir without collision). Put it on
/// durable storage separate from the live data dir.
#[serde(default)]
pub archive_dir: Option<String>,
}
/// Operation-timeout overrides (the optional `timeouts:` YAML block).
#[derive(Debug, Default, Deserialize)]
pub struct TimeoutsSpec {
/// Per-peer budget in SECONDS for the item/embedding leader broadcast and
/// the promote fan-out. Defaults to
/// [`super::forward::BROADCAST_PEER_TIMEOUT`] (2s) — right for
/// loopback/VPC; raise it for WAN topologies where a distant region cannot
/// answer in 2s. Must be ≥ 1 when given ([`load_topology`] rejects 0).
#[serde(default)]
pub broadcast_peer_secs: Option<u64>,
}
/// One region's declaration in the topology.
#[derive(Debug, Deserialize)]
pub struct RegionSpec {
/// Human-readable region name (e.g. `"us-east"`). Unique within the topology.
pub name: String,
/// This region's **advertised** gRPC replication address — a `host:port`
/// that may be a literal IP (`"127.0.0.1:9601"`) OR a DNS name
/// (`"tidaldb-1.tidaldb-peers.ns.svc.cluster.local:9601"`, m11p5). This is
/// the address SIBLING processes dial; with a DNS name tonic re-resolves
/// it on every reconnect, so a pod rescheduled onto a new IP becomes
/// reachable again without a peer restart. When omitted, an OS-assigned
/// loopback port is used — the right default for a single-process cluster
/// where peers never need a stable address. Explicit addresses are REQUIRED
/// in multi-process mode (see [`validate_multiproc`]) so sibling processes
/// can reach each other.
///
/// In multi-process mode this address also drives THIS region's local bind
/// UNLESS [`grpc_bind`](Self::grpc_bind) is set: a literal-IP `grpc_addr`
/// binds itself (today's behavior, byte-for-byte); a DNS `grpc_addr` (which
/// the local socket cannot bind) binds `0.0.0.0:<its port>`. See
/// [`super::transport::resolve_grpc_bind_addr`] for the derivation table.
#[serde(default)]
pub grpc_addr: Option<String>,
/// Optional explicit **local** gRPC bind address (a `SocketAddr`-parseable
/// `host:port`), distinct from the advertised [`grpc_addr`](Self::grpc_addr).
/// Set it when the advertised address is a DNS name the local socket cannot
/// bind and the default `0.0.0.0:<port>` is not what you want (e.g. binding
/// a specific interface). Absent ⇒ the bind is derived from `grpc_addr` per
/// the [`super::transport::resolve_grpc_bind_addr`] table. Validated as a
/// `SocketAddr` when present (a hostname here is rejected — a bind target
/// must be a concrete local address).
#[serde(default)]
pub grpc_bind: Option<String>,
/// This region's public HTTP address (host:port). Required in multi-process
/// mode for write/read forwarding and status aggregation (task 03);
/// unused in single-process mode.
#[serde(default)]
pub http_addr: Option<String>,
/// Optional TLS material for THIS region's gRPC transport (server side and
/// its client channels to peers; mTLS when the client pair is given). When
/// absent the transport runs plaintext — the right posture for the
/// loopback/VPC topologies this phase ships. Multi-process mode only;
/// the single-process cluster's self-loop never leaves the process.
#[serde(default)]
pub grpc_tls: Option<GrpcTlsSpec>,
/// Bind address for this region's Prometheus `/metrics` listener
/// (e.g. `"0.0.0.0:9091"`). When omitted, the node runs without a metrics
/// endpoint — set it in every production topology: cluster mode had NO
/// metrics listener before m11p1 and the roadmap calls that out as a gap.
#[serde(default)]
pub metrics_addr: Option<String>,
/// Optional placement label (m11p6), e.g. an availability zone (`"az-a"`),
/// demoting "region" from a shard identity to a placement hint. **Parsed and
/// accepted but NOT yet honored:** the read-affinity router that would prefer
/// a replica in the reader's zone is a later layer (L4/p8), so setting this
/// today has no runtime effect. Carried now so the schema is forward-stable.
#[serde(default)]
pub zone: Option<String>,
}
/// TLS material for one region's gRPC transport (the optional `grpc_tls:`
/// YAML block). Paths are read at startup by `tidal-net`; a missing or
/// malformed file fails the transport build loudly.
#[derive(Debug, Deserialize)]
pub struct GrpcTlsSpec {
/// CA certificate PEM (trust anchor for peer verification).
pub ca_cert: PathBuf,
/// This node's server certificate PEM.
pub server_cert: PathBuf,
/// This node's server private key PEM.
pub server_key: PathBuf,
/// Client certificate PEM for mTLS (optional; plain TLS without it).
#[serde(default)]
pub client_cert: Option<PathBuf>,
/// Client private key PEM for mTLS (paired with `client_cert`).
#[serde(default)]
pub client_key: Option<PathBuf>,
}
impl GrpcTlsSpec {
/// Project the YAML spec into the transport-layer [`tidal_net::config::TlsConfig`].
#[must_use]
pub(crate) fn to_tls_config(&self) -> tidal_net::config::TlsConfig {
tidal_net::config::TlsConfig {
ca_cert: self.ca_cert.clone(),
server_cert: self.server_cert.clone(),
server_key: self.server_key.clone(),
client_cert: self.client_cert.clone(),
client_key: self.client_key.clone(),
}
}
}
/// The shard a region owns. The mapping is 1:1 in M8 (one shard per region):
/// `RegionId(n)` owns exactly `ShardId(n)`. This function is the SINGLE seam
/// for that assumption — the multi-shard-regions follow-up (see ROADMAP known
/// gaps) changes the mapping here, not at the call sites.
#[must_use]
pub const fn shard_of_region(region: RegionId) -> ShardId {
ShardId(region.0)
}
const DEFAULT_TOPOLOGY_YAML: &str = include_str!("../../config/default-cluster.yaml");
/// Load the cluster topology spec from YAML (or the compiled-in default when
/// `path` is `None`).
///
/// # Errors
///
/// Returns [`ServerError`] if the file cannot be read or the YAML fails to parse.
pub fn load_topology(path: Option<&Path>) -> Result<TopologySpec> {
let raw = match path {
Some(p) => std::fs::read_to_string(p).map_err(|e| ServerError::io(p, e))?,
None => DEFAULT_TOPOLOGY_YAML.to_string(),
};
let spec: TopologySpec = serde_yml::from_str(&raw)
.map_err(|e| ServerError::SchemaConfig(format!("parse topology yaml: {e}")))?;
validate_spec_values(&spec)?;
Ok(spec)
}
/// Mode-independent value validation for a parsed topology (both cluster modes
/// load through [`load_topology`]). Rejects values that would otherwise be
/// silently coerced at runtime (e.g. `write_workers: 0` clamping to 1).
// One flat sequence of independent knob range-checks; splitting it would only
// scatter the validation table that operators read top-to-bottom.
#[allow(clippy::too_many_lines)]
fn validate_spec_values(spec: &TopologySpec) -> Result<()> {
if spec.write_workers == Some(0) {
return Err(ServerError::SchemaConfig(
"write_workers must be >= 1 (omit it to derive from available parallelism)".into(),
));
}
if spec.timeouts.broadcast_peer_secs == Some(0) {
return Err(ServerError::SchemaConfig(
"timeouts.broadcast_peer_secs must be >= 1 (omit it for the 2s default)".into(),
));
}
if let Some(n) = spec.replication.batch_max_events
&& !(1..=256).contains(&n)
{
return Err(ServerError::SchemaConfig(format!(
"replication.batch_max_events must be in [1, 256] (the WAL wire \
format's per-batch ceiling), got {n}"
)));
}
if let Some(w) = spec.replication.window
&& !(1..=64).contains(&w)
{
return Err(ServerError::SchemaConfig(format!(
"replication.window must be in [1, 64], got {w}"
)));
}
if spec.replication.retry_ms == Some(0) {
return Err(ServerError::SchemaConfig(
"replication.retry_ms must be >= 1 (omit it for the 100ms default)".into(),
));
}
if let Some(ack) = spec.replication.ack.as_deref()
&& !matches!(ack, "leader" | "quorum")
{
return Err(ServerError::SchemaConfig(format!(
"replication.ack must be \"leader\" or \"quorum\", got {ack:?}"
)));
}
if spec.replication.catchup_retry_ms == Some(0) {
return Err(ServerError::Cluster(
"replication.catchup_retry_ms must be >= 1 (omit it for the 30000ms \
default); 0 would hot-loop catch-up pulls against the leader"
.into(),
));
}
if spec.replication.quorum_timeout_ms == Some(0) {
return Err(ServerError::SchemaConfig(
"replication.quorum_timeout_ms must be >= 1 (omit it for the 2000ms default)".into(),
));
}
if spec.replication.snapshot_artifact_ttl_ms == Some(0) {
return Err(ServerError::SchemaConfig(
"replication.snapshot_artifact_ttl_ms must be >= 1 (omit it for the 600000ms \
default); 0 would expire a staged snapshot the instant it is created, defeating \
the single-flight artifact share for concurrent joiners"
.into(),
));
}
if spec.replication.learner_promote_lag == Some(0) {
return Err(ServerError::SchemaConfig(
"replication.learner_promote_lag must be >= 1 (omit it for the 1024 default); \
0 is the same perpetually-false `lag == 0` readiness hysteresis the design \
explicitly avoids (an open-loop load never converges to exactly 0 lag)"
.into(),
));
}
if let Some(n) = spec.wal.batch_size
&& !(1..=256).contains(&n)
{
return Err(ServerError::SchemaConfig(format!(
"wal.batch_size must be in [1, 256], got {n}"
)));
}
if spec.wal.batch_timeout_ms == Some(0) {
return Err(ServerError::SchemaConfig(
"wal.batch_timeout_ms must be >= 1 (omit it for the 10ms default)".into(),
));
}
// Election timing invariants (m11p4). The C2 lease bound is load-bearing
// for safety: the leader's check-quorum countdown starts at its last
// successful quorum contact while a follower's election timer starts at
// the last heartbeat it RECEIVED — the anchors differ by up to one
// heartbeat interval, so the deposed leader stops accepting writes
// before any successor can be elected only when
// `lease + heartbeat_interval < election_timeout_min`.
{
let hb = spec.election.heartbeat_interval();
let min = spec.election.election_timeout_min();
let max = spec.election.election_timeout_max();
let lease = spec.election.leader_lease();
if hb.is_zero() || min.is_zero() || lease.is_zero() {
return Err(ServerError::SchemaConfig(
"election timings must all be >= 1ms (omit them for the defaults)".into(),
));
}
if max <= min {
return Err(ServerError::SchemaConfig(format!(
"election.election_timeout_max_ms ({}) must exceed \
election_timeout_min_ms ({})",
max.as_millis(),
min.as_millis()
)));
}
if lease + hb >= min {
return Err(ServerError::SchemaConfig(format!(
"election.leader_lease_ms ({}) + heartbeat_interval_ms ({}) must be \
strictly below election_timeout_min_ms ({}): a deposed leader must stop \
accepting writes before any successor can be elected",
lease.as_millis(),
hb.as_millis(),
min.as_millis()
)));
}
}
validate_shards(spec)?;
Ok(())
}
impl TopologySpec {
/// Resolve the cluster-write-pool config from `write_workers`.
///
/// Shared by both `ClusterState::new` and `ShardReplica::new` so the
/// two cluster modes size their write pool the same way from the same field.
#[must_use]
pub(crate) fn write_pool_config(&self) -> ClusterWritePoolConfig {
self.write_workers.map_or_else(
ClusterWritePoolConfig::default,
ClusterWritePoolConfig::with_workers,
)
}
/// Resolve the per-peer broadcast/fan-out budget from `timeouts`, defaulting
/// to [`super::forward::BROADCAST_PEER_TIMEOUT`].
#[must_use]
pub(crate) fn broadcast_peer_timeout(&self) -> Duration {
self.timeouts
.broadcast_peer_secs
.map_or(super::forward::BROADCAST_PEER_TIMEOUT, Duration::from_secs)
}
/// Resolve the replication ship-queue config from the `replication:` block,
/// engine defaults for anything omitted.
#[must_use]
pub(crate) fn ship_queue_config(&self) -> tidaldb::replication::ShipQueueConfig {
let defaults = tidaldb::replication::ShipQueueConfig::default();
tidaldb::replication::ShipQueueConfig {
max_batch_events: self
.replication
.batch_max_events
.unwrap_or(defaults.max_batch_events),
max_batch_bytes: defaults.max_batch_bytes,
window: self.replication.window.unwrap_or(defaults.window),
retry_backoff: self
.replication
.retry_ms
.map_or(defaults.retry_backoff, Duration::from_millis),
leader_region: defaults.leader_region,
}
}
/// The `metrics_addr` declared for `region_name`, if any.
#[must_use]
pub(crate) fn metrics_addr_of(&self, region_name: &str) -> Option<String> {
self.regions
.iter()
.find(|r| r.name == region_name)
.and_then(|r| r.metrics_addr.clone())
}
/// The number of shard groups (m11p6): `shards.len()`, or 1 when `shards:`
/// is absent (the legacy single replicated log). Drives the gateway's
/// [`tidaldb::replication::ShardRouter`] (`Hash(n)` for `n > 1`, `Single`
/// for `n == 1`).
#[must_use]
pub fn shard_count(&self) -> usize {
self.shards.as_ref().map_or(1, Vec::len)
}
/// Resolve the shard-group assignment (m11p6) — the single seam that
/// generalizes the 1:1 [`shard_of_region`] mapping. With `shards:` present,
/// resolves each replica's node name to a [`RegionId`] and its gRPC address
/// (explicit, or the node's address with the shard id added to its port).
/// Absent ⇒ one group spanning every region (id 0, the topology `leader`,
/// addresses verbatim) — byte-for-byte today's single-log cluster (id 0 ⇒
/// port offset 0 ⇒ the declared address).
///
/// # Errors
///
/// Returns [`ServerError::SchemaConfig`] if a replica/leader names an
/// undeclared region, a group has no replicas, ids are not dense/unique, a
/// derived port overflows, or two shards on one node collide on a gRPC
/// address.
#[allow(clippy::too_many_lines)]
pub fn resolve_shard_groups(&self) -> Result<Vec<ResolvedShardGroup>> {
// The positional region→id map, overflow-guarded ONCE here (a RegionId is
// a u16). Every id below reuses this map rather than re-deriving the cast
// with a different overflow policy — one source of truth for the seam
// every durable id depends on.
let mut name_to_id: std::collections::HashMap<&str, RegionId> =
std::collections::HashMap::with_capacity(self.regions.len());
for (i, r) in self.regions.iter().enumerate() {
let id = RegionId(u16::try_from(i).map_err(|_| {
ServerError::SchemaConfig("topology declares more than 65535 regions".into())
})?);
name_to_id.insert(r.name.as_str(), id);
}
let region_of = |name: &str| -> Result<RegionId> {
name_to_id.get(name).copied().ok_or_else(|| {
ServerError::SchemaConfig(format!(
"shards: replica/leader names undeclared region '{name}'"
))
})
};
let Some(shards) = self.shards.as_ref() else {
// Legacy synthesis: one group, RF = all regions, addresses verbatim.
let leader = region_of(&self.leader)?;
let replicas = self
.regions
.iter()
.map(|r| ResolvedReplica {
region: name_to_id[r.name.as_str()],
name: r.name.clone(),
grpc_addr: r.grpc_addr.clone().unwrap_or_default(),
grpc_bind: r.grpc_bind.clone(),
})
.collect();
return Ok(vec![ResolvedShardGroup {
shard: ShardId(0),
leader,
leader_name: self.leader.clone(),
replicas,
}]);
};
let mut groups = Vec::with_capacity(shards.len());
for spec in shards {
if spec.replicas.is_empty() {
return Err(ServerError::SchemaConfig(format!(
"shard {} declares no replicas",
spec.id
)));
}
let mut replicas = Vec::with_capacity(spec.replicas.len());
let mut seen_nodes = std::collections::HashSet::new();
for r in &spec.replicas {
let region = region_of(&r.node)?;
if !seen_nodes.insert(region) {
return Err(ServerError::SchemaConfig(format!(
"shard {} lists node '{}' twice",
spec.id, r.node
)));
}
let node = &self.regions[usize::from(region.0)];
let grpc_addr = match &r.grpc_addr {
Some(a) => a.clone(),
None => offset_host_port(
node.grpc_addr.as_deref().ok_or_else(|| {
ServerError::SchemaConfig(format!(
"shard {} replica '{}' has no grpc_addr and node declares none \
to derive from",
spec.id, r.node
))
})?,
spec.id,
)?,
};
let grpc_bind = match &r.grpc_bind {
Some(b) => Some(b.clone()),
None => node
.grpc_bind
.as_deref()
.map(|b| offset_host_port(b, spec.id))
.transpose()?,
};
replicas.push(ResolvedReplica {
region,
name: r.node.clone(),
grpc_addr,
grpc_bind,
});
}
let leader_name = spec
.leader
.clone()
.unwrap_or_else(|| replicas[0].name.clone());
let leader = region_of(&leader_name)?;
if !replicas.iter().any(|r| r.region == leader) {
return Err(ServerError::SchemaConfig(format!(
"shard {} leader '{leader_name}' is not one of its replicas",
spec.id
)));
}
groups.push(ResolvedShardGroup {
shard: ShardId(spec.id),
leader,
leader_name,
replicas,
});
}
validate_no_port_collisions(&groups)?;
Ok(groups)
}
}
/// Add `offset` to the port of a `host:port` address; `offset == 0` returns the
/// input verbatim (the legacy-shard guarantee — no reformatting).
fn offset_host_port(addr: &str, offset: u16) -> Result<String> {
if offset == 0 {
return Ok(addr.to_string());
}
let (host, port) = addr.rsplit_once(':').ok_or_else(|| {
ServerError::SchemaConfig(format!(
"cannot derive a shard port from '{addr}' (no ':port' to offset)"
))
})?;
let base: u16 = port.parse().map_err(|_| {
ServerError::SchemaConfig(format!("'{addr}' port is not a number to offset"))
})?;
let derived = base.checked_add(offset).ok_or_else(|| {
ServerError::SchemaConfig(format!(
"deriving shard port from '{addr}' + {offset} overflows u16 — set an explicit \
grpc_addr for this replica"
))
})?;
Ok(format!("{host}:{derived}"))
}
/// Reject two shard replicas on the SAME node that resolve to the same gRPC
/// address (a derived/explicit port clash would make two of the node's shard
/// transports fight for one socket).
fn validate_no_port_collisions(groups: &[ResolvedShardGroup]) -> Result<()> {
let mut seen: std::collections::HashMap<(RegionId, &str), ShardId> =
std::collections::HashMap::new();
for g in groups {
for r in &g.replicas {
if let Some(prev) = seen.insert((r.region, r.grpc_addr.as_str()), g.shard) {
return Err(ServerError::SchemaConfig(format!(
"node '{}' resolves shards {} and {} to the same grpc_addr '{}' — set \
distinct ports (derived ports add the shard id to the node port)",
r.name, prev.0, g.shard.0, r.grpc_addr
)));
}
}
}
Ok(())
}
/// Structural validation of the optional `shards:` block (mode-independent;
/// address resolution is deferred to [`TopologySpec::resolve_shard_groups`]).
/// Checks dense unique ids in `[0, n)`, non-empty replicas, and that named
/// nodes/leaders exist — so a typo fails at load, not at routing time.
fn validate_shards(spec: &TopologySpec) -> Result<()> {
let Some(shards) = spec.shards.as_ref() else {
return Ok(());
};
if shards.is_empty() {
return Err(ServerError::SchemaConfig(
"shards: declared but empty (omit the key for the legacy single group)".into(),
));
}
let names: std::collections::HashSet<&str> =
spec.regions.iter().map(|r| r.name.as_str()).collect();
let mut ids: Vec<u16> = shards.iter().map(|s| s.id).collect();
ids.sort_unstable();
for (want, got) in ids.iter().enumerate() {
if usize::from(*got) != want {
return Err(ServerError::SchemaConfig(format!(
"shard ids must be dense and unique in [0, {}); got {ids:?}",
shards.len()
)));
}
}
for s in shards {
if s.replicas.is_empty() {
return Err(ServerError::SchemaConfig(format!(
"shard {} declares no replicas (RF must be >= 1)",
s.id
)));
}
for r in &s.replicas {
if !names.contains(r.node.as_str()) {
return Err(ServerError::SchemaConfig(format!(
"shard {} replica node '{}' is not a declared region",
s.id, r.node
)));
}
}
if let Some(l) = s.leader.as_deref()
&& !s.replicas.iter().any(|r| r.node == l)
{
return Err(ServerError::SchemaConfig(format!(
"shard {} leader '{l}' must be one of its replicas",
s.id
)));
}
}
Ok(())
}
/// Validate a topology for **multi-process** (`--region`) mode.
///
/// Every region must declare BOTH `grpc_addr` and `http_addr` (sibling
/// processes reach each other over gRPC and forward HTTP via these), region
/// names must be unique, `my_region` must name a declared region, and the
/// leader must be a declared region. Single-process validation is intentionally
/// looser (it auto-allocates gRPC ports and never forwards HTTP), so it lives in
/// `ClusterState::new`, not here.
///
/// # Errors
///
/// Returns [`ServerError::SchemaConfig`] naming the first problem found.
pub fn validate_multiproc(topology: &TopologySpec, my_region: &str) -> Result<()> {
if topology.regions.is_empty() {
return Err(ServerError::SchemaConfig(
"topology must declare at least one region".into(),
));
}
let mut seen = std::collections::HashSet::new();
for region in &topology.regions {
if !seen.insert(region.name.as_str()) {
return Err(ServerError::SchemaConfig(format!(
"duplicate region name '{}' in topology",
region.name
)));
}
// `grpc_addr` is the ADVERTISED replication address (m11p5): it must be
// a syntactic `host:port` but MAY be a DNS name — exactly the
// hostname-tolerant check `http_addr` already passes. The old
// parse-as-SocketAddr gate (the former `resolve_grpc_addr`) rejected
// hostnames at startup; that gate moves here so a DNS topology
// validates, and the bind-vs-advertise split (`resolve_grpc_bind_addr`)
// handles the local socket separately.
match region.grpc_addr.as_deref() {
None | Some("") => {
return Err(ServerError::SchemaConfig(format!(
"region '{}' must declare a grpc_addr in multi-process mode \
(siblings replicate over it)",
region.name
)));
}
Some(addr) => validate_host_port(&region.name, "grpc_addr", addr)?,
}
// `grpc_bind`, when present, is the LOCAL bind socket — it must be a
// concrete `SocketAddr` (a hostname cannot be bound). Absent ⇒ derived
// from `grpc_addr` at boot (see `resolve_grpc_bind_addr`).
if let Some(bind) = region.grpc_bind.as_deref() {
if bind.is_empty() {
return Err(ServerError::SchemaConfig(format!(
"region '{}' grpc_bind must be a host:port SocketAddr when present \
(omit it to derive the bind from grpc_addr)",
region.name
)));
}
bind.parse::<std::net::SocketAddr>().map_err(|e| {
ServerError::SchemaConfig(format!(
"region '{}' grpc_bind '{bind}' must parse as a bindable SocketAddr \
(a hostname is not a valid local bind target): {e}",
region.name
))
})?;
}
match region.http_addr.as_deref() {
None | Some("") => {
return Err(ServerError::SchemaConfig(format!(
"region '{}' must declare an http_addr in multi-process mode \
(used for write/read forwarding and status aggregation)",
region.name
)));
}
Some(addr) => validate_host_port(&region.name, "http_addr", addr)?,
}
}
if !seen.contains(my_region) {
return Err(ServerError::SchemaConfig(format!(
"--region '{my_region}' is not a declared region in the topology"
)));
}
if !seen.contains(topology.leader.as_str()) {
return Err(ServerError::SchemaConfig(format!(
"leader '{}' not found in regions",
topology.leader
)));
}
Ok(())
}
/// Syntactic validation of a declared `host:port` address (an optional
/// `http://`/`https://` prefix is tolerated, matching what
/// [`super::forward::peer_url`] accepts). Catches the config typos that
/// otherwise surface only as every forward to that peer timing out at runtime.
/// Reachability is intentionally NOT probed — sibling processes boot in any
/// order, so an unreachable-at-startup peer is normal.
fn validate_host_port(region: &str, field: &str, addr: &str) -> Result<()> {
let bare = addr
.strip_prefix("http://")
.or_else(|| addr.strip_prefix("https://"))
.unwrap_or(addr);
let bare = bare.trim_end_matches('/');
let bad = |why: &str| {
Err(ServerError::SchemaConfig(format!(
"region '{region}': {field} '{addr}' is not a valid host:port ({why})"
)))
};
let Some((host, port)) = bare.rsplit_once(':') else {
return bad("missing ':port'");
};
if host.is_empty() || host.contains([' ', '/']) {
return bad("empty or malformed host");
}
match port.parse::<u16>() {
Ok(0) => bad("port 0 is not reachable by peers"),
Ok(_) => Ok(()),
Err(_) => bad("port must be 1-65535"),
}
}
#[cfg(test)]
mod tests {
use super::*;
fn region(name: &str, grpc: Option<&str>, http: Option<&str>) -> RegionSpec {
RegionSpec {
name: name.into(),
grpc_addr: grpc.map(str::to_string),
grpc_bind: None,
http_addr: http.map(str::to_string),
grpc_tls: None,
metrics_addr: None,
zone: None,
}
}
fn full_topology() -> TopologySpec {
TopologySpec {
regions: vec![
region("us-east", Some("127.0.0.1:9601"), Some("127.0.0.1:9501")),
region("eu-west", Some("127.0.0.1:9602"), Some("127.0.0.1:9502")),
],
leader: "us-east".into(),
write_workers: None,
timeouts: TimeoutsSpec::default(),
replication: ReplicationSpec::default(),
wal: WalSpec::default(),
election: ElectionSpec::default(),
shards: None,
}
}
#[test]
fn multiproc_accepts_fully_declared_topology() {
validate_multiproc(&full_topology(), "eu-west").expect("fully-declared topology is valid");
}
#[test]
fn multiproc_rejects_missing_grpc_addr() {
let mut t = full_topology();
t.regions[1].grpc_addr = None;
let err = validate_multiproc(&t, "us-east").expect_err("missing grpc_addr must fail");
assert!(err.to_string().contains("grpc_addr"), "got {err}");
}
#[test]
fn multiproc_rejects_missing_http_addr() {
let mut t = full_topology();
t.regions[0].http_addr = None;
let err = validate_multiproc(&t, "us-east").expect_err("missing http_addr must fail");
assert!(err.to_string().contains("http_addr"), "got {err}");
}
#[test]
fn multiproc_rejects_unknown_my_region() {
let err = validate_multiproc(&full_topology(), "ap-south")
.expect_err("unknown --region must fail");
assert!(err.to_string().contains("ap-south"), "got {err}");
}
#[test]
fn multiproc_rejects_duplicate_names() {
let mut t = full_topology();
t.regions[1].name = "us-east".into();
let err = validate_multiproc(&t, "us-east").expect_err("duplicate region names must fail");
assert!(err.to_string().contains("duplicate"), "got {err}");
}
#[test]
fn multiproc_rejects_undeclared_leader() {
let mut t = full_topology();
t.leader = "ghost".into();
let err = validate_multiproc(&t, "us-east").expect_err("undeclared leader must fail");
assert!(err.to_string().contains("ghost"), "got {err}");
}
#[test]
fn multiproc_rejects_malformed_http_addr() {
for (addr, why) in [
("127.0.0.1", "missing port"),
("127.0.0.1:0", "port 0"),
("127.0.0.1:99999", "port out of range"),
(":9501", "empty host"),
("10.0.0.1:port", "non-numeric port"),
] {
let mut t = full_topology();
t.regions[1].http_addr = Some(addr.into());
let Err(err) = validate_multiproc(&t, "us-east") else {
panic!("http_addr '{addr}' ({why}) must be rejected");
};
assert!(err.to_string().contains("http_addr"), "got {err}");
}
}
#[test]
fn multiproc_accepts_hostname_grpc_addr() {
// m11p5: `grpc_addr` is the ADVERTISED address and may be a DNS name —
// the old SocketAddr-parse gate rejected this at startup, the whole
// point of the bind/advertise split.
for addr in [
"127.0.0.1:9601",
"tidaldb-1:9601",
"tidaldb-1.tidaldb-peers.tidaldb-cluster.svc.cluster.local:9601",
] {
let mut t = full_topology();
t.regions[1].grpc_addr = Some(addr.into());
validate_multiproc(&t, "us-east")
.unwrap_or_else(|e| panic!("grpc_addr '{addr}' must be accepted: {e}"));
}
}
#[test]
fn multiproc_rejects_malformed_grpc_addr() {
for (addr, why) in [
("127.0.0.1", "missing port"),
("127.0.0.1:0", "port 0"),
("127.0.0.1:99999", "port out of range"),
(":9601", "empty host"),
("10.0.0.1:port", "non-numeric port"),
] {
let mut t = full_topology();
t.regions[1].grpc_addr = Some(addr.into());
let Err(err) = validate_multiproc(&t, "us-east") else {
panic!("grpc_addr '{addr}' ({why}) must be rejected");
};
assert!(err.to_string().contains("grpc_addr"), "got {err}");
}
}
#[test]
fn multiproc_accepts_socketaddr_grpc_bind() {
// An explicit local bind must be a concrete SocketAddr.
let mut t = full_topology();
t.regions[1].grpc_bind = Some("0.0.0.0:9601".into());
validate_multiproc(&t, "us-east").expect("a SocketAddr grpc_bind is valid");
}
#[test]
fn multiproc_rejects_hostname_grpc_bind() {
// A hostname is NOT a bindable local address — `grpc_bind` is the bind,
// not the advertise, so it must parse as a SocketAddr.
for bind in ["tidaldb-1:9601", "0.0.0.0", ""] {
let mut t = full_topology();
t.regions[1].grpc_bind = Some(bind.into());
let Err(err) = validate_multiproc(&t, "us-east") else {
panic!("grpc_bind '{bind}' must be rejected (not a SocketAddr)");
};
assert!(err.to_string().contains("grpc_bind"), "got {err}");
}
}
#[test]
fn multiproc_accepts_scheme_prefixed_and_hostname_http_addr() {
for addr in [
"http://127.0.0.1:9501",
"https://tidal-eu.svc.cluster.local:9501",
"tidal-eu:9501",
] {
let mut t = full_topology();
t.regions[1].http_addr = Some(addr.into());
validate_multiproc(&t, "us-east")
.unwrap_or_else(|e| panic!("http_addr '{addr}' must be accepted: {e}"));
}
}
#[test]
fn spec_value_validation_rejects_zero_knobs() {
let mut t = full_topology();
t.write_workers = Some(0);
let err = validate_spec_values(&t).expect_err("write_workers: 0 must be rejected");
assert!(err.to_string().contains("write_workers"), "got {err}");
let mut t = full_topology();
t.timeouts.broadcast_peer_secs = Some(0);
let err = validate_spec_values(&t).expect_err("broadcast_peer_secs: 0 must be rejected");
assert!(err.to_string().contains("broadcast_peer_secs"), "got {err}");
// m11p5: a 0 snapshot-artifact TTL would expire an artifact the instant
// it is staged, defeating the single-flight share — rejected by name.
let mut t = full_topology();
t.replication.snapshot_artifact_ttl_ms = Some(0);
let err =
validate_spec_values(&t).expect_err("snapshot_artifact_ttl_ms: 0 must be rejected");
assert!(
err.to_string().contains("snapshot_artifact_ttl_ms"),
"got {err}"
);
}
#[test]
fn broadcast_peer_timeout_resolves_default_and_override() {
let mut t = full_topology();
assert_eq!(
t.broadcast_peer_timeout(),
super::super::forward::BROADCAST_PEER_TIMEOUT,
"omitted knob resolves to the compiled-in default"
);
t.timeouts.broadcast_peer_secs = Some(10);
assert_eq!(t.broadcast_peer_timeout(), Duration::from_secs(10));
}
#[test]
fn grpc_tls_spec_parses_and_projects() {
let yaml = r"
regions:
- name: us-east
grpc_addr: 127.0.0.1:9601
http_addr: 127.0.0.1:9501
grpc_tls:
ca_cert: /etc/tidal/ca.pem
server_cert: /etc/tidal/server.pem
server_key: /etc/tidal/server.key
leader: us-east
";
let spec: TopologySpec = serde_yml::from_str(yaml).expect("tls block parses");
let tls = spec.regions[0]
.grpc_tls
.as_ref()
.expect("grpc_tls present")
.to_tls_config();
assert_eq!(tls.ca_cert, PathBuf::from("/etc/tidal/ca.pem"));
assert!(tls.client_cert.is_none(), "mTLS pair defaults to off");
}
// ── m11p6 shard-group schema ─────────────────────────────────────────────
#[test]
fn absent_shards_synthesizes_one_group_verbatim() {
// The legacy guarantee: no `shards:` ⇒ one group, RF = all regions,
// leader = topology.leader, addresses byte-for-byte (offset 0).
let t = full_topology();
assert_eq!(t.shard_count(), 1);
let groups = t.resolve_shard_groups().expect("legacy synthesis");
assert_eq!(groups.len(), 1);
let g = &groups[0];
assert_eq!(g.shard, ShardId(0));
assert_eq!(g.leader, RegionId(0));
assert_eq!(g.leader_name, "us-east");
assert_eq!(g.replicas.len(), 2);
assert_eq!(g.replicas[0].region, RegionId(0));
assert_eq!(g.replicas[0].grpc_addr, "127.0.0.1:9601"); // verbatim
assert_eq!(g.replicas[1].grpc_addr, "127.0.0.1:9602"); // verbatim
}
fn three_node_yaml(shards: &str) -> TopologySpec {
let yaml = format!(
"regions:\n - {{ name: us-east, grpc_addr: \"127.0.0.1:9601\", \
http_addr: \"127.0.0.1:9501\", zone: az-a }}\n - {{ name: eu-west, \
grpc_addr: \"127.0.0.1:9611\", http_addr: \"127.0.0.1:9511\", zone: az-b }}\n \
- {{ name: ap-south, grpc_addr: \"127.0.0.1:9621\", http_addr: \"127.0.0.1:9521\" }}\n\
leader: us-east\n{shards}"
);
load_topology_from_str(&yaml).expect("topology parses + validates")
}
fn load_topology_from_str(raw: &str) -> Result<TopologySpec> {
let spec: TopologySpec = serde_yml::from_str(raw)
.map_err(|e| ServerError::SchemaConfig(format!("parse: {e}")))?;
validate_spec_values(&spec)?;
Ok(spec)
}
#[test]
fn explicit_shards_resolve_with_derived_and_explicit_ports() {
let t = three_node_yaml(
"shards:\n - id: 0\n leader: us-east\n replicas: [{node: us-east}, \
{node: eu-west}, {node: ap-south}]\n - id: 1\n leader: eu-west\n \
replicas: [{node: us-east, grpc_addr: \"127.0.0.1:7000\"}, {node: eu-west}, \
{node: ap-south}]\n",
);
assert_eq!(t.shard_count(), 2);
// `zone:` parses onto the RegionSpec (accepted, not yet honored).
assert_eq!(t.regions[0].zone.as_deref(), Some("az-a"));
let groups = t.resolve_shard_groups().expect("resolve");
assert_eq!(groups.len(), 2);
// Shard 0: derived = node port + 0 (verbatim).
assert_eq!(groups[0].shard, ShardId(0));
assert_eq!(groups[0].leader, RegionId(0));
assert_eq!(groups[0].replicas[0].grpc_addr, "127.0.0.1:9601");
assert_eq!(groups[0].replicas[1].grpc_addr, "127.0.0.1:9611");
// Shard 1: us-east explicit 7000; eu-west/ap-south derived = port + 1.
assert_eq!(groups[1].shard, ShardId(1));
assert_eq!(groups[1].leader, RegionId(1));
assert_eq!(groups[1].replicas[0].grpc_addr, "127.0.0.1:7000");
assert_eq!(groups[1].replicas[1].grpc_addr, "127.0.0.1:9612");
assert_eq!(groups[1].replicas[2].grpc_addr, "127.0.0.1:9622");
}
#[test]
fn shards_reject_non_dense_ids() {
let yaml = "shards:\n - {id: 0, replicas: [{node: us-east}]}\n \
- {id: 2, replicas: [{node: eu-west}]}\n";
let err = load_topology_from_str(&format!(
"regions:\n - {{name: us-east, grpc_addr: \"1:1\", http_addr: \"1:1\"}}\n \
- {{name: eu-west, grpc_addr: \"1:2\", http_addr: \"1:2\"}}\nleader: us-east\n{yaml}"
))
.expect_err("non-dense ids rejected");
assert!(err.to_string().contains("dense"), "got {err}");
}
#[test]
fn shards_reject_unknown_node_and_bad_leader() {
let bad_node = load_topology_from_str(
"regions:\n - {name: us-east, grpc_addr: \"1:1\", http_addr: \"1:1\"}\nleader: us-east\n\
shards:\n - {id: 0, replicas: [{node: ghost}]}\n",
)
.expect_err("unknown node");
assert!(bad_node.to_string().contains("ghost"), "got {bad_node}");
let bad_leader = load_topology_from_str(
"regions:\n - {name: us-east, grpc_addr: \"1:1\", http_addr: \"1:1\"}\n \
- {name: eu-west, grpc_addr: \"1:2\", http_addr: \"1:2\"}\nleader: us-east\n\
shards:\n - {id: 0, leader: eu-west, replicas: [{node: us-east}]}\n",
)
.expect_err("leader not a replica");
assert!(
bad_leader
.to_string()
.contains("must be one of its replicas"),
"got {bad_leader}"
);
}
#[test]
fn shards_reject_same_node_port_collision() {
// us-east hosts shards 0 and 1 but both resolve to the same explicit addr.
let t = load_topology_from_str(
"regions:\n - {name: us-east, grpc_addr: \"127.0.0.1:9601\", http_addr: \"1:1\"}\n \
- {name: eu-west, grpc_addr: \"127.0.0.1:9611\", http_addr: \"1:2\"}\nleader: us-east\n\
shards:\n - {id: 0, replicas: [{node: us-east, grpc_addr: \"127.0.0.1:5000\"}, \
{node: eu-west}]}\n - {id: 1, leader: eu-west, replicas: [{node: us-east, \
grpc_addr: \"127.0.0.1:5000\"}, {node: eu-west}]}\n",
)
.expect("structural validation passes");
let err = t
.resolve_shard_groups()
.expect_err("port collision rejected");
assert!(err.to_string().contains("same grpc_addr"), "got {err}");
}
}