tidaldb/tidal/tests/m12p6_graph_persistence.rs
jx12n a0399550d6 feat(m12p6): persist HNSW graph + bounded SIGTERM drain — boot loads, no rebuild
Boot now LOADS the per-slot HNSW graph instead of rebuilding it. Clean
shutdown writes {data_dir}/vector/<kind>__<slot>.usearch; the next open loads
it when it matches the durable corpus (seconds), falling back to a full rebuild
only when the graph is missing/stale/corrupt. Eliminates the multi-minute boot
rebuild (~50-70 min at 1M/1536-D) that let the WAL compact past a restarting
node and triggered the reseed cascade.

Graceful SIGTERM now actually runs the close: bounded_drain caps the post-signal
HTTP drain (TIDAL_SHUTDOWN_DRAIN_MS, default 15s) then runs the deterministic
close regardless — sibling keep-alive connections no longer block the drain past
the k8s 60s grace into a SIGKILL (which cannot run Drop). ClusterNode and
ShardReplica::shutdown are now &self (db handle is an ArcSwapOption) so the close
fires even when a stuck connection task holds an Arc.

Fix USearch insert to be a true upsert (remove+add): it was unconditional add,
which a multi:false index rejects on a reseeding follower's post-snapshot WAL
replay -> applied_events stalls -> catch-up deadlock -> unrecoverable cluster.

Also: circuit-breaker peer last-contact tracking; real k3s 1536-dim deploy +
recall findings (recall@10 0.9869, read p99 8.71ms @ 200rps @ 100k) in
docs/profiling/m12-cluster-deploy-findings.md; new tidal-stress k8s jobs and
m12p6 graph-persistence + SIGTERM tier-3 regression tests.
2026-06-15 13:09:20 -06:00

153 lines
5.6 KiB
Rust

#![allow(clippy::unwrap_used, clippy::cast_precision_loss)]
//! m12p6 — persisted HNSW graph: boot LOADS the graph instead of rebuilding it.
//!
//! Before this milestone every process boot rebuilt the `USearch` HNSW index by
//! re-inserting every durable embedding at `ef_construction=400`, single-core —
//! ~5.5 min at 100k/128-D, ~50-70 min at 1M/1536-D. That outage is long enough
//! for the WAL to compact past a restarting node and trigger a reseed cascade.
//!
//! The fix saves each slot's graph to `{data_dir}/vector/<kind>__<slot>.usearch`
//! on clean shutdown and LOADS it on the next open when it matches the durable
//! corpus (fast, seconds), falling back to the rebuild only when the graph is
//! missing / stale / corrupt.
//!
//! # UAT Scenario
//!
//! ```
//! Given: A persistent db with an Item "content" HNSW slot and N indexed vectors
//! When: db.shutdown() — persists the graph to {data_dir}/vector/
//! Then: the per-slot .usearch graph file exists on disk
//! And: reopening the db serves the correct nearest neighbor from the LOADED
//! graph (no full rebuild required)
//! ```
use std::collections::HashMap;
use tidaldb::{
TempTidalHome, TidalDb,
schema::{DecaySpec, EntityId, EntityKind, SchemaBuilder, Window},
};
// dim 4000 ⇒ the dimension-aware brute-force→HNSW crossover floors to 1000, so a
// 1000-vector corpus selects the production USearch (HNSW) backend — the backend
// whose multi-minute rebuild this persistence eliminates. Keeping N at the floor
// keeps the test's one-time build (ef_construction=400) to a few seconds while
// still exercising the real HNSW save/load path end-to-end.
const DIM: usize = 4000;
const N: u64 = 1000;
/// `SplitMix64` deterministic generator (no seeded-RNG dev-dependency).
const fn splitmix64(state: &mut u64) -> u64 {
*state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = *state;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^ (z >> 31)
}
/// A deterministic, reproducible vector keyed by entity id.
fn vector_for(id: u64) -> Vec<f32> {
let mut state = id.wrapping_mul(0x2545_F491_4F6C_DD1D).wrapping_add(1);
(0..DIM)
.map(|_| {
let bits = (splitmix64(&mut state) >> 40) as u32;
(bits as f32 / 4096.0 / 4096.0) - 0.5
})
.collect()
}
fn embedding_schema() -> tidaldb::schema::Schema {
let mut b = SchemaBuilder::new();
let _ = b
.signal("view", EntityKind::Item, DecaySpec::Permanent)
.windows(&[Window::AllTime])
.velocity(false)
.add();
b.embedding_slot("content", EntityKind::Item, DIM);
b.build().unwrap()
}
/// On clean shutdown the HNSW graph is saved; on reopen it is LOADED (not
/// rebuilt) and serves the correct nearest neighbor.
#[test]
fn hnsw_graph_persisted_on_shutdown_and_loaded_on_reopen() {
let home = TempTidalHome::new().unwrap();
let graph_file = home.path().join("vector").join("item__content.usearch");
// First open: write N HNSW-backed embeddings, prove the live index serves a
// result, then clean shutdown (which persists the graph).
{
let db = TidalDb::builder()
.with_data_dir(home.path())
.with_schema(embedding_schema())
.open()
.unwrap();
for id in 1..=N {
db.write_item_with_metadata(EntityId::new(id), &HashMap::new())
.unwrap();
db.write_item_embedding(EntityId::new(id), &vector_for(id))
.unwrap();
}
// Live index serves item 42 as its own nearest neighbor.
let live = db.vector_search_items(&vector_for(42), 1, None).unwrap();
assert_eq!(live.len(), 1);
assert_eq!(live[0].id, 42, "live index must serve the nearest neighbor");
assert!(
!graph_file.exists(),
"graph file must NOT exist before shutdown (only the live in-memory \
index has been built so far)"
);
db.shutdown().unwrap();
}
// The clean shutdown must have persisted the HNSW graph to disk.
assert!(
graph_file.exists(),
"shutdown must persist the slot's HNSW graph to {}",
graph_file.display()
);
let graph_bytes = std::fs::metadata(&graph_file).unwrap().len();
assert!(
graph_bytes > 0,
"persisted HNSW graph must be non-empty, got {graph_bytes} bytes"
);
// Reopen: the open path must LOAD the persisted graph (its count matches the
// 1000 durable embeddings) and serve the same nearest neighbor — no full
// re-insert rebuild required.
let db = TidalDb::builder()
.with_data_dir(home.path())
.with_schema(embedding_schema())
.open()
.unwrap();
let near = db.vector_search_items(&vector_for(42), 1, None).unwrap();
assert_eq!(near.len(), 1, "reopened (loaded) index must serve a result");
assert_eq!(
near[0].id, 42,
"loaded graph must serve item 42's own vector as its nearest neighbor"
);
// A different query also resolves correctly against the loaded graph — the
// loaded graph is functionally identical to the one built before shutdown.
let near_7 = db.vector_search_items(&vector_for(7), 1, None).unwrap();
assert_eq!(
near_7[0].id, 7,
"loaded graph resolves a second query correctly"
);
// The loaded graph carries the full corpus (all 1000 vectors searchable).
let top = db.vector_search_items(&vector_for(42), 1000, None).unwrap();
assert_eq!(
top.len(),
N as usize,
"loaded graph must contain the entire persisted corpus"
);
db.shutdown().unwrap();
}