Resolves every finding in docs/reviews/M0-M10-code-review-2026-06-08-pass2.md across the engine, network, server, and CLI crates: session restore, replication/CRDT, WAL format and recovery, storage indexes, query/ranking executors, cohort/community governance, and scatter-gather routing. Adds regression tests: - review_pass2_creator_search_filter - review_pass2_d_replication - review_pass2_query_for_session - review_pass2_storage_indexes_bitmap_cache - review_pass2_zone_a_sessions Verified: cargo clippy -D warnings and full test suite green across all crates.
1294 lines
55 KiB
Rust
1294 lines
55 KiB
Rust
//! Entity state rebuild from durable storage and periodic checkpoint thread.
|
|
|
|
use std::{
|
|
collections::HashMap,
|
|
path::PathBuf,
|
|
sync::{
|
|
Arc,
|
|
atomic::{AtomicBool, AtomicU64, Ordering},
|
|
},
|
|
time::Duration,
|
|
};
|
|
|
|
use roaring::RoaringBitmap;
|
|
|
|
use super::{metadata::deserialize_metadata, metrics::MetricsState, storage_box::StorageBox};
|
|
use crate::{
|
|
cohort::CohortSignalLedger,
|
|
query::suggest::SuggestionIndex,
|
|
replication::{ShardId, state::ReplicationState},
|
|
schema::{EntityId, TidalError, Timestamp},
|
|
signals::{DEFAULT_MAX_SIGNAL_ENTRIES, SignalLedger, trim_cold_entries},
|
|
storage::{
|
|
StorageEngine, Tag, encode_key,
|
|
indexes::{bitmap::BitmapIndex, range::RangeIndex},
|
|
},
|
|
};
|
|
|
|
/// Suffix for the single follower-replication-state checkpoint row.
|
|
const REPLICATION_STATE_SUFFIX: &[u8] = b"hwm";
|
|
|
|
/// Storage key for the follower replication high-water-mark checkpoint.
|
|
///
|
|
/// One row at entity ID 0 under [`Tag::ReplicationState`], holding the
|
|
/// JSON-serialized per-shard `applied_seqno`.
|
|
#[must_use]
|
|
fn replication_state_key() -> Vec<u8> {
|
|
encode_key(
|
|
EntityId::new(0),
|
|
Tag::ReplicationState,
|
|
REPLICATION_STATE_SUFFIX,
|
|
)
|
|
}
|
|
|
|
/// Durably persist the follower's per-shard replication high-water-mark.
|
|
///
|
|
/// Serializes [`ReplicationState::to_checkpoint_bytes`] and commits it (with an
|
|
/// fsync via `flush`) so a follower crash cannot reset its idempotency boundary
|
|
/// to 0. Called from the periodic checkpoint thread and from shutdown. Failure
|
|
/// is non-fatal at the call site: a missed checkpoint means the next open
|
|
/// restores an older high-water-mark (correct, just re-applies a few re-shipped
|
|
/// segments idempotently), but it still signals a failing disk.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns `TidalError::Storage` if the write or flush fails.
|
|
pub(super) fn checkpoint_replication_state(
|
|
storage: &dyn StorageEngine,
|
|
replication_state: &ReplicationState,
|
|
) -> crate::Result<()> {
|
|
let bytes = replication_state.to_checkpoint_bytes();
|
|
let mut batch = crate::storage::WriteBatch::new();
|
|
batch.put(replication_state_key(), bytes);
|
|
storage.write_batch(batch)?;
|
|
storage.flush()?;
|
|
Ok(())
|
|
}
|
|
|
|
/// Restore a follower [`ReplicationState`] from its durable checkpoint.
|
|
///
|
|
/// Reads the persisted per-shard high-water-mark and reconstructs a
|
|
/// `ReplicationState` tracking `shards`. Shards absent from the checkpoint (or
|
|
/// the first-boot / corrupt-bytes cases) start at 0 — replaying from 0 is
|
|
/// always correct, just slower. The returned state is the SINGLE Arc the
|
|
/// receiver advances and the lag gauge reads (obs-REPL-1): the caller must not
|
|
/// build a second `ReplicationState` for the same node.
|
|
#[must_use]
|
|
pub(super) fn restore_replication_state(
|
|
storage: &dyn StorageEngine,
|
|
shards: &[ShardId],
|
|
) -> ReplicationState {
|
|
match storage.get(&replication_state_key()) {
|
|
Ok(Some(bytes)) => {
|
|
let restored = ReplicationState::from_checkpoint_bytes(&bytes, shards);
|
|
tracing::info!(
|
|
shards = shards.len(),
|
|
"follower replication high-water-mark restored from checkpoint"
|
|
);
|
|
restored
|
|
}
|
|
Ok(None) => ReplicationState::new(shards),
|
|
Err(e) => {
|
|
tracing::warn!(
|
|
error = %e,
|
|
"replication-state checkpoint read failed; \
|
|
follower starts at applied seqno 0 (re-applies re-shipped segments)"
|
|
);
|
|
ReplicationState::new(shards)
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── Index health metrics handles ────────────────────────────────────────────
|
|
|
|
/// Handles to live index instances for the periodic checkpoint-thread refresh.
|
|
///
|
|
/// Always carries the text-index liveness handles (item + creator syncer and
|
|
/// the shared unhealthy latch) so the thread can poll text-search health and
|
|
/// schedule a resync rebuild regardless of the `metrics` feature — text-search
|
|
/// liveness is a correctness signal, not a monitoring nicety.
|
|
///
|
|
/// When the `metrics` feature is enabled it additionally carries `Arc`/clone
|
|
/// references to the embedding registry and bitmap indexes so the thread can
|
|
/// read their current sizes for the Prometheus gauges.
|
|
pub(super) struct IndexMetricsHandles {
|
|
/// Item text index, for `is_healthy()` polling and rebuild scheduling.
|
|
pub text_index: Option<Arc<crate::text::TextIndex>>,
|
|
/// Creator text index, polled identically to the item index.
|
|
pub creator_text_index: Option<Arc<crate::text::TextIndex>>,
|
|
/// Latched true by this thread once either syncer reports unhealthy; read by
|
|
/// `TidalDb::health_check` to flip the DB to degraded.
|
|
pub text_unhealthy: Arc<AtomicBool>,
|
|
#[cfg(feature = "metrics")]
|
|
pub embedding_registry:
|
|
Arc<std::sync::RwLock<crate::storage::vector::registry::EmbeddingSlotRegistry>>,
|
|
#[cfg(feature = "metrics")]
|
|
pub bitmap_category: crate::storage::indexes::bitmap::BitmapIndex,
|
|
#[cfg(feature = "metrics")]
|
|
pub bitmap_format: crate::storage::indexes::bitmap::BitmapIndex,
|
|
#[cfg(feature = "metrics")]
|
|
pub bitmap_creator: crate::storage::indexes::bitmap::BitmapIndex,
|
|
#[cfg(feature = "metrics")]
|
|
pub bitmap_tag: crate::storage::indexes::bitmap::BitmapIndex,
|
|
}
|
|
|
|
/// Rebuild in-memory entity state from durable storage on restart.
|
|
///
|
|
/// Scans the users keyspace for relationship edges. Populates:
|
|
/// 1. `user_state.blocked` from `RelationshipType::Blocks` edges
|
|
/// 2. `user_state.seen` (hidden items) from `RelationshipType::Hide` edges
|
|
/// 3. `user_state.follows` from `RelationshipType::Follows` edges
|
|
/// 4. `interaction_ledger` from `RelationshipType::InteractionWeight` edges
|
|
///
|
|
/// The `creator_items` bitmap is NOT rebuilt here: it is populated by the shared
|
|
/// [`ItemIndexes::index`] (called from [`rebuild_item_indexes`]) so the live
|
|
/// write path and the rebuild path use one implementation and cannot drift.
|
|
///
|
|
/// For ephemeral mode, all engines are empty, so this is effectively a no-op.
|
|
pub(super) fn rebuild_entity_state(
|
|
storage: &StorageBox,
|
|
user_state: &crate::entities::UserStateIndex,
|
|
interaction_ledger: &crate::entities::InteractionLedger,
|
|
) -> crate::Result<()> {
|
|
use crate::{
|
|
entities::relationship::{
|
|
RelationshipType, deserialize_relationship_value, parse_relationship_to,
|
|
},
|
|
storage::keys::parse_key,
|
|
};
|
|
|
|
// Scan the users keyspace for all relationship edges.
|
|
// The relationship key format is:
|
|
// [from_entity_id: 8 BE][0x00][Tag::Rel (0x04)][rel_type: 1][to_entity_id: 8 BE]
|
|
// We scan with an empty prefix to get all keys, then filter for Tag::Rel.
|
|
let mut rel_count = 0u64;
|
|
for entry in storage.users_engine().scan_prefix(&[]) {
|
|
let (key, value) = entry.map_err(TidalError::from)?;
|
|
|
|
// Only process relationship keys (Tag::Rel = 0x04).
|
|
if let Some((from_id, Tag::Rel, suffix)) = parse_key(&key) {
|
|
// suffix = [rel_type: 1 byte][to_entity_id: 8 BE]
|
|
if suffix.is_empty() {
|
|
continue;
|
|
}
|
|
let rel_type_byte = suffix[0];
|
|
let Some(rel_type) = RelationshipType::from_byte(rel_type_byte) else {
|
|
continue;
|
|
};
|
|
let Some(to_id) = parse_relationship_to(&key) else {
|
|
continue;
|
|
};
|
|
let from_id_u64 = from_id.as_u64();
|
|
|
|
match rel_type {
|
|
RelationshipType::Blocks => {
|
|
user_state.add_block_creator(from_id_u64, to_id.as_u64());
|
|
rel_count += 1;
|
|
}
|
|
RelationshipType::Hide => {
|
|
user_state.add_hide_item(from_id_u64, to_id.as_u64());
|
|
rel_count += 1;
|
|
}
|
|
RelationshipType::Follows => {
|
|
// Forward: user -> followed creator
|
|
user_state.add_follow(from_id_u64, to_id.as_u64());
|
|
// Reverse: creator -> follower users
|
|
user_state.add_creator_follower(to_id.as_u64(), from_id_u64);
|
|
rel_count += 1;
|
|
}
|
|
RelationshipType::InteractionWeight => {
|
|
// Reconstruct interaction weight from the stored edge value.
|
|
if let Some((weight, ts_nanos)) = deserialize_relationship_value(&value) {
|
|
interaction_ledger.record(from_id_u64, to_id.as_u64(), weight, ts_nanos);
|
|
rel_count += 1;
|
|
}
|
|
}
|
|
RelationshipType::Mute => {
|
|
// Mute edges do not have in-memory state (yet).
|
|
rel_count += 1;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
if rel_count > 0 {
|
|
tracing::info!(
|
|
relationships = rel_count,
|
|
"entity state rebuilt from durable storage"
|
|
);
|
|
}
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Classify a storage key as an indexable item-metadata row.
|
|
///
|
|
/// Returns the row's [`EntityId`] iff the key is a `Tag::Meta` key whose suffix
|
|
/// is *not* an `EMB:` embedding row (those share `Tag::Meta` but carry a
|
|
/// serialized vector, not searchable/index-able metadata). This is the
|
|
/// load-bearing "which rows are item metadata" rule that every item-scan loop
|
|
/// (`rebuild_item_indexes`, the text-index rebuilds, the suggestion rebuild)
|
|
/// must apply identically — factoring it here makes it impossible for one loop
|
|
/// to silently drift (e.g. forget the `EMB:` skip and index raw vectors).
|
|
fn item_meta_entity_id(key: &[u8]) -> Option<EntityId> {
|
|
use crate::storage::keys::parse_key;
|
|
|
|
match parse_key(key) {
|
|
Some((entity_id, Tag::Meta, suffix)) if !suffix.starts_with(b"EMB:") => Some(entity_id),
|
|
_ => None,
|
|
}
|
|
}
|
|
|
|
/// Borrowed references to every in-memory item index, bundled so the write path
|
|
/// and the restart-rebuild path share one indexing implementation.
|
|
///
|
|
/// The two paths previously parsed the same metadata keys
|
|
/// (`category`/`format`/`creator_id`/`tags`/`duration`/`created_at`)
|
|
/// independently and had already drifted — the write path defaulted a missing
|
|
/// `created_at` and populated `creator_items`, the rebuild path did neither.
|
|
/// Drift here is a correctness bug: an item's recency ordering or
|
|
/// creator-following set silently changes across a restart. Routing both paths
|
|
/// through [`index_item_metadata`] makes divergence impossible by construction.
|
|
pub(super) struct ItemIndexes<'a> {
|
|
pub category: &'a BitmapIndex,
|
|
pub format: &'a BitmapIndex,
|
|
pub creator: &'a BitmapIndex,
|
|
pub tag: &'a BitmapIndex,
|
|
pub duration: &'a RangeIndex<u32>,
|
|
pub created_at: &'a RangeIndex<u64>,
|
|
pub creator_items: &'a crate::entities::CreatorItemsBitmap,
|
|
}
|
|
|
|
impl ItemIndexes<'_> {
|
|
/// Scrub every prior index entry for `id_u32` before re-indexing it under
|
|
/// new metadata.
|
|
///
|
|
/// The bitmap/range indexes ([`BitmapIndex`]/[`RangeIndex`]) only ever
|
|
/// *insert*, so on an item overwrite a changed value would leave the entity
|
|
/// indexed under BOTH the old and the new value — a phantom hit on
|
|
/// RETRIEVE/SEARCH metadata filters and a corrupted recency order (W22). The
|
|
/// scrub is value-agnostic: [`delete_entity`](crate::storage::indexes::RangeIndex::delete_entity)
|
|
/// removes *all* of the entity's entries and is a no-op when absent.
|
|
/// `creator_items` is keyed by creator id rather than item id, so the OLD
|
|
/// creator is taken from `old_metadata`. Call this on overwrite immediately
|
|
/// before [`index`](Self::index); the durable store is overwritten
|
|
/// separately, so only the live in-memory index needs reconciling.
|
|
pub fn scrub(&self, id_u32: u32, old_metadata: &HashMap<String, String>) {
|
|
self.category.delete_entity(id_u32);
|
|
self.format.delete_entity(id_u32);
|
|
self.creator.delete_entity(id_u32);
|
|
self.tag.delete_entity(id_u32);
|
|
self.duration.delete_entity(id_u32);
|
|
self.created_at.delete_entity(id_u32);
|
|
if let Some(creator_id) = old_metadata
|
|
.get("creator_id")
|
|
.and_then(|v| v.parse::<u64>().ok())
|
|
{
|
|
self.creator_items.remove_item(creator_id, id_u32);
|
|
}
|
|
}
|
|
|
|
/// Index one item's metadata into every in-memory index.
|
|
///
|
|
/// `metadata` must already carry a `created_at` value (the write path
|
|
/// materializes the `Timestamp::now()` default into the persisted metadata
|
|
/// *before* calling this, so the rebuild path reads the identical value and
|
|
/// reproduces the exact recency ordering). A row whose persisted metadata
|
|
/// genuinely lacks `created_at` (a legacy row written before the default was
|
|
/// materialized) simply has no recency-index entry — the same as before —
|
|
/// but is otherwise indexed identically by both paths.
|
|
pub fn index(&self, id_u32: u32, metadata: &HashMap<String, String>) {
|
|
// Bitmap indexes.
|
|
if let Some(val) = metadata.get("category") {
|
|
self.category.insert(id_u32, val);
|
|
}
|
|
if let Some(val) = metadata.get("format") {
|
|
self.format.insert(id_u32, val);
|
|
}
|
|
if let Some(val) = metadata.get("creator_id") {
|
|
self.creator.insert(id_u32, val);
|
|
// Populate creator-items for the `following` profile / `unblocked`
|
|
// predicate. This was missing from the rebuild path before the
|
|
// shared indexer, so creator-following queries returned nothing
|
|
// after a restart until each item was rewritten.
|
|
if let Ok(creator_id) = val.parse::<u64>() {
|
|
self.creator_items.add_item(creator_id, id_u32);
|
|
}
|
|
}
|
|
if let Some(tags) = metadata.get("tags") {
|
|
for tag in tags.split(',') {
|
|
let tag = tag.trim();
|
|
if !tag.is_empty() {
|
|
self.tag.insert(id_u32, tag);
|
|
}
|
|
}
|
|
}
|
|
|
|
// Range indexes.
|
|
if let Some(val) = metadata.get("duration") {
|
|
match val.parse::<u32>() {
|
|
Ok(dur) => self.duration.insert(id_u32, dur),
|
|
Err(e) => tracing::warn!(
|
|
entity_id = u64::from(id_u32),
|
|
value = %val,
|
|
error = %e,
|
|
"failed to parse 'duration' metadata; item will not be indexed by duration"
|
|
),
|
|
}
|
|
}
|
|
if let Some(val) = metadata.get("created_at") {
|
|
match val.parse::<u64>() {
|
|
Ok(ts) => self.created_at.insert(id_u32, ts),
|
|
Err(e) => tracing::warn!(
|
|
entity_id = u64::from(id_u32),
|
|
value = %val,
|
|
error = %e,
|
|
"failed to parse 'created_at' metadata; item will not be indexed by recency"
|
|
),
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Rebuild item indexes (universe bitmap + bitmap/range indexes) from durable storage.
|
|
///
|
|
/// Scans the items keyspace for `Tag::Meta` keys and populates the in-memory
|
|
/// universe bitmap and all item indexes (category, format, creator, tags,
|
|
/// duration, `created_at`, plus the creator-items bitmap) from the persisted
|
|
/// metadata via the shared [`ItemIndexes::index`]. This ensures `/health`
|
|
/// reports the correct item count and queries work immediately after a restart
|
|
/// without rewriting items.
|
|
///
|
|
/// Because the write path now persists a defaulted `created_at` into metadata,
|
|
/// the rebuild reproduces the exact recency ordering the live path produced.
|
|
///
|
|
/// For ephemeral mode the engine is empty, so this is a no-op.
|
|
#[allow(clippy::cast_possible_truncation)]
|
|
pub(super) fn rebuild_item_indexes(
|
|
storage: &StorageBox,
|
|
universe: &mut RoaringBitmap,
|
|
indexes: &ItemIndexes<'_>,
|
|
) -> crate::Result<()> {
|
|
let mut count = 0u64;
|
|
let scan_start = std::time::Instant::now();
|
|
|
|
for entry in storage.items_engine().scan_prefix(&[]) {
|
|
let (key, value) = entry.map_err(TidalError::from)?;
|
|
|
|
if let Some(entity_id) = item_meta_entity_id(&key) {
|
|
// The write path (db/items.rs) REJECTS any item id > u32::MAX before
|
|
// it ever lands in storage, so the only way one reaches here is a
|
|
// corrupt or hostile durable store. Mirror the write path's refusal:
|
|
// skip the row entirely rather than truncate it into a colliding u32
|
|
// slot, which would alias a real lower id and manufacture phantom
|
|
// query hits. Failing safe on corrupt input beats silently corrupting
|
|
// the in-memory universe/indexes.
|
|
if entity_id.as_u64() > u64::from(u32::MAX) {
|
|
tracing::warn!(
|
|
entity_id = entity_id.as_u64(),
|
|
"entity ID exceeds u32::MAX during rebuild; skipping \
|
|
(write path never persists such ids)"
|
|
);
|
|
continue;
|
|
}
|
|
// In range by the guard above, so the narrowing is exact.
|
|
let id_u32 = entity_id.as_u64() as u32;
|
|
let meta = deserialize_metadata(&value);
|
|
|
|
// Universe bitmap.
|
|
universe.insert(id_u32);
|
|
|
|
// All other indexes (shared with the write path).
|
|
indexes.index(id_u32, &meta);
|
|
|
|
count += 1;
|
|
if count.is_multiple_of(10_000) {
|
|
tracing::info!(rebuilt = count, "item index rebuild in progress");
|
|
}
|
|
}
|
|
}
|
|
|
|
if count > 0 {
|
|
tracing::info!(
|
|
items = count,
|
|
elapsed_ms = scan_start.elapsed().as_millis(),
|
|
"item indexes rebuilt from durable storage"
|
|
);
|
|
}
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Background thread body: checkpoint signal state to storage every 30 seconds.
|
|
///
|
|
/// Checkpoints the global signal ledger, the cohort signal ledger, the
|
|
/// per-contributor community ledger, the co-engagement index, and the per-user
|
|
/// preference vectors (each writes its own `WriteBatch`). Every one of these is
|
|
/// derived state with no WAL backing for its per-user/per-contributor identity,
|
|
/// so a periodic checkpoint is what bounds their post-crash loss window to the
|
|
/// checkpoint interval instead of "everything since the last clean shutdown."
|
|
///
|
|
/// After each successful checkpoint, compacts WAL segments that are fully
|
|
/// covered by the checkpoint. Compaction failure is non-fatal: a warning is
|
|
/// logged and the next checkpoint cycle will retry.
|
|
///
|
|
/// Polls the shutdown flag every 500ms so the thread exits promptly when
|
|
/// `shutdown_inner()` is called. Only runs in persistent mode (ephemeral opens
|
|
/// never spawn this thread).
|
|
///
|
|
/// The `Arc` arguments are intentionally passed by value: the thread must own
|
|
/// them for its entire lifetime (references cannot satisfy the `'static` bound
|
|
/// required by `std::thread::spawn`).
|
|
#[allow(
|
|
clippy::needless_pass_by_value,
|
|
clippy::too_many_arguments,
|
|
clippy::too_many_lines
|
|
)]
|
|
pub(super) fn run_checkpoint_thread(
|
|
shutdown: Arc<AtomicBool>,
|
|
ledger: Arc<SignalLedger>,
|
|
cohort_ledger: Arc<CohortSignalLedger>,
|
|
community_ledger: Arc<crate::governance::CommunityLedger>,
|
|
co_engagement: Arc<crate::entities::CoEngagementIndex>,
|
|
preference_vectors: Arc<crate::entities::PreferenceVectors>,
|
|
replication_state: Arc<ReplicationState>,
|
|
storage: Box<dyn StorageEngine + Send + Sync>,
|
|
last_wal_seq: Arc<AtomicU64>,
|
|
wal_dir: Option<PathBuf>,
|
|
metrics: Arc<MetricsState>,
|
|
index_handles: IndexMetricsHandles,
|
|
) {
|
|
const CHECKPOINT_INTERVAL: Duration = Duration::from_secs(30);
|
|
/// Index health metrics (Tantivy, `USearch`, bitmap) refresh every 10s -- 3x more
|
|
/// frequent than checkpoints so operators get near-real-time index visibility.
|
|
const INDEX_METRICS_INTERVAL: Duration = Duration::from_secs(10);
|
|
const POLL_INTERVAL: Duration = Duration::from_millis(500);
|
|
|
|
let mut elapsed = Duration::ZERO;
|
|
let mut index_metrics_elapsed = Duration::ZERO;
|
|
loop {
|
|
std::thread::sleep(POLL_INTERVAL);
|
|
if shutdown.load(Ordering::Acquire) {
|
|
break;
|
|
}
|
|
elapsed += POLL_INTERVAL;
|
|
index_metrics_elapsed += POLL_INTERVAL;
|
|
|
|
// Every 10s (3x faster than checkpoint): poll text-index liveness — a
|
|
// correctness signal — and, when the `metrics` feature is on, also
|
|
// refresh the index-size gauges. The text-health poll runs regardless of
|
|
// feature so search liveness is always observed.
|
|
if index_metrics_elapsed >= INDEX_METRICS_INTERVAL {
|
|
index_metrics_elapsed = Duration::ZERO;
|
|
poll_text_index_health(&index_handles, storage.as_ref());
|
|
#[cfg(feature = "metrics")]
|
|
refresh_index_metrics(&index_handles, &metrics);
|
|
}
|
|
|
|
if elapsed >= CHECKPOINT_INTERVAL {
|
|
elapsed = Duration::ZERO;
|
|
|
|
// Update signal hot entries gauge.
|
|
#[cfg(feature = "metrics")]
|
|
{
|
|
metrics
|
|
.signal_hot_entries
|
|
.store(ledger.entries().len() as u64, Ordering::Relaxed);
|
|
}
|
|
|
|
// (index health metrics refreshed every 10s in the block above)
|
|
|
|
// Trim signal ledger if over the memory budget (5M entries ~5.4 GB).
|
|
let entry_count = ledger.entries().len();
|
|
if entry_count > DEFAULT_MAX_SIGNAL_ENTRIES {
|
|
tracing::info!(
|
|
entry_count,
|
|
max_entries = DEFAULT_MAX_SIGNAL_ENTRIES,
|
|
"signal ledger exceeds memory budget — trimming cold entries"
|
|
);
|
|
let evicted = trim_cold_entries(ledger.entries(), DEFAULT_MAX_SIGNAL_ENTRIES);
|
|
tracing::info!(
|
|
evicted,
|
|
remaining = ledger.entries().len(),
|
|
"signal ledger trim complete"
|
|
);
|
|
}
|
|
|
|
let seq = last_wal_seq.load(Ordering::Relaxed);
|
|
let meta = crate::signals::checkpoint::CheckpointMeta {
|
|
checkpoint_time_ns: Timestamp::now().as_nanos(),
|
|
wal_sequence: seq,
|
|
payload_hash: [0u8; 32], // computed by checkpoint()
|
|
};
|
|
if let Err(e) = ledger.checkpoint(storage.as_ref(), meta) {
|
|
tracing::error!(error = %e, "periodic signal checkpoint failed");
|
|
metrics
|
|
.checkpoint_failures_total
|
|
.fetch_add(1, Ordering::Relaxed);
|
|
} else {
|
|
tracing::debug!("periodic signal checkpoint written");
|
|
|
|
// Update checkpoint age metric.
|
|
#[cfg(feature = "metrics")]
|
|
{
|
|
let now_ns = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.unwrap_or_default()
|
|
.as_nanos() as u64;
|
|
// Release pairs with the Acquire load in
|
|
// `MetricsState::is_degraded` (and `health_check`): a reader
|
|
// that observes this fresh timestamp must also observe that
|
|
// the checkpoint it timestamps actually completed, so the
|
|
// staleness derivation never trips on a half-published value.
|
|
metrics.last_checkpoint_ns.store(now_ns, Ordering::Release);
|
|
}
|
|
|
|
// Compact WAL segments covered by the checkpoint.
|
|
// This runs AFTER the checkpoint is durable, so deleted
|
|
// segments are guaranteed to be redundant. The WAL writer thread
|
|
// is still live here, so we MUST use the online variant, which
|
|
// never deletes the segment the writer is appending to (deleting
|
|
// it would unlink the inode out from under the open FD and lose
|
|
// post-checkpoint writes on Linux).
|
|
//
|
|
// NOTE: the WAL checkpoint *marker* is deliberately NOT advanced
|
|
// here — only on clean shutdown (lifecycle.rs) and via the
|
|
// test-only force_replication_checkpoint. Advancing it on the racy
|
|
// periodic snapshot would trade the existing safe over-count
|
|
// behavior (a crash replays a few already-materialized events) for
|
|
// a potential lost-write (the marker would suppress replay of an
|
|
// event the concurrent snapshot missed). For eventually-consistent
|
|
// signal state, over-count self-heals via decay; an acked lost
|
|
// write does not. The follower's replicated aggregate is therefore
|
|
// fully recoverable from checkpoint + WAL replay, and its persisted
|
|
// high-water-mark (checkpointed below) is the leader-seqno
|
|
// idempotency boundary that re-shipped segments are gated against.
|
|
if let Some(ref dir) = wal_dir
|
|
&& seq > 0
|
|
{
|
|
match crate::wal::compaction::compact_wal_online(dir, seq) {
|
|
Ok(result) => {
|
|
#[cfg(feature = "metrics")]
|
|
{
|
|
metrics
|
|
.wal_compacted_segments_total
|
|
.fetch_add(result.segments_deleted as u64, Ordering::Relaxed);
|
|
}
|
|
let _ = result; // suppress unused warning when metrics disabled
|
|
}
|
|
Err(e) => {
|
|
tracing::warn!(error = %e, "WAL compaction after checkpoint failed");
|
|
}
|
|
}
|
|
|
|
// Update WAL lag bytes: sum only segments NOT yet compacted
|
|
// by the checkpoint, which is exactly what the gauge's doc
|
|
// ("Bytes of WAL segments not yet compacted") promises. `seq`
|
|
// is the checkpoint sequence we just compacted at.
|
|
#[cfg(feature = "metrics")]
|
|
{
|
|
let lag = compute_wal_lag_bytes(dir, seq);
|
|
metrics.wal_lag_bytes.store(lag, Ordering::Relaxed);
|
|
}
|
|
}
|
|
}
|
|
// Checkpoint cohort signal state with the same meta.
|
|
if cohort_ledger.entry_count() > 0
|
|
&& let Err(e) = cohort_ledger.checkpoint(storage.as_ref(), meta)
|
|
{
|
|
tracing::error!(error = %e, "periodic cohort checkpoint failed");
|
|
}
|
|
|
|
// Checkpoint the derived per-user / per-contributor ledgers that have
|
|
// no WAL backstop. A periodic checkpoint bounds their post-crash loss
|
|
// window to the interval instead of "everything since the last clean
|
|
// shutdown" (the community ledger is the SOLE durable copy of
|
|
// per-contributor identity, so this is its only periodic durability).
|
|
//
|
|
// The cohort ledger is already checkpointed above with `meta`, so the
|
|
// shared helper is called WITHOUT a cohort argument here to avoid a
|
|
// double cohort write in the same cycle.
|
|
let _ = checkpoint_secondary_ledgers(
|
|
storage.as_ref(),
|
|
None,
|
|
community_ledger.as_ref(),
|
|
co_engagement.as_ref(),
|
|
preference_vectors.as_ref(),
|
|
);
|
|
|
|
// Persist the follower replication high-water-mark (BLOCKER 6): a
|
|
// restarted follower must restore its idempotency boundary, not
|
|
// reset to 0 and double-count every re-shipped segment. Non-fatal —
|
|
// a missed checkpoint restores an older (still-correct) boundary.
|
|
if let Err(e) =
|
|
checkpoint_replication_state(storage.as_ref(), replication_state.as_ref())
|
|
{
|
|
tracing::error!(error = %e, "periodic replication-state checkpoint failed");
|
|
}
|
|
}
|
|
}
|
|
|
|
// `metrics` is only read inside `#[cfg(feature = "metrics")]` blocks above;
|
|
// `index_handles` is always read by `poll_text_index_health`.
|
|
let _ = &metrics;
|
|
}
|
|
|
|
/// Checkpoint the *secondary derived* ledgers — cohort (optional), community,
|
|
/// co-engagement, and preference vectors — into `storage`, with ONE consistent
|
|
/// set of guards and error handling.
|
|
///
|
|
/// This is the single shared implementation behind every secondary-checkpoint
|
|
/// site: the periodic checkpoint thread (above), shutdown
|
|
/// ([`TidalDb::shutdown_inner`](crate::db::TidalDb)), and backup
|
|
/// ([`TidalDb::create_backup`](crate::db::TidalDb)). Previously each site
|
|
/// open-coded the same four-checkpoint block with subtly different guards
|
|
/// (`entry_count() > 0` / `edge_count() > 0` / `!is_empty()` / none) and
|
|
/// divergent error handling; a new derived ledger or a changed guard had to be
|
|
/// edited in four places or one path silently skipped a checkpoint. Centralizing
|
|
/// it makes that drift impossible.
|
|
///
|
|
/// `cohort` is `Some((ledger, meta))` only where the cohort write is wanted in
|
|
/// this pass (shutdown / backup). The periodic thread passes `None` because it
|
|
/// already checkpoints the cohort ledger with its own `meta` earlier in the same
|
|
/// cycle — passing `Some` here would write it twice.
|
|
///
|
|
/// None of these ledgers has a WAL backstop for its per-user/per-contributor
|
|
/// identity, so this checkpoint is what bounds their post-crash loss window. Each
|
|
/// is independent: a failure is logged and the remaining checkpoints still run.
|
|
/// The FIRST error is returned so callers that must surface it (shutdown folds it
|
|
/// into its `first_err`) can, while callers that only need best-effort (backup,
|
|
/// periodic) discard it.
|
|
pub(super) fn checkpoint_secondary_ledgers(
|
|
storage: &dyn StorageEngine,
|
|
cohort: Option<(
|
|
&CohortSignalLedger,
|
|
crate::signals::checkpoint::CheckpointMeta,
|
|
)>,
|
|
community_ledger: &crate::governance::CommunityLedger,
|
|
co_engagement: &crate::entities::CoEngagementIndex,
|
|
preference_vectors: &crate::entities::PreferenceVectors,
|
|
) -> Option<TidalError> {
|
|
let mut first_err: Option<TidalError> = None;
|
|
if let Some((cohort_ledger, meta)) = cohort
|
|
&& cohort_ledger.entry_count() > 0
|
|
&& let Err(e) = cohort_ledger.checkpoint(storage, meta)
|
|
{
|
|
tracing::error!(error = %e, "cohort checkpoint failed");
|
|
first_err.get_or_insert(e);
|
|
}
|
|
// The community ledger is the SOLE durable copy of per-contributor identity,
|
|
// so it is always checkpointed (no emptiness guard): an empty checkpoint is a
|
|
// cheap no-op, but skipping it on a transiently-empty in-memory map could
|
|
// leave a stale durable snapshot.
|
|
if let Err(e) = community_ledger.checkpoint(storage) {
|
|
tracing::error!(error = %e, "community checkpoint failed");
|
|
first_err.get_or_insert(e);
|
|
}
|
|
if co_engagement.edge_count() > 0
|
|
&& let Err(e) = co_engagement.checkpoint(storage)
|
|
{
|
|
tracing::error!(error = %e, "co-engagement checkpoint failed");
|
|
first_err.get_or_insert(e);
|
|
}
|
|
if !preference_vectors.is_empty()
|
|
&& let Err(e) = preference_vectors.checkpoint(storage)
|
|
{
|
|
tracing::error!(error = %e, "preference-vector checkpoint failed");
|
|
first_err.get_or_insert(e);
|
|
}
|
|
first_err
|
|
}
|
|
|
|
/// Poll text-index liveness and schedule a resync rebuild when unhealthy.
|
|
///
|
|
/// Runs every 10s on the checkpoint thread regardless of the `metrics` feature
|
|
/// (text-search liveness is a correctness signal). Reads
|
|
/// [`TextIndex::is_healthy`](crate::text::TextIndex::is_healthy) for both the
|
|
/// item and creator syncers. When either is unhealthy it latches
|
|
/// `handles.text_unhealthy` (so `health_check` flips the DB to degraded) and
|
|
/// attempts to rebuild the ITEM index from the durable item store the thread
|
|
/// already owns — the only store handle available to this thread. The creator
|
|
/// index resync is left to the foreground `flush_creator_text_index`/health
|
|
/// path; the latch still surfaces its unhealth.
|
|
///
|
|
/// A rebuild that itself fails leaves the latch set (the next reopen or a manual
|
|
/// resync recovers); a rebuild that succeeds clears the latch for the item side,
|
|
/// and the syncer's own `is_healthy()` flag governs from then on.
|
|
fn poll_text_index_health(handles: &IndexMetricsHandles, items_engine: &dyn StorageEngine) {
|
|
let item_ok = handles
|
|
.text_index
|
|
.as_ref()
|
|
.is_none_or(|idx| idx.is_healthy());
|
|
let creator_ok = handles
|
|
.creator_text_index
|
|
.as_ref()
|
|
.is_none_or(|idx| idx.is_healthy());
|
|
|
|
if item_ok && creator_ok {
|
|
return; // healthy -> nothing to do (do not clear: only a rebuild clears)
|
|
}
|
|
|
|
// Latch so health_check / the HTTP surface report degraded.
|
|
handles.text_unhealthy.store(true, Ordering::Release);
|
|
tracing::warn!(
|
|
item_ok,
|
|
creator_ok,
|
|
"text index syncer unhealthy; scheduling rebuild from durable storage"
|
|
);
|
|
|
|
// Resync the item index from the durable store this thread owns.
|
|
if !item_ok && let Some(idx) = handles.text_index.as_ref() {
|
|
match rebuild_text_index_from_engine(idx, items_engine) {
|
|
Ok(indexed) => {
|
|
tracing::info!(indexed, "item text index rebuilt; clearing unhealthy latch");
|
|
// Clear only if the creator side is also healthy; otherwise leave
|
|
// the latch set to reflect the still-unhealthy creator syncer.
|
|
if creator_ok {
|
|
handles.text_unhealthy.store(false, Ordering::Release);
|
|
}
|
|
}
|
|
Err(e) => {
|
|
tracing::error!(error = %e, "item text index rebuild failed; remains degraded");
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Rebuild a Tantivy text index from a durable items storage engine.
|
|
///
|
|
/// Scans `Tag::Meta` keys (skipping `EMB:` embedding rows, whose values are
|
|
/// vectors not metadata), deserializes each item's metadata, and feeds them to
|
|
/// [`TextIndex::rebuild_from`](crate::text::TextIndex::rebuild_from) which clears
|
|
/// and re-indexes atomically. Returns the number of items re-indexed.
|
|
fn rebuild_text_index_from_engine(
|
|
idx: &crate::text::TextIndex,
|
|
items_engine: &dyn StorageEngine,
|
|
) -> crate::Result<usize> {
|
|
let mut items: Vec<(crate::schema::EntityId, HashMap<String, String>)> = Vec::new();
|
|
for entry in items_engine.scan_prefix(&[]) {
|
|
let (key, value) = entry.map_err(TidalError::from)?;
|
|
if let Some(entity_id) = item_meta_entity_id(&key) {
|
|
items.push((entity_id, deserialize_metadata(&value)));
|
|
}
|
|
}
|
|
let count = items.len();
|
|
idx.rebuild_from(items.into_iter())?;
|
|
idx.reload_reader()?;
|
|
Ok(count)
|
|
}
|
|
|
|
/// Unconditionally rebuild the item and creator Tantivy text indexes from the
|
|
/// durable entity stores at open, and fold the suggestion-index rebuild into the
|
|
/// same item scan.
|
|
///
|
|
/// # Why this exists (BLOCKER 7)
|
|
///
|
|
/// The text index is DERIVED state fed by a best-effort async syncer. Items are
|
|
/// persisted durably to the entity store and *then* enqueued to the syncer; a
|
|
/// hard crash within the syncer's batch window (default 2s) leaves those items
|
|
/// in the durable store but never committed to Tantivy. The old design relied on
|
|
/// a Tantivy commit-payload sequence number for recovery, but that machinery was
|
|
/// never wired (every write enqueued `seq: 0`) and never read at open, so those
|
|
/// items vanished from BM25 search permanently — silently, with health GREEN.
|
|
///
|
|
/// The fix mirrors the vector index, which is unconditionally rebuilt from the
|
|
/// durable embeddings at open: here we unconditionally rebuild the text indexes
|
|
/// from the durable entity stores (the source of truth) on every reopen. Any
|
|
/// item/creator present in the store but missing from Tantivy is re-indexed.
|
|
///
|
|
/// # Efficiency
|
|
///
|
|
/// The item side performs a SINGLE scan of the items engine and drives BOTH the
|
|
/// suggestion index and the item text index from it (the suggestion rebuild was
|
|
/// previously its own separate scan). The creator side is a separate scan of the
|
|
/// creators engine because creators live in their own store and are absent from
|
|
/// every item-keyed pass. `EMB:` embedding rows (items) are skipped, matching
|
|
/// the live write path's text exclusions.
|
|
///
|
|
/// Failures are non-fatal and logged: a failed text rebuild leaves search
|
|
/// degraded for that index, but the durable store is intact and the
|
|
/// health-driven resync (`poll_text_index_health`) remains a backstop.
|
|
pub(super) fn rebuild_text_indexes_at_open(
|
|
storage: &StorageBox,
|
|
item_text_index: Option<&crate::text::TextIndex>,
|
|
creator_text_index: Option<&crate::text::TextIndex>,
|
|
suggestion_index: &SuggestionIndex,
|
|
) {
|
|
rebuild_item_text_and_suggestions_at_open(storage, item_text_index, suggestion_index);
|
|
rebuild_creator_text_at_open(storage, creator_text_index);
|
|
}
|
|
|
|
/// Single item scan that rebuilds the suggestion index and (if present) the item
|
|
/// text index. See [`rebuild_text_indexes_at_open`] for rationale.
|
|
fn rebuild_item_text_and_suggestions_at_open(
|
|
storage: &StorageBox,
|
|
item_text_index: Option<&crate::text::TextIndex>,
|
|
suggestion_index: &SuggestionIndex,
|
|
) {
|
|
// One pass over every persisted item, feeding both derived indexes.
|
|
let mut text_items: Vec<(EntityId, HashMap<String, String>)> = Vec::new();
|
|
let mut suggestions = 0u64;
|
|
for entry in storage.items_engine().scan_prefix(&[]) {
|
|
let Ok((key, value)) = entry else { continue };
|
|
let Some(entity_id) = item_meta_entity_id(&key) else {
|
|
continue;
|
|
};
|
|
let meta = deserialize_metadata(&value);
|
|
if let Some(title) = meta.get("title") {
|
|
suggestion_index.index_title(title);
|
|
suggestions += 1;
|
|
}
|
|
if item_text_index.is_some() {
|
|
text_items.push((entity_id, meta));
|
|
}
|
|
}
|
|
|
|
if suggestions > 0 {
|
|
tracing::info!(
|
|
items = suggestions,
|
|
"suggestion index rebuilt from durable storage"
|
|
);
|
|
}
|
|
|
|
if let Some(idx) = item_text_index {
|
|
let count = text_items.len();
|
|
// The health flag is freshly `true` on a just-opened index, so a
|
|
// successful rebuild needs no explicit clear — search is in sync.
|
|
match idx
|
|
.rebuild_from(text_items.into_iter())
|
|
.and_then(|()| idx.reload_reader())
|
|
{
|
|
Ok(()) => {
|
|
if count > 0 {
|
|
tracing::info!(
|
|
items = count,
|
|
"item text index rebuilt from durable storage at open"
|
|
);
|
|
}
|
|
}
|
|
Err(e) => tracing::error!(
|
|
error = %e,
|
|
"item text index rebuild at open failed; BM25 search degraded \
|
|
until the next health-driven resync"
|
|
),
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Scan the creators engine and rebuild the creator text index from it.
|
|
///
|
|
/// Creators are stored via `serialize_entity(None, metadata)` (like users), so
|
|
/// each value is deserialized with [`deserialize_entity`](crate::entities::deserialize_entity)
|
|
/// rather than the raw-metadata codec used for items.
|
|
fn rebuild_creator_text_at_open(
|
|
storage: &StorageBox,
|
|
creator_text_index: Option<&crate::text::TextIndex>,
|
|
) {
|
|
use crate::storage::keys::parse_key;
|
|
|
|
let Some(idx) = creator_text_index else {
|
|
return; // no creator text fields declared
|
|
};
|
|
|
|
let mut creators: Vec<(EntityId, HashMap<String, String>)> = Vec::new();
|
|
for entry in storage.creators_engine().scan_prefix(&[]) {
|
|
let Ok((key, value)) = entry else { continue };
|
|
if let Some((entity_id, Tag::Meta, _suffix)) = parse_key(&key) {
|
|
let (_emb, meta) = crate::entities::deserialize_entity(&value);
|
|
creators.push((entity_id, meta));
|
|
}
|
|
}
|
|
|
|
let count = creators.len();
|
|
match idx
|
|
.rebuild_from(creators.into_iter())
|
|
.and_then(|()| idx.reload_reader())
|
|
{
|
|
Ok(()) => {
|
|
if count > 0 {
|
|
tracing::info!(
|
|
creators = count,
|
|
"creator text index rebuilt from durable storage at open"
|
|
);
|
|
}
|
|
}
|
|
Err(e) => tracing::error!(
|
|
error = %e,
|
|
"creator text index rebuild at open failed; creator BM25 search degraded"
|
|
),
|
|
}
|
|
}
|
|
|
|
/// Refresh index health metrics from the live index handles.
|
|
///
|
|
/// Called once per checkpoint cycle (~30s). Reads current stats from the
|
|
/// Tantivy text index, `USearch` embedding registry, and bitmap indexes, then
|
|
/// stores them into the corresponding `MetricsState` atomic gauges.
|
|
///
|
|
/// All stores use `Relaxed` ordering because these are monitoring gauges --
|
|
/// a slightly stale value is acceptable, and no other thread depends on the
|
|
/// freshness of any individual gauge.
|
|
#[cfg(feature = "metrics")]
|
|
fn refresh_index_metrics(handles: &IndexMetricsHandles, metrics: &MetricsState) {
|
|
// Tantivy text index.
|
|
if let Some(ref text) = handles.text_index {
|
|
let (segments, docs) = text.index_stats();
|
|
metrics
|
|
.tantivy_segment_count
|
|
.store(segments as u64, Ordering::Relaxed);
|
|
metrics.tantivy_indexed_docs.store(docs, Ordering::Relaxed);
|
|
}
|
|
|
|
// USearch embedding registry.
|
|
if let Ok(registry) = handles.embedding_registry.read() {
|
|
let (vectors, bytes) = registry.index_stats();
|
|
metrics
|
|
.usearch_vector_count
|
|
.store(vectors, Ordering::Relaxed);
|
|
metrics
|
|
.usearch_index_size_bytes
|
|
.store(bytes, Ordering::Relaxed);
|
|
}
|
|
|
|
// Bitmap indexes: sum cardinality across all four index types.
|
|
let cardinality = handles.bitmap_category.total_cardinality()
|
|
+ handles.bitmap_format.total_cardinality()
|
|
+ handles.bitmap_creator.total_cardinality()
|
|
+ handles.bitmap_tag.total_cardinality();
|
|
metrics
|
|
.bitmap_index_cardinality
|
|
.store(cardinality, Ordering::Relaxed);
|
|
}
|
|
|
|
/// Sum the file sizes of WAL segments NOT yet compacted by the checkpoint at
|
|
/// `checkpoint_seq` — i.e. the true "lag behind the checkpoint."
|
|
///
|
|
/// # Why not "all segments"
|
|
///
|
|
/// This previously summed EVERY segment, including ones already covered by the
|
|
/// checkpoint (deletable on the next compaction) and the active segment, so the
|
|
/// number it stored into `metrics.wal_lag_bytes` did not match that gauge's
|
|
/// documentation ("Bytes of WAL segments not yet compacted"). A segment is
|
|
/// "not yet compacted" exactly when [`compact_wal_online`](crate::wal::compaction::compact_wal_online)
|
|
/// would retain it: that compactor deletes segments whose `first_seq < floor`,
|
|
/// where `floor = min(checkpoint_seq, active_first_seq)` (the active segment is
|
|
/// always protected because it still holds uncheckpointed tail events). We
|
|
/// mirror that floor here and sum only the segments with `first_seq >= floor`,
|
|
/// which makes the gauge faithful to its own HELP string.
|
|
///
|
|
/// Run immediately AFTER `compact_wal_online`, the covered segments are already
|
|
/// gone, so in steady state this equals the on-disk retained bytes — but the
|
|
/// floor computation keeps the gauge honest even if compaction was skipped (e.g.
|
|
/// `seq == 0`) or fell behind.
|
|
///
|
|
/// Returns 0 if the directory cannot be read or contains no segments.
|
|
/// Errors on individual file metadata reads are treated as 0 bytes
|
|
/// (non-fatal: this is a best-effort monitoring metric).
|
|
#[cfg(feature = "metrics")]
|
|
fn compute_wal_lag_bytes(wal_dir: &std::path::Path, checkpoint_seq: u64) -> u64 {
|
|
let Ok(segments) = crate::wal::segment::list_segments(wal_dir) else {
|
|
return 0;
|
|
};
|
|
|
|
// The active segment (max `first_seq`) is never compacted away while the
|
|
// writer is live; clamp the floor so it is always counted as lag. This is
|
|
// the identical floor `compact_wal_online` uses, so "lag" == "what survived
|
|
// compaction" by construction.
|
|
let floor = match segments.iter().map(|(first_seq, _)| *first_seq).max() {
|
|
Some(active_first_seq) => checkpoint_seq.min(active_first_seq),
|
|
None => checkpoint_seq,
|
|
};
|
|
|
|
segments
|
|
.iter()
|
|
.filter(|(first_seq, _)| *first_seq >= floor)
|
|
.map(|(_, path)| std::fs::metadata(path).map(|m| m.len()).unwrap_or(0))
|
|
.sum()
|
|
}
|
|
|
|
#[cfg(all(test, feature = "metrics"))]
|
|
#[allow(clippy::unwrap_used)]
|
|
mod wal_lag_tests {
|
|
use super::compute_wal_lag_bytes;
|
|
use crate::{replication::ShardId, wal::segment::segment_filename};
|
|
|
|
/// Write a segment file of `len` bytes whose name encodes `first_seq`, and
|
|
/// return its byte length so callers can build expected sums.
|
|
fn write_segment(dir: &std::path::Path, first_seq: u64, len: usize) -> u64 {
|
|
let path = dir.join(segment_filename(ShardId::SINGLE, first_seq));
|
|
std::fs::write(&path, vec![0u8; len]).unwrap();
|
|
len as u64
|
|
}
|
|
|
|
/// The lag gauge must EXCLUDE segments fully covered by the checkpoint and
|
|
/// INCLUDE both the straddling/active segment and any post-checkpoint
|
|
/// segment — matching the gauge's "not yet compacted" documentation. The old
|
|
/// sum-everything implementation over-counted the covered segments.
|
|
#[test]
|
|
fn lag_excludes_covered_segments_and_counts_the_active_tail() {
|
|
let dir = tempfile::tempdir().unwrap();
|
|
|
|
// Three segments. Checkpoint at seq=200 means:
|
|
// seg first_seq=1 -> covered (1 < floor) -> EXCLUDED
|
|
// seg first_seq=200 -> active/straddling tail -> INCLUDED
|
|
// (200 is the max first_seq, so floor = min(200,200) = 200)
|
|
let _covered = write_segment(dir.path(), 1, 4096);
|
|
let active = write_segment(dir.path(), 200, 1024);
|
|
|
|
let lag = compute_wal_lag_bytes(dir.path(), 200);
|
|
assert_eq!(
|
|
lag, active,
|
|
"lag must count only the uncompacted active segment, not the covered one"
|
|
);
|
|
}
|
|
|
|
/// A second post-checkpoint segment is also lag; only the truly-covered
|
|
/// segment is dropped from the sum.
|
|
#[test]
|
|
fn lag_counts_all_uncompacted_segments() {
|
|
let dir = tempfile::tempdir().unwrap();
|
|
|
|
// floor = min(checkpoint_seq=50, active_first_seq=300) = 50.
|
|
// first_seq=1 -> 1 < 50 -> EXCLUDED (covered)
|
|
// first_seq=100 -> >= 50 -> INCLUDED
|
|
// first_seq=300 -> active -> INCLUDED
|
|
let _covered = write_segment(dir.path(), 1, 8192);
|
|
let mid = write_segment(dir.path(), 100, 2048);
|
|
let active = write_segment(dir.path(), 300, 512);
|
|
|
|
let lag = compute_wal_lag_bytes(dir.path(), 50);
|
|
assert_eq!(lag, mid + active, "every uncompacted segment counts as lag");
|
|
}
|
|
|
|
/// With no checkpoint yet (`checkpoint_seq == 0`) nothing is covered, so the
|
|
/// whole on-disk WAL is lag — there is nothing the checkpoint has caught up to.
|
|
#[test]
|
|
fn lag_with_zero_checkpoint_counts_everything() {
|
|
let dir = tempfile::tempdir().unwrap();
|
|
let a = write_segment(dir.path(), 1, 1000);
|
|
let b = write_segment(dir.path(), 50, 2000);
|
|
let lag = compute_wal_lag_bytes(dir.path(), 0);
|
|
assert_eq!(lag, a + b, "no checkpoint => all WAL bytes are lag");
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
#[allow(clippy::unwrap_used)]
|
|
mod rebuild_tests {
|
|
use std::collections::HashMap;
|
|
|
|
use roaring::RoaringBitmap;
|
|
|
|
use super::{ItemIndexes, rebuild_entity_state, rebuild_item_indexes};
|
|
use crate::{
|
|
db::{metadata::serialize_metadata, storage_box::StorageBox},
|
|
entities::{
|
|
CreatorItemsBitmap, InteractionLedger, UserStateIndex,
|
|
relationship::{
|
|
RelationshipType, encode_relationship_key, serialize_relationship_value,
|
|
},
|
|
},
|
|
schema::{EntityId, Timestamp},
|
|
storage::{
|
|
InMemoryBackend, Tag, encode_key,
|
|
indexes::{bitmap::BitmapIndex, range::RangeIndex},
|
|
},
|
|
};
|
|
|
|
fn empty_memory_storage() -> StorageBox {
|
|
StorageBox::Memory {
|
|
items: InMemoryBackend::new(),
|
|
users: InMemoryBackend::new(),
|
|
creators: InMemoryBackend::new(),
|
|
}
|
|
}
|
|
|
|
/// Write one item-metadata row exactly as the live write path does
|
|
/// (`encode_key(id, Tag::Meta, b"") -> serialize_metadata(map)`).
|
|
fn put_item_meta(storage: &StorageBox, id: EntityId, meta: &HashMap<String, String>) {
|
|
let key = encode_key(id, Tag::Meta, b"");
|
|
storage
|
|
.items_engine()
|
|
.put(&key, &serialize_metadata(meta))
|
|
.unwrap();
|
|
}
|
|
|
|
fn meta(pairs: &[(&str, &str)]) -> HashMap<String, String> {
|
|
pairs
|
|
.iter()
|
|
.map(|(k, v)| ((*k).to_owned(), (*v).to_owned()))
|
|
.collect()
|
|
}
|
|
|
|
/// Direct contract test for `rebuild_entity_state`: durable relationship
|
|
/// edges (block / hide / follow / interaction) must reconstruct the identical
|
|
/// in-memory `UserStateIndex` sets and interaction ledger after a restart.
|
|
#[test]
|
|
fn rebuild_entity_state_reconstructs_relationship_state() {
|
|
let storage = empty_memory_storage();
|
|
let ts = Timestamp::now();
|
|
let (user, creator, item) = (EntityId::new(1), EntityId::new(100), EntityId::new(7));
|
|
|
|
// Persist edges into the users keyspace exactly as the write path does.
|
|
let edges = [
|
|
(user, RelationshipType::Blocks, creator),
|
|
(user, RelationshipType::Hide, item),
|
|
(user, RelationshipType::Follows, creator),
|
|
(user, RelationshipType::InteractionWeight, creator),
|
|
];
|
|
for (from, rel, to) in edges {
|
|
let key = encode_relationship_key(from, rel, to);
|
|
let value = serialize_relationship_value(2.5, ts);
|
|
storage.users_engine().put(&key, &value).unwrap();
|
|
}
|
|
|
|
let user_state = UserStateIndex::new();
|
|
let interaction = InteractionLedger::new();
|
|
rebuild_entity_state(&storage, &user_state, &interaction).unwrap();
|
|
|
|
// Blocks -> blocked_creators.
|
|
assert!(
|
|
user_state.blocked_creators(1).contains(&100),
|
|
"block edge must rebuild into blocked_creators"
|
|
);
|
|
// Hide -> hidden_items.
|
|
assert!(
|
|
user_state.hidden_items(1).contains(7),
|
|
"hide edge must rebuild into hidden_items"
|
|
);
|
|
// Follows -> followed_creators (and reverse follower set).
|
|
assert!(
|
|
user_state.followed_creators(1).contains(&100),
|
|
"follow edge must rebuild into followed_creators"
|
|
);
|
|
// InteractionWeight -> interaction ledger carries the persisted weight.
|
|
assert!(
|
|
interaction.score(1, 100, ts.as_nanos()) > 0.0,
|
|
"interaction edge must rebuild a non-zero decayed weight"
|
|
);
|
|
}
|
|
|
|
/// Direct contract test for `rebuild_item_indexes`: a persisted item's
|
|
/// metadata must reconstruct the identical universe bitmap, every metadata
|
|
/// index, the `creator_items` bitmap, and the `created_at` recency entry that the
|
|
/// shared `ItemIndexes::index` write path would produce.
|
|
#[test]
|
|
fn rebuild_item_indexes_reconstructs_every_index() {
|
|
let storage = empty_memory_storage();
|
|
|
|
// One fully-tagged item, written via the canonical metadata codec.
|
|
put_item_meta(
|
|
&storage,
|
|
EntityId::new(42),
|
|
&meta(&[
|
|
("category", "jazz"),
|
|
("format", "audio"),
|
|
("creator_id", "100"),
|
|
("tags", "smooth,live"),
|
|
("duration", "180"),
|
|
("created_at", "1000"),
|
|
]),
|
|
);
|
|
|
|
let mut universe = RoaringBitmap::new();
|
|
let category = BitmapIndex::new("category");
|
|
let format = BitmapIndex::new("format");
|
|
let creator = BitmapIndex::new("creator_id");
|
|
let tag = BitmapIndex::new("tags");
|
|
let duration = RangeIndex::<u32>::new("duration");
|
|
let created_at = RangeIndex::<u64>::new("created_at");
|
|
let creator_items = CreatorItemsBitmap::new();
|
|
let indexes = ItemIndexes {
|
|
category: &category,
|
|
format: &format,
|
|
creator: &creator,
|
|
tag: &tag,
|
|
duration: &duration,
|
|
created_at: &created_at,
|
|
creator_items: &creator_items,
|
|
};
|
|
|
|
rebuild_item_indexes(&storage, &mut universe, &indexes).unwrap();
|
|
|
|
assert!(universe.contains(42), "item must enter the universe bitmap");
|
|
assert!(
|
|
category.get("jazz").is_some_and(|bm| bm.contains(42)),
|
|
"category index must be rebuilt"
|
|
);
|
|
assert!(
|
|
format.get("audio").is_some_and(|bm| bm.contains(42)),
|
|
"format index must be rebuilt"
|
|
);
|
|
assert!(
|
|
creator.get("100").is_some_and(|bm| bm.contains(42)),
|
|
"creator index must be rebuilt"
|
|
);
|
|
assert!(
|
|
tag.get("smooth").is_some_and(|bm| bm.contains(42))
|
|
&& tag.get("live").is_some_and(|bm| bm.contains(42)),
|
|
"each comma-separated tag must be rebuilt"
|
|
);
|
|
assert!(
|
|
creator_items.get(100).is_some_and(|bm| bm.contains(42)),
|
|
"creator_items bitmap must be rebuilt for the `following` profile"
|
|
);
|
|
// Recency: the created_at value must land in the range index.
|
|
let recent = std::ops::Bound::Included(&1000u64);
|
|
let recent = created_at.range(recent, std::ops::Bound::Included(&1000u64));
|
|
assert!(
|
|
recent.contains(42),
|
|
"created_at recency entry must be rebuilt at the persisted value"
|
|
);
|
|
}
|
|
|
|
/// Corrupt/hostile input regression (WARNING fix): the write path rejects any
|
|
/// item id > `u32::MAX` before it ever persists, so such a row can only reach
|
|
/// the rebuild from a corrupt store. The rebuild must SKIP it — never truncate
|
|
/// it into a colliding u32 slot that aliases a real lower id and manufactures
|
|
/// phantom query hits.
|
|
#[test]
|
|
fn rebuild_item_indexes_skips_over_u32_ids_instead_of_colliding() {
|
|
let storage = empty_memory_storage();
|
|
|
|
// A legitimate low item id, and a hostile id that truncates onto it
|
|
// (`(1<<32) + 5` -> 5). If the rebuild truncated, both would land on
|
|
// slot 5 with last-write-wins metadata corruption.
|
|
let low = EntityId::new(5);
|
|
let over = EntityId::new((1u64 << 32) + 5);
|
|
put_item_meta(&storage, low, &meta(&[("category", "real")]));
|
|
put_item_meta(&storage, over, &meta(&[("category", "phantom")]));
|
|
|
|
let mut universe = RoaringBitmap::new();
|
|
let category = BitmapIndex::new("category");
|
|
let format = BitmapIndex::new("format");
|
|
let creator = BitmapIndex::new("creator_id");
|
|
let tag = BitmapIndex::new("tags");
|
|
let duration = RangeIndex::<u32>::new("duration");
|
|
let created_at = RangeIndex::<u64>::new("created_at");
|
|
let creator_items = CreatorItemsBitmap::new();
|
|
let indexes = ItemIndexes {
|
|
category: &category,
|
|
format: &format,
|
|
creator: &creator,
|
|
tag: &tag,
|
|
duration: &duration,
|
|
created_at: &created_at,
|
|
creator_items: &creator_items,
|
|
};
|
|
|
|
rebuild_item_indexes(&storage, &mut universe, &indexes).unwrap();
|
|
|
|
// The legitimate row is indexed; the over-range row is skipped, so slot 5
|
|
// is NOT cross-contaminated with the "phantom" category.
|
|
assert!(universe.contains(5), "the valid low id must be indexed");
|
|
assert!(
|
|
category.get("real").is_some_and(|bm| bm.contains(5)),
|
|
"the valid low id keeps its true category"
|
|
);
|
|
assert!(
|
|
category.get("phantom").is_none(),
|
|
"the over-u32::MAX row must be skipped, not truncated into slot 5"
|
|
);
|
|
}
|
|
}
|