tidaldb/docs/ops/prometheus-alerts.yaml
jx12n ad4134e280 chore: doc consolidation, seven-dimension review fixes, and commit hooks
- Eliminate the tidal/ self-contained doc mirror; docs now have two canonical
  homes (root *.md and docs/), with planning/specs/research/reviews moved up
- Remove stale .agents/skills and .ai mirrors; canonicalize skills under .claude/
- Add pre-commit hook + scripts/check-docs.sh doc-guard + scripts/install-hooks.sh
- Implement M0-M10 seven-dimension review findings across engine, net, server,
  and tidalctl (durability, replication, query, WAL, storage, CLI hardening)
2026-06-08 22:46:28 -06:00

132 lines
6.1 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# =============================================================================
# DESIGN-REFERENCE RULE SET — NOT LOADED BY ANY ALERTMANAGER TODAY.
# =============================================================================
# This file is reference/design config. It is NOT wired into any running
# alerting pipeline:
# - vmalert (pilot/dev) loads ONLY ops/vmalert/rules/*.yaml
# - prod CRDs live ONLY in infra/k8s/prom/prometheus-rules/*.yaml
# Nothing mounts or imports docs/ops/prometheus-alerts.yaml.
# Do not read these as live pager rules.
#
# Provenance: every metric referenced below (tidaldb_health_ok,
# tidaldb_checkpoint_age_seconds, tidaldb_checkpoint_failures_total,
# tidaldb_wal_lag_bytes, tidaldb_signal_hot_entries, tidaldb_degradation_level,
# tidaldb_active_sessions, tidaldb_rate_limited_total,
# tidaldb_tantivy_segment_count, tidaldb_retrieve_latency_us_bucket,
# tidaldb_search_latency_us_bucket) IS really emitted today by
# tidal/src/db/metrics/mod.rs. So these rules are promotable — they
# are not orphaned against phantom metrics.
#
# Promotion path (the right long-term fix — do it deliberately, do NOT
# hand-copy these as live pager rules without the steps below):
# 1. Add a vmalert rule file: ops/vmalert/rules/tidaldb.yaml, translating
# each rule and stamping the canonical labels every routed rule carries —
# severity + capability + surface (+ optional component). Map per the
# canonical severity taxonomy in ops/alerting.md:
# labels {severity: critical} = pager class (TidalDBDown only here),
# {severity: warning} = non-paging / business-hours,
# {severity: info} = pure-FYI.
# Add capability: tidaldb and a surface label per rule, and a
# runbook_url annotation:
# https://github.com/orchard9/tidaldb/blob/main/docs/runbooks/<slug>.md
# 2. Add the prod CRD twin: infra/k8s/prom/prometheus-rules/tidaldb.yaml,
# kept byte-aligned in expr/threshold with the vmalert file.
# 3. Verify the alerts fire against real metrics (scrape tidaldb, force a
# degraded state) before declaring them on-call-ready.
# Until steps 13 land, this file stays a design reference only.
# =============================================================================
groups:
- name: tidaldb
interval: 30s
rules:
- alert: TidalDBDown
expr: tidaldb_health_ok == 0
for: 1m
labels: { severity: critical }
annotations:
summary: "tidalDB is unhealthy"
description: "tidaldb_health_ok is 0 — database is unhealthy or shut down."
- alert: TidalDBCheckpointStale
expr: tidaldb_checkpoint_age_seconds > 300
for: 2m
labels: { severity: warning }
annotations:
summary: "Signal checkpoint not running"
description: "{{ $value }}s since last checkpoint (threshold: 300s). Signal durability at risk."
- alert: TidalDBCheckpointFailures
expr: increase(tidaldb_checkpoint_failures_total[5m]) > 0
labels: { severity: warning }
annotations:
summary: "Signal checkpoint failures detected"
description: "Checkpoint failures in last 5m. Check disk space and storage errors."
- alert: TidalDBWALDiskPressure
expr: tidaldb_wal_lag_bytes > 1000000000
for: 5m
labels: { severity: warning }
annotations:
summary: "WAL disk usage exceeds 1GB"
description: "{{ $value | humanize1024 }}B of WAL uncompacted. Compaction may be stuck."
- alert: TidalDBSignalBacklog
expr: tidaldb_signal_hot_entries > 4000000
for: 5m
labels: { severity: warning }
annotations:
summary: "Signal ledger over 80% of capacity"
description: "{{ $value }} hot entries (threshold: 4M / 80% of 5M budget)."
- alert: TidalDBDegradedRanking
expr: tidaldb_degradation_level > 0
for: 2m
labels: { severity: warning }
annotations:
summary: "Ranking quality degraded"
description: "Degradation level {{ $value }} active. Scale up or reduce load."
- alert: TidalDBSessionLeak
# tidaldb_active_sessions is a GAUGE, so rate()/increase() are invalid
# (those operators only count counter resets). deriv() fits a least-squares
# slope over the window and is the gauge-correct way to detect sustained
# growth: > 0.03 sessions/sec is ~> 9 new sessions over a 5m window.
expr: deriv(tidaldb_active_sessions[5m]) > 0.03 and tidaldb_active_sessions > 100
for: 5m
labels: { severity: warning }
annotations:
summary: "Active session count growing steadily"
description: "{{ $value }} active sessions and trending up (deriv > 0.03/s). Agents may not be closing sessions."
- alert: TidalDBHighRateLimiting
expr: rate(tidaldb_rate_limited_total[5m]) > 100
for: 5m
labels: { severity: info }
annotations:
summary: "Sustained rate limiting"
description: "{{ $value }}/s rate-limited writes. Review agent rate limit config."
- alert: TidalDBTantivySegmentBloat
expr: tidaldb_tantivy_segment_count > 30
for: 10m
labels: { severity: warning }
annotations:
summary: "Tantivy segment count elevated"
description: "{{ $value }} segments (threshold: 30). Text syncer may be stalled."
- alert: TidalDBSlowRetrieve
expr: histogram_quantile(0.95, rate(tidaldb_retrieve_latency_us_bucket[5m])) > 500000
for: 5m
labels: { severity: warning }
annotations:
summary: "Retrieve p95 latency exceeds 500ms"
description: "p95 retrieve latency is {{ $value | humanizeDuration }}. Check signal ledger load and degradation level."
- alert: TidalDBSlowSearch
expr: histogram_quantile(0.95, rate(tidaldb_search_latency_us_bucket[5m])) > 1000000
for: 5m
labels: { severity: warning }
annotations:
summary: "Search p95 latency exceeds 1s"
description: "p95 search latency is {{ $value | humanizeDuration }}. Check Tantivy segment count and ANN index health."