Two outages in one session came from this file claiming `replicas: 0` while the cluster served traffic: a plain `kubectl apply` scaled it to zero, twice. Source now states production intent. Parking stays an explicit `kubectl scale` divergence recorded in k3s-fleet/cluster-state.yaml, and scripts/restore-fleet.sh remains the guarded path back - its storage, image, and per-node capacity preflights are exactly what a bare apply does not do. Pins server:m12-consumer-readpath-20260818@sha256:9191233d..., the build with the transport-aware read budget and the scatter-degraded counters. Verified after the rolling update: 3/3 Ready, all regions lag 0, every replica answering 3/3 shards with degraded=false on the default budget.
345 lines
17 KiB
YAML
345 lines
17 KiB
YAML
# The tidalDB CLUSTER: ONE StatefulSet, every pod a region (m11p5 §4).
|
||
#
|
||
# MUTUALLY EXCLUSIVE with the standalone set in k8s/. Both source workloads are
|
||
# parked at 0. `scripts/restore-fleet.sh` restores one selected data plane:
|
||
# cardinality 1 in namespace `tidaldb`, or three `cluster --region` processes in
|
||
# namespace `tidaldb-cluster` with real quorum-ack writes. Never run both.
|
||
#
|
||
# WHY ONE StatefulSet (not one-per-region): the m11p5 bind/advertise split lets
|
||
# every pod mount the SAME topology ConfigMap (peers are advertised by per-pod
|
||
# DNS; the local socket binds 0.0.0.0), so a single StatefulSet with stable pod
|
||
# identities tidaldb-{0,1,2} IS the three regions. Scaling is `kubectl scale`
|
||
# (see docs/runbooks/kubernetes.md): pod N>=3 auto-seed-joins as a learner and
|
||
# auto-promotes to a voter — no file edits, no per-pod manifests.
|
||
apiVersion: apps/v1
|
||
kind: StatefulSet
|
||
metadata:
|
||
name: tidaldb
|
||
namespace: tidaldb-cluster
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
spec:
|
||
serviceName: tidaldb-peers # the headless peer Service — stable per-pod DNS
|
||
# PRODUCTION as of 2026-08-18: three voters is the desired state, not a value
|
||
# only the restore script may set. `replicas: 0` lived here while the cluster
|
||
# was parked and it cost two outages in one session: a plain `kubectl apply` of
|
||
# this file scaled a LIVE cluster to zero, twice, because the manifest still
|
||
# claimed parked. Source states intent; parking is an explicit `kubectl scale`
|
||
# divergence recorded in k3s-fleet/cluster-state.yaml, and
|
||
# scripts/restore-fleet.sh remains the guarded path back (its storage, image,
|
||
# and per-node capacity preflights are what a bare apply does not do).
|
||
replicas: 3
|
||
# Parallel: bring all pods up at once. There is no ordered-bootstrap
|
||
# dependency — siblings boot in any order (an unreachable-at-startup peer is
|
||
# normal; the election + catch-up timer converge them). Ordered start would
|
||
# only serialize a 3-region cold boot for no benefit.
|
||
podManagementPolicy: Parallel
|
||
selector:
|
||
matchLabels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
template:
|
||
metadata:
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
annotations:
|
||
# Plain-Prometheus scrape hints (per-pod :9091, unauthenticated — keep
|
||
# cluster-internal). The Operator-native path is a PodMonitor/ServiceMonitor.
|
||
prometheus.io/scrape: "true"
|
||
prometheus.io/port: "9091"
|
||
prometheus.io/path: "/metrics"
|
||
spec:
|
||
# SIGTERM flips readiness to 503 (pod leaves the client Service), drains
|
||
# in-flight requests, then checkpoints + fsyncs the WAL AND saves every
|
||
# shard's HNSW graph before exit (m12p6). The graph save is the long pole at
|
||
# the production shape (~32k vectors/slot × 3 shards, serialized + fsynced),
|
||
# so the grace must cover it or k8s SIGKILLs mid-save and the next boot
|
||
# rebuilds. 600s is a generous ceiling; the bounded drain (below) starts the
|
||
# save early, and a clean save typically finishes in well under a minute.
|
||
terminationGracePeriodSeconds: 600
|
||
# Spread the three pods across distinct nodes so a single node loss takes
|
||
# at most one voter — preserving quorum (2 of 3). ScheduleAnyway (not
|
||
# DoNotSchedule) so a smaller cluster still schedules, just less spread.
|
||
topologySpreadConstraints:
|
||
- maxSkew: 1
|
||
topologyKey: kubernetes.io/hostname
|
||
whenUnsatisfiable: ScheduleAnyway
|
||
labelSelector:
|
||
matchLabels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
securityContext:
|
||
runAsNonRoot: true
|
||
runAsUser: 10001 # the fixed `tidal` uid (docker/deploy/Dockerfile)
|
||
runAsGroup: 10001
|
||
fsGroup: 10001 # makes the mounted PVC group-writable by the runtime user
|
||
seccompProfile:
|
||
type: RuntimeDefault
|
||
initContainers:
|
||
- name: init-datadir
|
||
image: busybox@sha256:73aaf090f3d85aa34ee199857f03fa3a95c8ede2ffd4cc2cdb5b94e566b11662
|
||
imagePullPolicy: IfNotPresent
|
||
command: ["sh", "-c", "mkdir -p /data/db && chown -R 10001:10001 /data/db"]
|
||
securityContext:
|
||
allowPrivilegeEscalation: false
|
||
runAsUser: 10001
|
||
runAsGroup: 10001
|
||
volumeMounts:
|
||
- name: data
|
||
mountPath: /data
|
||
containers:
|
||
- name: tidaldb
|
||
image: registry.threesix.ai/tidal/server:m12-consumer-readpath-20260818@sha256:9191233d012e41fd92bb6d0efea83b60010995a6fc6e3ebc894f3d7c5f72872a
|
||
imagePullPolicy: IfNotPresent
|
||
# The image ENTRYPOINT is the bare binary. We override the command with
|
||
# a tiny /bin/sh wrapper (the bookworm-slim runtime HAS a shell) so we
|
||
# can branch on the pod ordinal: pods 0-2 are the initial voter set
|
||
# (plain topology boot); pods >=3 are SCALE-UP and must seed-join as
|
||
# learners. Keeping this in args (no initContainer, no extra image)
|
||
# means the whole scale story is readable in this one file.
|
||
#
|
||
# POD_NAME is e.g. "tidaldb-4"; ORD is its trailing ordinal. For ORD<3
|
||
# we boot from the topology file (the region IS this pod's name). For
|
||
# ORD>=3 we ALSO pass --seed (any peer; the headless Service load-
|
||
# balances to a live one) + this pod's advertised DNS addresses, and
|
||
# the node learns its roster/id/term from a seed and joins as a learner.
|
||
# The topology ConfigMap is STILL mounted+passed for the behavioral knob
|
||
# blocks (replication/election/...), required even for a --seed boot
|
||
# (m11p5 §3.5); its regions: list is ignored for a seed joiner's roster.
|
||
command: ["/bin/sh", "-c"]
|
||
args:
|
||
- |
|
||
set -eu
|
||
ORD="${POD_NAME##*-}"
|
||
# Per-pod STABLE DNS (headless Service) — what this pod ADVERTISES
|
||
# for peers to dial it directly. The headless Service publishes
|
||
# not-ready addresses (so a pod has DNS before it is Ready), so it
|
||
# resolves to EVERY pod incl. still-joining ones.
|
||
DOMAIN="tidaldb-peers.tidaldb-cluster.svc.cluster.local"
|
||
# READY-ONLY client Service (ClusterIP VIP) — the seed-join discovery
|
||
# target. It excludes not-ready pods, so a joiner always reaches a
|
||
# LIVE serving peer instead of round-robining onto a not-ready pod
|
||
# (often ITSELF, since the headless Service includes the joiner) and
|
||
# failing discovery for the whole 120s window — the real T4 scale-up
|
||
# blocker. Carries its own cert SAN (certs.yaml).
|
||
SEED_SVC="tidaldb.tidaldb-cluster.svc.cluster.local"
|
||
# Common args for every pod.
|
||
set -- cluster \
|
||
--listen 0.0.0.0:9500 \
|
||
--data-dir /data/db \
|
||
--schema /etc/tidal-server/schema/schema.yaml \
|
||
--topology /etc/tidal-server/cluster-topology.yaml \
|
||
--experimental-cluster
|
||
if [ "$ORD" -ge 3 ]; then
|
||
# SCALE-UP pod: seed-join as a learner. Discover a live leader via
|
||
# the READY-ONLY client Service ($SEED_SVC); advertise THIS pod's
|
||
# stable per-pod DNS ($DOMAIN) for gRPC (9601) and HTTP (9500) so
|
||
# peers dial it directly. --metrics gives the joiner a metrics
|
||
# listener (it has no topology entry).
|
||
# m11p7/m12p5: the :9500 plane serves TLS, and `peer_url` honors an
|
||
# explicit URL scheme VERBATIM (forward.rs) — so the seed MUST be
|
||
# `https://`, not `http://` (with `http://` the joiner dials
|
||
# plaintext to the TLS port and seed-join fails). The discovery
|
||
# target is the ready-only client Service, NOT the headless peers
|
||
# Service, so a joiner never round-robins onto a not-ready pod
|
||
# (incl. itself) and burns the 120s discovery window — both were
|
||
# real T4 scale-up blockers.
|
||
set -- "$@" \
|
||
--seed "https://${SEED_SVC}:9500" \
|
||
--advertise-grpc "${POD_NAME}.${DOMAIN}:9601" \
|
||
--advertise-http "${POD_NAME}.${DOMAIN}:9500" \
|
||
--metrics 0.0.0.0:9091
|
||
fi
|
||
exec tidal-server "$@"
|
||
env:
|
||
# Region identity == pod name (tidaldb-0/1/2/...). For ORD<3 this
|
||
# MUST match a region declared in the topology ConfigMap; the names
|
||
# line up by construction (regions are named after the pod identities).
|
||
- name: POD_NAME
|
||
valueFrom:
|
||
fieldRef:
|
||
fieldPath: metadata.name
|
||
- name: TIDAL_REGION
|
||
valueFrom:
|
||
fieldRef:
|
||
fieldPath: metadata.name
|
||
- name: TIDAL_API_KEY
|
||
valueFrom:
|
||
secretKeyRef:
|
||
name: tidaldb-credentials
|
||
key: TIDAL_API_KEY
|
||
# m11p7: the cluster key (mints/verifies per-node internal tokens that
|
||
# authenticate inter-node HTTP). A file mount (not an inline env) so a
|
||
# rotation of the Secret is picked up WITHOUT a pod restart by the
|
||
# credential poller. Distinct secret data key from the bearer.
|
||
- name: TIDAL_CLUSTER_KEY_FILE
|
||
value: /etc/tidaldb/cluster-key/cluster-key
|
||
- name: TIDAL_SERVER_LOG
|
||
value: info
|
||
- name: TIDAL_ALLOW_EXPERIMENTAL_CLUSTER
|
||
value: "1"
|
||
# m12p6: shorten the post-SIGTERM in-flight drain so the (long) HNSW
|
||
# graph save starts promptly within the grace window instead of after
|
||
# the full 15s default. 3s is ample for loopback/in-cluster drain.
|
||
- name: TIDAL_SHUTDOWN_DRAIN_MS
|
||
value: "3000"
|
||
ports:
|
||
- name: http
|
||
containerPort: 9500
|
||
# One gRPC port per hosted shard group (m11p6/m12p4). With the
|
||
# 3-group `shards:` block enabled in the topology ConfigMap, every
|
||
# pod replicates all three groups and binds a derived port per
|
||
# group: shard 0 → 9601, shard 1 → 9602, shard 2 → 9603
|
||
# (`node base port + shard id`; see topology-configmap.yaml). The
|
||
# headless Service reaches each by pod DNS, so these are declared
|
||
# for clarity/NetworkPolicy; the bind itself is driven by the
|
||
# topology. Collapse back to a single `grpc` port if `shards:` is
|
||
# removed (legacy single group).
|
||
- name: grpc
|
||
containerPort: 9601
|
||
- name: grpc-1
|
||
containerPort: 9602
|
||
- name: grpc-2
|
||
containerPort: 9603
|
||
- name: metrics
|
||
containerPort: 9091
|
||
# Three probes map to the three health endpoints. The readinessProbe is
|
||
# now CLUSTER-AWARE (m11p5 §4): /health returns 503 while shutting down,
|
||
# quarantined, removed/decommissioned, or a joiner/install boot has not
|
||
# yet first-converged (lag <= learner_promote_lag, sticky-ready after).
|
||
# A restarted PVC-retained voter is Ready on today's terms (no
|
||
# regression). The full predicate is documented in the kubernetes.md
|
||
# runbook so probe behavior is diagnosable.
|
||
# m11p7: the HTTP plane on :9500 serves TLS (inter-node mTLS), so every
|
||
# probe must use scheme HTTPS. kubelet does NOT verify the server cert
|
||
# for httpGet probes, so the cert's DNS-only SANs (no pod IP) are fine.
|
||
startupProbe:
|
||
httpGet:
|
||
path: /health/startup
|
||
port: http
|
||
scheme: HTTPS
|
||
periodSeconds: 5
|
||
failureThreshold: 240 # ~20 min: HNSW index rebuild/load at 1536-dim is CPU-bound (100k ~5min single-core; headroom for 1M gate)
|
||
livenessProbe:
|
||
httpGet:
|
||
path: /health/live
|
||
port: http
|
||
scheme: HTTPS
|
||
# A transiently saturated query runtime must shed readiness before
|
||
# kubelet turns load into a cascading restart. Six 10s failures give
|
||
# the process roughly one minute to recover while still detecting a
|
||
# genuinely wedged runtime.
|
||
periodSeconds: 10
|
||
timeoutSeconds: 5
|
||
failureThreshold: 6
|
||
readinessProbe:
|
||
httpGet:
|
||
path: /health # cluster-aware: 503 joiner/quarantined/draining
|
||
port: http
|
||
scheme: HTTPS
|
||
periodSeconds: 10
|
||
timeoutSeconds: 3
|
||
failureThreshold: 3
|
||
resources:
|
||
requests:
|
||
# CONSUMER-SIZED, not gate-sized (2026-08-18).
|
||
#
|
||
# The 2-core reservation this replaced came from the Ref-A soak
|
||
# envelope: the busiest leader sustained ~1.7 cores and peaked at
|
||
# ~2.5 under the 200 rps write-heavy mix. That is the ENDURANCE
|
||
# GATE's price, and three voters at 2 cores need 6,000m plus 2,000m
|
||
# free on each of the three nodes their local-path volumes are
|
||
# pinned to. The live fleet is 82-91% committed on requests, so that
|
||
# contract cannot be placed and the cluster stayed parked for it.
|
||
#
|
||
# 300m is what the tightest pinned node (k3s-agent-1, 355m free)
|
||
# can actually reserve for a voter, and it is honest for a FIRST
|
||
# CONSUMER's load - not for the gate. The limit below keeps the
|
||
# measured burst reachable without reserving it.
|
||
#
|
||
# This is provisioned optimism with named detectors: if real write
|
||
# volume approaches the knee, TidalDBClusterQuorumLag,
|
||
# TidalDBClusterCommitIndexStall, TidalDBClusterWritePoolShedding
|
||
# and TidalDBClusterQuorumTimeouts fire before users see it. Raise
|
||
# the request (or add shards - writes hash-route across shard
|
||
# groups) rather than waiting for a stall.
|
||
cpu: "300m"
|
||
# Baseline working set was ~3.6 GiB before load, and full placement
|
||
# means every pod holds the WHOLE 1536-D corpus. This is a resident
|
||
# footprint, not a gate artifact: it stays at 4 GiB.
|
||
memory: 4Gi
|
||
limits:
|
||
# Two cores keeps the measured query/apply burst reachable on the
|
||
# tightest node without reserving it. Note the ratio: a burstable
|
||
# pod whose neighbours are also bursting gets CFS-throttled, which
|
||
# is precisely how CockroachDB was pushed into multi-second Raft
|
||
# stalls on this fleet with nodes 70% idle. The quorum alerts above
|
||
# are the detector for that; more request is the fix.
|
||
cpu: "2"
|
||
# Four independent OOMKills occurred at 3.97-4.00 GiB. Six GiB is
|
||
# measured peak plus 50% recovery/profiling headroom; the exact
|
||
# internal growth source still requires heap/allocation profiling.
|
||
memory: 6Gi
|
||
securityContext:
|
||
allowPrivilegeEscalation: false
|
||
readOnlyRootFilesystem: true # writes only /data (PVC) and /tmp (emptyDir)
|
||
capabilities:
|
||
drop: ["ALL"]
|
||
volumeMounts:
|
||
- name: data
|
||
mountPath: /data
|
||
- name: schema
|
||
mountPath: /etc/tidal-server/schema
|
||
readOnly: true
|
||
- name: topology
|
||
mountPath: /etc/tidal-server/cluster-topology.yaml
|
||
subPath: cluster-topology.yaml
|
||
readOnly: true
|
||
# m11p7 inter-node TLS material (cert-manager Secret). The grpc_tls
|
||
# block in the topology points at these paths. A renewal rewrites the
|
||
# Secret; the kubelet swaps the `..data` symlink and tidalDB's cert
|
||
# poller hot-swaps with zero connection drop.
|
||
- name: cluster-tls
|
||
mountPath: /etc/tidaldb/tls
|
||
readOnly: true
|
||
- name: cluster-key
|
||
mountPath: /etc/tidaldb/cluster-key
|
||
readOnly: true
|
||
- name: tmp
|
||
mountPath: /tmp
|
||
volumes:
|
||
- name: schema
|
||
configMap:
|
||
name: tidaldb-schema
|
||
- name: topology
|
||
configMap:
|
||
name: tidaldb-cluster-topology
|
||
# m11p7: the cert-manager-issued node cert (tls.crt/tls.key/ca.crt).
|
||
- name: cluster-tls
|
||
secret:
|
||
secretName: tidaldb-cluster-tls
|
||
# m11p7: the cluster key for per-node internal tokens (own Secret key).
|
||
- name: cluster-key
|
||
secret:
|
||
secretName: tidaldb-credentials
|
||
items:
|
||
- key: TIDAL_CLUSTER_KEY
|
||
path: cluster-key
|
||
- name: tmp
|
||
emptyDir: {}
|
||
volumeClaimTemplates:
|
||
- metadata:
|
||
name: data
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
backup.orchard9.ai/class: expendable
|
||
backup.orchard9.ai/method: tidal-stress-reseed
|
||
spec:
|
||
accessModes: ["ReadWriteOnce"]
|
||
storageClassName: local-path
|
||
resources:
|
||
requests:
|
||
storage: 5Gi
|