Pin the live cluster to the rc7 amd64 image (@sha256:171505745b801dcf231b531de6167dbc309a7182957811cbc2228f0a302572b1, 6-layer manifest, imagetools-verified) carrying the write-burst false-partition fix (tidal-net record_timeout). Deployed via `kubectl set image`; rolling update completed 3/3 Ready. Reseeds across the rollout were the safe snapshot-install fallback (rejoining behind WAL retention), all converged lag=0, no loop, no loss.
306 lines
16 KiB
YAML
306 lines
16 KiB
YAML
# The tidalDB CLUSTER: ONE StatefulSet, every pod a region (m11p5 §4).
|
||
#
|
||
# MUTUALLY EXCLUSIVE with the standalone set in k8s/ (namespace `tidaldb`,
|
||
# replicas: 1, `standalone` subcommand). This is namespace `tidaldb-cluster`,
|
||
# replicas: 3, the `cluster --region` subcommand, real quorum-ack writes,
|
||
# automatic election, and elastic membership. Deploy ONE or the OTHER per
|
||
# namespace — never both.
|
||
#
|
||
# WHY ONE StatefulSet (not one-per-region): the m11p5 bind/advertise split lets
|
||
# every pod mount the SAME topology ConfigMap (peers are advertised by per-pod
|
||
# DNS; the local socket binds 0.0.0.0), so a single StatefulSet with stable pod
|
||
# identities tidaldb-{0,1,2} IS the three regions. Scaling is `kubectl scale`
|
||
# (see docs/runbooks/kubernetes.md): pod N>=3 auto-seed-joins as a learner and
|
||
# auto-promotes to a voter — no file edits, no per-pod manifests.
|
||
apiVersion: apps/v1
|
||
kind: StatefulSet
|
||
metadata:
|
||
name: tidaldb
|
||
namespace: tidaldb-cluster
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
spec:
|
||
serviceName: tidaldb-peers # the headless peer Service — stable per-pod DNS
|
||
replicas: 3 # the initial voter set; scale up/down per the runbook
|
||
# Parallel: bring all pods up at once. There is no ordered-bootstrap
|
||
# dependency — siblings boot in any order (an unreachable-at-startup peer is
|
||
# normal; the election + catch-up timer converge them). Ordered start would
|
||
# only serialize a 3-region cold boot for no benefit.
|
||
podManagementPolicy: Parallel
|
||
selector:
|
||
matchLabels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
template:
|
||
metadata:
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
annotations:
|
||
# Plain-Prometheus scrape hints (per-pod :9091, unauthenticated — keep
|
||
# cluster-internal). The Operator-native path is a PodMonitor/ServiceMonitor.
|
||
prometheus.io/scrape: "true"
|
||
prometheus.io/port: "9091"
|
||
prometheus.io/path: "/metrics"
|
||
spec:
|
||
# SIGTERM flips readiness to 503 (pod leaves the client Service), drains
|
||
# in-flight requests, then checkpoints + fsyncs the WAL AND saves every
|
||
# shard's HNSW graph before exit (m12p6). The graph save is the long pole at
|
||
# the production shape (~32k vectors/slot × 3 shards, serialized + fsynced),
|
||
# so the grace must cover it or k8s SIGKILLs mid-save and the next boot
|
||
# rebuilds. 600s is a generous ceiling; the bounded drain (below) starts the
|
||
# save early, and a clean save typically finishes in well under a minute.
|
||
terminationGracePeriodSeconds: 600
|
||
# Spread the three pods across distinct nodes so a single node loss takes
|
||
# at most one voter — preserving quorum (2 of 3). ScheduleAnyway (not
|
||
# DoNotSchedule) so a smaller cluster still schedules, just less spread.
|
||
topologySpreadConstraints:
|
||
- maxSkew: 1
|
||
topologyKey: kubernetes.io/hostname
|
||
whenUnsatisfiable: ScheduleAnyway
|
||
labelSelector:
|
||
matchLabels:
|
||
app.kubernetes.io/name: tidaldb
|
||
app.kubernetes.io/component: cluster-node
|
||
securityContext:
|
||
runAsNonRoot: true
|
||
runAsUser: 10001 # the fixed `tidal` uid (docker/deploy/Dockerfile)
|
||
runAsGroup: 10001
|
||
fsGroup: 10001 # makes the mounted PVC group-writable by the runtime user
|
||
seccompProfile:
|
||
type: RuntimeDefault
|
||
initContainers:
|
||
- name: init-datadir
|
||
image: busybox:1.36
|
||
imagePullPolicy: IfNotPresent
|
||
command: ["sh", "-c", "mkdir -p /data/db && chown -R 10001:10001 /data/db"]
|
||
securityContext:
|
||
allowPrivilegeEscalation: false
|
||
runAsUser: 10001
|
||
runAsGroup: 10001
|
||
volumeMounts:
|
||
- name: data
|
||
mountPath: /data
|
||
containers:
|
||
- name: tidaldb
|
||
image: registry.threesix.ai/tidal/server@sha256:171505745b801dcf231b531de6167dbc309a7182957811cbc2228f0a302572b1 # m12-writeburst-rc7 (LIVE; tidaldb commit 1b5bcba). Adds the write-burst false-partition fix on top of rc6 (bd211e7338): a follower whose 1536-D HNSW apply momentarily starves its transport runtime makes a leader ship RPC miss the 10s deadline -> tonic DeadlineExceeded was counted as a transport failure (record_failure) -> both followers' breakers latched Open -> commit stalled -> ack=quorum 503-stormed with no self-heal. Fix (tidal-net, heuristic set only; commit.rs/election.rs/vote path UNTOUCHED): record_timeout opens the breaker ONLY when there's no recent proof of life (last_contact stale -> genuine blackhole still detected); DeadlineExceeded|Cancelled route to it, genuine Unavailable still opens. Verified: 6 unit + 2 real-gRPC integration + slow-fsync multi-process regression. Inherits rc6's seed-join + reseed-loop + election-divergence fixes and rc13's WAL_RETENTION_SEGMENTS=16. amd64 manifest (6 layers, imagetools-verified).
|
||
imagePullPolicy: IfNotPresent
|
||
# The image ENTRYPOINT is the bare binary. We override the command with
|
||
# a tiny /bin/sh wrapper (the bookworm-slim runtime HAS a shell) so we
|
||
# can branch on the pod ordinal: pods 0-2 are the initial voter set
|
||
# (plain topology boot); pods >=3 are SCALE-UP and must seed-join as
|
||
# learners. Keeping this in args (no initContainer, no extra image)
|
||
# means the whole scale story is readable in this one file.
|
||
#
|
||
# POD_NAME is e.g. "tidaldb-4"; ORD is its trailing ordinal. For ORD<3
|
||
# we boot from the topology file (the region IS this pod's name). For
|
||
# ORD>=3 we ALSO pass --seed (any peer; the headless Service load-
|
||
# balances to a live one) + this pod's advertised DNS addresses, and
|
||
# the node learns its roster/id/term from a seed and joins as a learner.
|
||
# The topology ConfigMap is STILL mounted+passed for the behavioral knob
|
||
# blocks (replication/election/...), required even for a --seed boot
|
||
# (m11p5 §3.5); its regions: list is ignored for a seed joiner's roster.
|
||
command: ["/bin/sh", "-c"]
|
||
args:
|
||
- |
|
||
set -eu
|
||
ORD="${POD_NAME##*-}"
|
||
# Per-pod STABLE DNS (headless Service) — what this pod ADVERTISES
|
||
# for peers to dial it directly. The headless Service publishes
|
||
# not-ready addresses (so a pod has DNS before it is Ready), so it
|
||
# resolves to EVERY pod incl. still-joining ones.
|
||
DOMAIN="tidaldb-peers.tidaldb-cluster.svc.cluster.local"
|
||
# READY-ONLY client Service (ClusterIP VIP) — the seed-join discovery
|
||
# target. It excludes not-ready pods, so a joiner always reaches a
|
||
# LIVE serving peer instead of round-robining onto a not-ready pod
|
||
# (often ITSELF, since the headless Service includes the joiner) and
|
||
# failing discovery for the whole 120s window — the real T4 scale-up
|
||
# blocker. Carries its own cert SAN (certs.yaml).
|
||
SEED_SVC="tidaldb.tidaldb-cluster.svc.cluster.local"
|
||
# Common args for every pod.
|
||
set -- cluster \
|
||
--listen 0.0.0.0:9500 \
|
||
--data-dir /data/db \
|
||
--schema /etc/tidal-server/schema/schema.yaml \
|
||
--topology /etc/tidal-server/cluster-topology.yaml \
|
||
--experimental-cluster
|
||
if [ "$ORD" -ge 3 ]; then
|
||
# SCALE-UP pod: seed-join as a learner. Discover a live leader via
|
||
# the READY-ONLY client Service ($SEED_SVC); advertise THIS pod's
|
||
# stable per-pod DNS ($DOMAIN) for gRPC (9601) and HTTP (9500) so
|
||
# peers dial it directly. --metrics gives the joiner a metrics
|
||
# listener (it has no topology entry).
|
||
# m11p7/m12p5: the :9500 plane serves TLS, and `peer_url` honors an
|
||
# explicit URL scheme VERBATIM (forward.rs) — so the seed MUST be
|
||
# `https://`, not `http://` (with `http://` the joiner dials
|
||
# plaintext to the TLS port and seed-join fails). The discovery
|
||
# target is the ready-only client Service, NOT the headless peers
|
||
# Service, so a joiner never round-robins onto a not-ready pod
|
||
# (incl. itself) and burns the 120s discovery window — both were
|
||
# real T4 scale-up blockers.
|
||
set -- "$@" \
|
||
--seed "https://${SEED_SVC}:9500" \
|
||
--advertise-grpc "${POD_NAME}.${DOMAIN}:9601" \
|
||
--advertise-http "${POD_NAME}.${DOMAIN}:9500" \
|
||
--metrics 0.0.0.0:9091
|
||
fi
|
||
exec tidal-server "$@"
|
||
env:
|
||
# Region identity == pod name (tidaldb-0/1/2/...). For ORD<3 this
|
||
# MUST match a region declared in the topology ConfigMap; the names
|
||
# line up by construction (regions are named after the pod identities).
|
||
- name: POD_NAME
|
||
valueFrom:
|
||
fieldRef:
|
||
fieldPath: metadata.name
|
||
- name: TIDAL_REGION
|
||
valueFrom:
|
||
fieldRef:
|
||
fieldPath: metadata.name
|
||
- name: TIDAL_API_KEY
|
||
valueFrom:
|
||
secretKeyRef:
|
||
name: tidaldb-credentials
|
||
key: TIDAL_API_KEY
|
||
# m11p7: the cluster key (mints/verifies per-node internal tokens that
|
||
# authenticate inter-node HTTP). A file mount (not an inline env) so a
|
||
# rotation of the Secret is picked up WITHOUT a pod restart by the
|
||
# credential poller. Distinct secret data key from the bearer.
|
||
- name: TIDAL_CLUSTER_KEY_FILE
|
||
value: /etc/tidaldb/cluster-key/cluster-key
|
||
- name: TIDAL_SERVER_LOG
|
||
value: info
|
||
- name: TIDAL_ALLOW_EXPERIMENTAL_CLUSTER
|
||
value: "1"
|
||
# m12p6: shorten the post-SIGTERM in-flight drain so the (long) HNSW
|
||
# graph save starts promptly within the grace window instead of after
|
||
# the full 15s default. 3s is ample for loopback/in-cluster drain.
|
||
- name: TIDAL_SHUTDOWN_DRAIN_MS
|
||
value: "3000"
|
||
ports:
|
||
- name: http
|
||
containerPort: 9500
|
||
# One gRPC port per hosted shard group (m11p6/m12p4). With the
|
||
# 3-group `shards:` block enabled in the topology ConfigMap, every
|
||
# pod replicates all three groups and binds a derived port per
|
||
# group: shard 0 → 9601, shard 1 → 9602, shard 2 → 9603
|
||
# (`node base port + shard id`; see topology-configmap.yaml). The
|
||
# headless Service reaches each by pod DNS, so these are declared
|
||
# for clarity/NetworkPolicy; the bind itself is driven by the
|
||
# topology. Collapse back to a single `grpc` port if `shards:` is
|
||
# removed (legacy single group).
|
||
- name: grpc
|
||
containerPort: 9601
|
||
- name: grpc-1
|
||
containerPort: 9602
|
||
- name: grpc-2
|
||
containerPort: 9603
|
||
- name: metrics
|
||
containerPort: 9091
|
||
# Three probes map to the three health endpoints. The readinessProbe is
|
||
# now CLUSTER-AWARE (m11p5 §4): /health returns 503 while shutting down,
|
||
# quarantined, removed/decommissioned, or a joiner/install boot has not
|
||
# yet first-converged (lag <= learner_promote_lag, sticky-ready after).
|
||
# A restarted PVC-retained voter is Ready on today's terms (no
|
||
# regression). The full predicate is documented in the kubernetes.md
|
||
# runbook so probe behavior is diagnosable.
|
||
# m11p7: the HTTP plane on :9500 serves TLS (inter-node mTLS), so every
|
||
# probe must use scheme HTTPS. kubelet does NOT verify the server cert
|
||
# for httpGet probes, so the cert's DNS-only SANs (no pod IP) are fine.
|
||
startupProbe:
|
||
httpGet:
|
||
path: /health/startup
|
||
port: http
|
||
scheme: HTTPS
|
||
periodSeconds: 5
|
||
failureThreshold: 240 # ~20 min: HNSW index rebuild/load at 1536-dim is CPU-bound (100k ~5min single-core; headroom for 1M gate)
|
||
livenessProbe:
|
||
httpGet:
|
||
path: /health/live
|
||
port: http
|
||
scheme: HTTPS
|
||
periodSeconds: 10
|
||
timeoutSeconds: 3
|
||
failureThreshold: 3
|
||
readinessProbe:
|
||
httpGet:
|
||
path: /health # cluster-aware: 503 joiner/quarantined/draining
|
||
port: http
|
||
scheme: HTTPS
|
||
periodSeconds: 10
|
||
timeoutSeconds: 3
|
||
failureThreshold: 3
|
||
resources:
|
||
requests:
|
||
cpu: "500m"
|
||
memory: 1Gi
|
||
limits:
|
||
# m12 read-SLA fix: 2->3. The cgroup cpu quota is what the engine's
|
||
# available_parallelism() reads (SEARCH_GATE / worker_threads sizing).
|
||
# At "2" a cross-shard search burst starved the async reactor + the
|
||
# election/heartbeat/apply loops (reads hung to the 30s route timeout;
|
||
# the starved control plane churned elections -> reseed self-exit).
|
||
# "3" leaves ~1 core for kubelet/system on the 4-core nodes; requests
|
||
# stay at 500m so the pod still schedules (server nodes alloc=3).
|
||
cpu: "3"
|
||
memory: 4Gi # m12/1536: 100k×1536 HNSW load peaks ~1.9Gi; 1M needs more headroom (nodes have 13Gi allocatable)
|
||
securityContext:
|
||
allowPrivilegeEscalation: false
|
||
readOnlyRootFilesystem: true # writes only /data (PVC) and /tmp (emptyDir)
|
||
capabilities:
|
||
drop: ["ALL"]
|
||
volumeMounts:
|
||
- name: data
|
||
mountPath: /data
|
||
- name: schema
|
||
mountPath: /etc/tidal-server/schema
|
||
readOnly: true
|
||
- name: topology
|
||
mountPath: /etc/tidal-server/cluster-topology.yaml
|
||
subPath: cluster-topology.yaml
|
||
readOnly: true
|
||
# m11p7 inter-node TLS material (cert-manager Secret). The grpc_tls
|
||
# block in the topology points at these paths. A renewal rewrites the
|
||
# Secret; the kubelet swaps the `..data` symlink and tidalDB's cert
|
||
# poller hot-swaps with zero connection drop.
|
||
- name: cluster-tls
|
||
mountPath: /etc/tidaldb/tls
|
||
readOnly: true
|
||
- name: cluster-key
|
||
mountPath: /etc/tidaldb/cluster-key
|
||
readOnly: true
|
||
- name: tmp
|
||
mountPath: /tmp
|
||
volumes:
|
||
- name: schema
|
||
configMap:
|
||
name: tidaldb-schema
|
||
- name: topology
|
||
configMap:
|
||
name: tidaldb-cluster-topology
|
||
# m11p7: the cert-manager-issued node cert (tls.crt/tls.key/ca.crt).
|
||
- name: cluster-tls
|
||
secret:
|
||
secretName: tidaldb-cluster-tls
|
||
# m11p7: the cluster key for per-node internal tokens (own Secret key).
|
||
- name: cluster-key
|
||
secret:
|
||
secretName: tidaldb-credentials
|
||
items:
|
||
- key: TIDAL_CLUSTER_KEY
|
||
path: cluster-key
|
||
- name: tmp
|
||
emptyDir: {}
|
||
volumeClaimTemplates:
|
||
- metadata:
|
||
name: data
|
||
labels:
|
||
app.kubernetes.io/name: tidaldb
|
||
spec:
|
||
accessModes: ["ReadWriteOnce"]
|
||
storageClassName: local-path
|
||
resources:
|
||
requests:
|
||
storage: 5Gi
|