tidaldb/k8s/cluster/statefulset.yaml
jx12n 31ee612f27 feat(m12p4): sharded ingestion — scatter-gather pool + cross-shard unified reads (L4)
Scale write throughput across data-shard groups while keeping a single unified
read surface:

- scatter_gather.rs: pooled fan-out across shard groups (replaces per-request
  client construction); cross-shard query results merged on one node
- cluster/node.rs: cross-shard read routing — a read on any node gathers from
  every shard group's leader and unions results
- cluster/forward.rs: fix h2 204 forward-relay bug (relay_forwarded skips body
  for 1xx/204/304 — synthesized JSON body on a 204 triggered HTTP/2 RST_STREAM
  on the real mTLS plane)
- dto.rs: cross-shard query/result DTOs
- k8s/cluster/: enable 3-group `shards:` topology (statefulset, service-peers,
  topology-configmap)
- k8s/cluster-local-kind/: local-kind overlay to run the T5 gate without Ref-A
- tidal-stress/k8s/stress-job-t5.yaml: 2-generator sharded throughput job
- tests: cluster_cross_shard_reads.rs + multiproc support; ran real on kind
- docs/profiling/m12p4-t5-sharded-throughput.md: T5 throughput findings
2026-06-14 15:17:35 -06:00

271 lines
12 KiB
YAML

# The tidalDB CLUSTER: ONE StatefulSet, every pod a region (m11p5 §4).
#
# MUTUALLY EXCLUSIVE with the standalone set in k8s/ (namespace `tidaldb`,
# replicas: 1, `standalone` subcommand). This is namespace `tidaldb-cluster`,
# replicas: 3, the `cluster --region` subcommand, real quorum-ack writes,
# automatic election, and elastic membership. Deploy ONE or the OTHER per
# namespace — never both.
#
# WHY ONE StatefulSet (not one-per-region): the m11p5 bind/advertise split lets
# every pod mount the SAME topology ConfigMap (peers are advertised by per-pod
# DNS; the local socket binds 0.0.0.0), so a single StatefulSet with stable pod
# identities tidaldb-{0,1,2} IS the three regions. Scaling is `kubectl scale`
# (see docs/runbooks/kubernetes.md): pod N>=3 auto-seed-joins as a learner and
# auto-promotes to a voter — no file edits, no per-pod manifests.
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: tidaldb
namespace: tidaldb-cluster
labels:
app.kubernetes.io/name: tidaldb
app.kubernetes.io/component: cluster-node
spec:
serviceName: tidaldb-peers # the headless peer Service — stable per-pod DNS
replicas: 3 # the initial voter set; scale up/down per the runbook
# Parallel: bring all pods up at once. There is no ordered-bootstrap
# dependency — siblings boot in any order (an unreachable-at-startup peer is
# normal; the election + catch-up timer converge them). Ordered start would
# only serialize a 3-region cold boot for no benefit.
podManagementPolicy: Parallel
selector:
matchLabels:
app.kubernetes.io/name: tidaldb
app.kubernetes.io/component: cluster-node
template:
metadata:
labels:
app.kubernetes.io/name: tidaldb
app.kubernetes.io/component: cluster-node
annotations:
# Plain-Prometheus scrape hints (per-pod :9091, unauthenticated — keep
# cluster-internal). The Operator-native path is a PodMonitor/ServiceMonitor.
prometheus.io/scrape: "true"
prometheus.io/port: "9091"
prometheus.io/path: "/metrics"
spec:
# SIGTERM flips readiness to 503 (pod leaves the client Service), drains
# in-flight requests, then checkpoints + fsyncs the WAL before exit. The
# cluster path also wants the leader-lease/heartbeat windows to lapse so a
# successor is elected cleanly. 60s covers the whole sequence.
terminationGracePeriodSeconds: 60
# Spread the three pods across distinct nodes so a single node loss takes
# at most one voter — preserving quorum (2 of 3). ScheduleAnyway (not
# DoNotSchedule) so a smaller cluster still schedules, just less spread.
topologySpreadConstraints:
- maxSkew: 1
topologyKey: kubernetes.io/hostname
whenUnsatisfiable: ScheduleAnyway
labelSelector:
matchLabels:
app.kubernetes.io/name: tidaldb
app.kubernetes.io/component: cluster-node
securityContext:
runAsNonRoot: true
runAsUser: 10001 # the fixed `tidal` uid (docker/deploy/Dockerfile)
runAsGroup: 10001
fsGroup: 10001 # makes the mounted PVC group-writable by the runtime user
seccompProfile:
type: RuntimeDefault
initContainers:
- name: init-datadir
image: busybox:1.36
imagePullPolicy: IfNotPresent
command: ["sh", "-c", "mkdir -p /data/db && chown -R 10001:10001 /data/db"]
securityContext:
allowPrivilegeEscalation: false
runAsUser: 10001
runAsGroup: 10001
volumeMounts:
- name: data
mountPath: /data
containers:
- name: tidaldb
image: registry.threesix.ai/tidal/server@sha256:8b136de21b969adedee37fdcb0cac15ecdd77b9beb5c5f31be3a5d15340323b1 # m11-44b768b (p6 sharding + p7 mTLS + p8 ops + p9 correctness)
imagePullPolicy: IfNotPresent
# The image ENTRYPOINT is the bare binary. We override the command with
# a tiny /bin/sh wrapper (the bookworm-slim runtime HAS a shell) so we
# can branch on the pod ordinal: pods 0-2 are the initial voter set
# (plain topology boot); pods >=3 are SCALE-UP and must seed-join as
# learners. Keeping this in args (no initContainer, no extra image)
# means the whole scale story is readable in this one file.
#
# POD_NAME is e.g. "tidaldb-4"; ORD is its trailing ordinal. For ORD<3
# we boot from the topology file (the region IS this pod's name). For
# ORD>=3 we ALSO pass --seed (any peer; the headless Service load-
# balances to a live one) + this pod's advertised DNS addresses, and
# the node learns its roster/id/term from a seed and joins as a learner.
# The topology ConfigMap is STILL mounted+passed for the behavioral knob
# blocks (replication/election/...), required even for a --seed boot
# (m11p5 §3.5); its regions: list is ignored for a seed joiner's roster.
command: ["/bin/sh", "-c"]
args:
- |
set -eu
ORD="${POD_NAME##*-}"
DOMAIN="tidaldb-peers.tidaldb-cluster.svc.cluster.local"
# Common args for every pod.
set -- cluster \
--listen 0.0.0.0:9500 \
--data-dir /data/db \
--schema /etc/tidal-server/schema/schema.yaml \
--topology /etc/tidal-server/cluster-topology.yaml \
--experimental-cluster
if [ "$ORD" -ge 3 ]; then
# SCALE-UP pod: seed-join as a learner. Advertise THIS pod's
# stable per-pod DNS for gRPC (9601) and HTTP (9500); point --seed
# at the headless Service (it resolves to a live peer). --metrics
# gives the joiner a metrics listener (it has no topology entry).
set -- "$@" \
--seed "http://${DOMAIN}:9500" \
--advertise-grpc "${POD_NAME}.${DOMAIN}:9601" \
--advertise-http "${POD_NAME}.${DOMAIN}:9500" \
--metrics 0.0.0.0:9091
fi
exec tidal-server "$@"
env:
# Region identity == pod name (tidaldb-0/1/2/...). For ORD<3 this
# MUST match a region declared in the topology ConfigMap; the names
# line up by construction (regions are named after the pod identities).
- name: POD_NAME
valueFrom:
fieldRef:
fieldPath: metadata.name
- name: TIDAL_REGION
valueFrom:
fieldRef:
fieldPath: metadata.name
- name: TIDAL_API_KEY
valueFrom:
secretKeyRef:
name: tidaldb-credentials
key: TIDAL_API_KEY
# m11p7: the cluster key (mints/verifies per-node internal tokens that
# authenticate inter-node HTTP). A file mount (not an inline env) so a
# rotation of the Secret is picked up WITHOUT a pod restart by the
# credential poller. Distinct secret data key from the bearer.
- name: TIDAL_CLUSTER_KEY_FILE
value: /etc/tidaldb/cluster-key/cluster-key
- name: TIDAL_SERVER_LOG
value: info
- name: TIDAL_ALLOW_EXPERIMENTAL_CLUSTER
value: "1"
ports:
- name: http
containerPort: 9500
# One gRPC port per hosted shard group (m11p6/m12p4). With the
# 3-group `shards:` block enabled in the topology ConfigMap, every
# pod replicates all three groups and binds a derived port per
# group: shard 0 → 9601, shard 1 → 9602, shard 2 → 9603
# (`node base port + shard id`; see topology-configmap.yaml). The
# headless Service reaches each by pod DNS, so these are declared
# for clarity/NetworkPolicy; the bind itself is driven by the
# topology. Collapse back to a single `grpc` port if `shards:` is
# removed (legacy single group).
- name: grpc
containerPort: 9601
- name: grpc-1
containerPort: 9602
- name: grpc-2
containerPort: 9603
- name: metrics
containerPort: 9091
# Three probes map to the three health endpoints. The readinessProbe is
# now CLUSTER-AWARE (m11p5 §4): /health returns 503 while shutting down,
# quarantined, removed/decommissioned, or a joiner/install boot has not
# yet first-converged (lag <= learner_promote_lag, sticky-ready after).
# A restarted PVC-retained voter is Ready on today's terms (no
# regression). The full predicate is documented in the kubernetes.md
# runbook so probe behavior is diagnosable.
# m11p7: the HTTP plane on :9500 serves TLS (inter-node mTLS), so every
# probe must use scheme HTTPS. kubelet does NOT verify the server cert
# for httpGet probes, so the cert's DNS-only SANs (no pod IP) are fine.
startupProbe:
httpGet:
path: /health/startup
port: http
scheme: HTTPS
periodSeconds: 5
failureThreshold: 60 # ~5 min for large-DB WAL replay / index load
livenessProbe:
httpGet:
path: /health/live
port: http
scheme: HTTPS
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 3
readinessProbe:
httpGet:
path: /health # cluster-aware: 503 joiner/quarantined/draining
port: http
scheme: HTTPS
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 3
resources:
requests:
cpu: "250m"
memory: 256Mi
limits:
cpu: "2"
memory: 2Gi # size from docs/ops/capacity-planning.md
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true # writes only /data (PVC) and /tmp (emptyDir)
capabilities:
drop: ["ALL"]
volumeMounts:
- name: data
mountPath: /data
- name: schema
mountPath: /etc/tidal-server/schema
readOnly: true
- name: topology
mountPath: /etc/tidal-server/cluster-topology.yaml
subPath: cluster-topology.yaml
readOnly: true
# m11p7 inter-node TLS material (cert-manager Secret). The grpc_tls
# block in the topology points at these paths. A renewal rewrites the
# Secret; the kubelet swaps the `..data` symlink and tidalDB's cert
# poller hot-swaps with zero connection drop.
- name: cluster-tls
mountPath: /etc/tidaldb/tls
readOnly: true
- name: cluster-key
mountPath: /etc/tidaldb/cluster-key
readOnly: true
- name: tmp
mountPath: /tmp
volumes:
- name: schema
configMap:
name: tidaldb-schema
- name: topology
configMap:
name: tidaldb-cluster-topology
# m11p7: the cert-manager-issued node cert (tls.crt/tls.key/ca.crt).
- name: cluster-tls
secret:
secretName: tidaldb-cluster-tls
# m11p7: the cluster key for per-node internal tokens (own Secret key).
- name: cluster-key
secret:
secretName: tidaldb-credentials
items:
- key: TIDAL_CLUSTER_KEY
path: cluster-key
- name: tmp
emptyDir: {}
volumeClaimTemplates:
- metadata:
name: data
labels:
app.kubernetes.io/name: tidaldb
spec:
accessModes: ["ReadWriteOnce"]
storageClassName: local-path
resources:
requests:
storage: 5Gi