# T5 — sharded quorum-write throughput at scale (m12p4 / G3). # # Goal: prove the 3-group × RF=3 topology scales WRITES ~S× over a single group. # With `shards:` enabled (k8s/cluster/topology-configmap.yaml) the regular # `/signals`//`/items`//`/embeddings` surface hash-routes each entity's write to # its OWNING group's leader, so the three balanced leaders (tidaldb-0/1/2 lead # shards 0/1/2) absorb writes in parallel instead of one leader funnelling all. # # TWO generators. The m11p6 knee was the GENERATOR's in-flight cap (~4k rps), # not the engine (~30% CPU at 3k/s). A single generator cannot saturate three # parallel leaders, so this Job runs `parallelism: 2` (completionMode: Indexed): # two generator pods drive the ramp concurrently and their throughput sums. # Both spread writes across all three `--target`s (no `--leader-url` pin), so # every pod gateways AND leads one group — no single forwarding bottleneck. # # Aggregate the two pods' `ok_per_sec` (each logs its own JSON summary) for the # cluster total. Gate: aggregate ≥ 5,000 quorum signal writes/s AND ≥ 2.5× the # single-group baseline (stress-job-m11p6-baseline.yaml), zero acked loss per # group, 0% error. The single-generator absolute on constrained hardware may sit # below 5k; the SCALING RATIO (3-group vs single-group) is the architecture # proof and is hardware-independent. # # TLS: the :9500 plane serves a private-CA cert (tidaldb-cluster-tls). Each # generator trusts it via the mounted ca.crt (--ca-cert) — verified TLS, never # --insecure. Targets are the pod-DNS SANs (https://), never pod IPs. # # Apply: kubectl apply -f tidal-stress/k8s/stress-job-t5.yaml # Watch: kubectl logs -f job/tidal-stress-t5 -n tidaldb-cluster --all-containers --prefix # Sum: grep '"ok_per_sec"' across both pods' --json-summary outputs # Rearm: kubectl delete job tidal-stress-t5 -n tidaldb-cluster apiVersion: batch/v1 kind: Job metadata: name: tidal-stress-t5 namespace: tidaldb-cluster labels: app.kubernetes.io/name: tidal-stress app.kubernetes.io/part-of: tidaldb spec: backoffLimit: 0 ttlSecondsAfterFinished: 7200 # Two generator pods, both running to completion — the "second stress # generator" m12p4 calls for. Indexed so each pod has a stable # JOB_COMPLETION_INDEX (0/1) in its logs for per-generator attribution. completions: 2 parallelism: 2 completionMode: Indexed template: metadata: labels: app.kubernetes.io/name: tidal-stress app.kubernetes.io/part-of: tidaldb spec: restartPolicy: Never automountServiceAccountToken: false # Spread the two generators onto distinct nodes so neither generator pod # competes with the other for CPU — each must be free to drive its full # in-flight budget (the m11p6 knee was generator CPU/in-flight, not engine). topologySpreadConstraints: - maxSkew: 1 topologyKey: kubernetes.io/hostname whenUnsatisfiable: ScheduleAnyway labelSelector: matchLabels: app.kubernetes.io/name: tidal-stress securityContext: runAsNonRoot: true runAsUser: 1000 runAsGroup: 1000 seccompProfile: type: RuntimeDefault containers: - name: stress image: registry.threesix.ai/tidal/stress@sha256:e130aa871f5df17a14a9e13e7df606c602b95a03d8eba49490ff7481e6e2b2b3 # m11-44b768b (TLS-aware) imagePullPolicy: IfNotPresent args: # Spread writes across all three pods (each gateways + leads one # group); NO --leader-url pin, so the forwarding load is balanced. - --target - https://tidaldb-0.tidaldb-peers.tidaldb-cluster.svc.cluster.local:9500 - --target - https://tidaldb-1.tidaldb-peers.tidaldb-cluster.svc.cluster.local:9500 - --target - https://tidaldb-2.tidaldb-peers.tidaldb-cluster.svc.cluster.local:9500 - --ca-cert - /etc/tidaldb/tls/ca.crt - --ack - quorum - --ramp - peach-100k - --stage-secs - "120" - --mix - writes - --corpus - "20000" - --users - "100000" # Each generator drives its own in-flight budget; two pods double the # aggregate offered load over a single generator. - --max-inflight - "5000" - --poll-status # m11p9 machine-readable gates: a per-generator JSON summary plus # hard ceilings. 0% error + no knee are the per-group zero-acked-loss # / SLO bar; aggregate the two pods' ok_per_sec for the 5k/2.5x gate. - --json-summary - /tmp/t5-summary.json - --max-error-pct - "0" env: - name: TIDAL_API_KEY valueFrom: secretKeyRef: name: tidaldb-credentials key: TIDAL_API_KEY - name: TIDAL_STRESS_LOG value: warn resources: requests: cpu: 500m memory: 256Mi limits: cpu: "3" memory: 1Gi securityContext: allowPrivilegeEscalation: false readOnlyRootFilesystem: true capabilities: drop: ["ALL"] volumeMounts: - name: cluster-tls mountPath: /etc/tidaldb/tls readOnly: true - name: tmp mountPath: /tmp volumes: - name: cluster-tls secret: secretName: tidaldb-cluster-tls items: - key: ca.crt path: ca.crt - name: tmp emptyDir: {}