/** * Runbook section 1 — the cluster is up and converged. * * CAP-001 membership is complete and every voter is Ready * CAP-002 every node has converged: zero lag, no reseed, one agreed leader * * Why each node is asked individually rather than asking the aggregate: * `GET /cluster/status` can report a peer it holds no frontier report for as * `applied_events: 0` and then derive lag against that zero, so a fully * converged peer shows up as the leader's entire history behind. The per-node * `/cluster/status/local` is the authoritative view. See CAP-012 and * docs/ops/observability.md section 4. */ import { expect, test } from '@playwright/test'; import { kubectl, waitForPodsReady, withPortForward } from '../support/cluster'; import { observed, recordJson } from '../support/evidence'; import { NAMESPACE, POD_NAMES, PORT_CLIENT, apiKey } from '../support/env'; /** One shard group's replication position on one node. */ type ShardStatus = { shard: number; is_leader: boolean; leader: string; term: number; role: string; applied_events: number; leader_seqno: number; lag_events: number; reseed_required: boolean; reseeding: boolean; }; type LocalStatus = { region: string; leader: string; reseed_required: boolean; reseeding: boolean; quarantined: boolean; partitioned: string[]; applied_events: number; lag_events: number; shards: ShardStatus[]; }; test.describe('section 1 — cluster convergence', () => { test('membership is exactly the known voter set and every pod is Ready', async ({}, testInfo) => { const result = await observed(testInfo, 'get pods wide', () => kubectl([ '-n', NAMESPACE, 'get', 'pods', '-l', 'app.kubernetes.io/name=tidaldb', '-o', 'jsonpath={range .items[*]}{.metadata.name}{"\\t"}{.status.containerStatuses[0].ready}{"\\t"}{.status.containerStatuses[0].restartCount}{"\\t"}{.status.phase}{"\\n"}{end}', ]), ); expect(result.code, result.stderr).toBe(0); const pods = result.stdout .trim() .split('\n') .filter((line) => line.trim() !== '') .map((line) => { const [name, ready, restarts, phase] = line.split('\t'); return { name, ready: ready === 'true', restarts: Number.parseInt(restarts, 10), phase, }; }); await recordJson(testInfo, 'pod-inventory', pods); // An unexpected extra pod means a scale operation is mid-flight or an // orphan survived — either way the voter set is not what the runbook // assumes, so assert the exact set rather than a minimum count. expect( pods.map((p) => p.name).sort(), 'voter set must be exactly the documented pods', ).toEqual([...POD_NAMES].sort()); for (const pod of pods) { expect(pod.phase, `${pod.name} phase`).toBe('Running'); } // Readiness is polled, not sampled: `reseed_self_restart: true` makes a // bounded exit(0)/reinstall cycle DESIGNED behavior, so one unlucky sample // would report a converging cluster as a broken one. const readiness = await waitForPodsReady(NAMESPACE, POD_NAMES); await recordJson(testInfo, 'pod-readiness-timeline', readiness); expect( readiness.converged, `pods did not all reach Ready within the budget; final=${JSON.stringify(readiness.final)}`, ).toBe(true); }); test('every node reports zero lag, no reseed, and agrees on one leader', async ({ playwright, }, testInfo) => { const key = apiKey(); const statuses: LocalStatus[] = []; for (const pod of POD_NAMES) { const status = await test.step(`read ${pod} /cluster/status/local`, async () => withPortForward(NAMESPACE, pod, PORT_CLIENT, async (forward) => { // The pod serves TLS with the internal cluster CA, whose leaf is // issued for in-cluster DNS names — a 127.0.0.1 tunnel cannot // validate it. ignoreHTTPSErrors is scoped to this one context: a // local port-forward to a named pod, never the public endpoint. const context = await playwright.request.newContext({ ignoreHTTPSErrors: true, extraHTTPHeaders: { authorization: `Bearer ${key}` }, }); try { const response = await context.get( `https://127.0.0.1:${forward.localPort}/cluster/status/local`, ); expect(response.status(), `${pod} status endpoint`).toBe(200); return (await response.json()) as LocalStatus; } finally { await context.dispose(); } })); statuses.push(status); } await recordJson( testInfo, 'per-node-convergence', statuses.map((status) => ({ region: status.region, leader: status.leader, reseed_required: status.reseed_required, reseeding: status.reseeding, quarantined: status.quarantined, partitioned: status.partitioned, shards: status.shards.map((shard) => ({ shard: shard.shard, leader: shard.leader, term: shard.term, role: shard.role, applied_events: shard.applied_events, lag_events: shard.lag_events, reseed_required: shard.reseed_required, })), })), ); expect(statuses.length, 'one status per pod').toBe(POD_NAMES.length); for (const status of statuses) { expect(status.region, 'each pod reports its own region').toBeTruthy(); // reseed_required surviving a restart is the m11p5 livelock signature. expect(status.reseed_required, `${status.region} must not require reseed`).toBe(false); expect(status.reseeding, `${status.region} must not be reseeding`).toBe(false); expect(status.quarantined, `${status.region} must not be quarantined`).toBe(false); expect(status.partitioned, `${status.region} must see no partitions`).toEqual([]); expect(status.shards.length, `${status.region} shard groups`).toBe(3); for (const shard of status.shards) { expect( shard.lag_events, `${status.region} shard ${shard.shard} must have zero lag`, ).toBe(0); expect( shard.reseed_required, `${status.region} shard ${shard.shard} must not require reseed`, ).toBe(false); expect( shard.applied_events, `${status.region} shard ${shard.shard} should have applied real events`, ).toBeGreaterThan(0); } } // Agreement, not identity: leadership legitimately moves between runs (it // moved from tidaldb-2 to tidaldb-1 during this suite's development), so // pinning a node name would produce a test that fails on a healthy // election. What must hold is that every node names the SAME leader for // each shard group — disagreement is split brain. for (let shardIndex = 0; shardIndex < 3; shardIndex += 1) { const leaders = [ ...new Set( statuses.map( (status) => status.shards.find((s) => s.shard === shardIndex)?.leader, ), ), ]; expect( leaders, `all nodes must agree on the leader of shard ${shardIndex}`, ).toHaveLength(1); expect(leaders[0], `shard ${shardIndex} must have a leader`).toBeTruthy(); } }); });