From 9f3f5e7f86193b1cbbcd6e4f96941ebb596d34c0 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 11:17:52 +0300
Subject: [PATCH 01/64] feat(ledger): Final check one-click sign-off +
tee-overlap dedupe + rate-display fixes
Field feedback from recent production migrations:
- Final check (dashboard card + POST /control/final-check + /final-check.txt):
runs chunk states, DLQ, the full source recount with cd-checksum
fingerprints and the sampled content audit, applies the tee/cutover
interpretation itself, and answers 'is it safe to decommission the old
cluster?' as PASS / PASS WITH NOTES / FAIL in plain sentences. The source
audit now takes an upToMs cutover clamp: post-cutover windows (where the
mirror keeps feeding the old side) are excluded and explained instead of
flagging as false mismatches.
- Tee-overlap dedupe (POST /control/dedupe-overlap): removes the duplicates
a run without LEDGER_CD_UPPER_BOUND created on a mirrored cutover. The
migrated copy's _id exists in old Mongo, the native one's doesn't, so the
cleanup is exact. Dry-run by default; execute is licensed by a completed
dry run over the same window. Must run before old-cluster teardown.
- Rate display: /stats gains a 10-min clusterSlow window so a freshly opened
dashboard tab shows a real docs/s immediately (huge-chunk runs showed 0);
completed runs show the whole-run average from the run timeline instead of
the surviving pod's own counter (4-pod production run showed 5,060 vs the real
~20,300); armed-confirm window 4s -> 8s.
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 46 +++++
src/http/ledger-viz-route.ts | 90 +++++++++-
src/runtime/dedupe-overlap.ts | 140 +++++++++++++++
src/runtime/final-check.ts | 199 +++++++++++++++++++++
src/runtime/ledger-engine.ts | 52 +++++-
src/runtime/ledger-rebuild.ts | 28 ++-
src/target/staging-manager.ts | 36 ++++
tests/integration/dedupe-overlap.test.ts | 123 +++++++++++++
tests/integration/final-check.test.ts | 213 +++++++++++++++++++++++
9 files changed, 913 insertions(+), 14 deletions(-)
create mode 100644 src/runtime/dedupe-overlap.ts
create mode 100644 src/runtime/final-check.ts
create mode 100644 tests/integration/dedupe-overlap.test.ts
create mode 100644 tests/integration/final-check.test.ts
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 5a06e4d..8cef6d9 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -55,6 +55,52 @@ once, for minutes, at cutover — never for the migration.
| A doc CRASHES the process every time (poison pill) | After 3 crash-retries the chunk is auto-split instead of retried; repeated splitting converges on a ≤1-min window quarantined as a tiny failed chunk — everything else migrates (verified: 20k-doc drill localized 1 poison doc to a 2-doc window in 25 restarts) | Inspect the few source docs in the failed chunk's cd window; fix/remove them, then `POST /control/retry-failed` |
| Live ClickHouse itself must be rebuilt | Live events still sit in the Kafka log; history still sits in frozen Mongo | Recreate table → reset ONLY the ClickHouse-sink connector's offsets to earliest (aggregator groups untouched) → re-run the migrator |
+## Final check — the one-click sign-off
+
+Don't interpret audit buckets by hand: the **Final check** runs everything
+(chunk states, DLQ, full source recount, cd-checksum fingerprints, sampled
+content comparison), applies the tee/cutover rules itself, and answers the
+only question that matters — *is it safe to decommission the old cluster?* —
+as **PASS / PASS WITH NOTES / FAIL** in plain sentences with the action named
+on every red line.
+
+- Dashboard: the **Final check** card → *Run final check*. On tee/mirror runs
+ without a stored bound, type the cutover time into the field first.
+- SSH-only:
+
+```bash
+# start (add {"cutoverMs": } for mirror runs without a stored bound)
+curl -s -X POST localhost:PORT/control/final-check -H 'content-type: application/json' -d '{}'
+# read the verdict (re-run until it says PASS/FAIL; shows progress while running)
+curl -s localhost:PORT/final-check.txt
+```
+
+A stored/env cd bound is picked up automatically as the cutover. Post-cutover
+source windows are excluded and explained in a note — divergence there is the
+mirror still feeding the old side, not data loss. Run it while the old
+cluster is still up: the source is the reference.
+
+## Tee-overlap dedupe — fixing a missing bound after the fact
+
+A mirrored cutover migrated WITHOUT `LEDGER_CD_UPPER_BOUND` copies the
+mirror's re-ingested docs on top of natively ingested rows: every event in
+the overlap window (tee flip → migration completion) exists twice in
+ClickHouse. The copies are separable — the migrated copy's `_id` exists in
+the old cluster's Mongo; the native one's doesn't — so cleanup is exact and
+loses nothing. **Must run before the old cluster is decommissioned** (old
+Mongo is the separator).
+
+```bash
+# 1. DRY RUN (counts only): fromMs = tee flip / IP swap, toMs = migration completion
+curl -s -X POST localhost:PORT/control/dedupe-overlap -H 'content-type: application/json' \
+ -d '{"fromMs": 1789700000000, "toMs": 1789794970435}'
+curl -s localhost:PORT/api/dedupe-overlap # totals.chMatched = the duplicates
+# 2. EXECUTE (refused unless the dry run over the SAME window completed first)
+curl -s -X POST localhost:PORT/control/dedupe-overlap -H 'content-type: application/json' \
+ -d '{"fromMs": 1789700000000, "toMs": 1789794970435, "execute": true}'
+# 3. re-run the Final check with the same cutover to confirm
+```
+
## Verification cheat sheet
```sql
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index b336a87..7974ce9 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -300,6 +300,15 @@ const PAGE = `
Destructive actions ask for a second click. Every action shows a receipt.
+
+
Final check — one click answers: is it safe to decommission the old source? (chunks + DLQ + full source recount + checksums + content samples — interpreted for you)
+
+
+
+
+
Not run. Run it after the migration completes — it recounts every window against the source, so give it time on big runs; progress shows here. SSH-only: curl -X POST :PORT/control/final-check then curl :PORT/final-check.txt
+
+
Collections
Waiting for first chunk…
@@ -513,7 +522,7 @@ async function control(action, okMsg, btn, needsConfirm) {
btn.dataset.label = btn.textContent;
btn.textContent = 'Click again to confirm';
btn.classList.add('armed');
- setTimeout(() => { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 4000);
+ setTimeout(() => { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 8000);
return;
}
if (btn) { armed.delete(btn); if (btn.dataset.label) { btn.textContent = btn.dataset.label; btn.classList.remove('armed'); } btn.disabled = true; }
@@ -534,7 +543,7 @@ async function startRebuild(btn, force) {
btn.dataset.label = btn.textContent;
btn.textContent = 'Click again to confirm';
btn.classList.add('armed');
- setTimeout(() => { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 4000);
+ setTimeout(() => { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 8000);
return;
}
armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); btn.disabled = true;
@@ -720,6 +729,59 @@ function updatePhaseBadges() {
}
updatePhaseBadges();
+async function startFinalCheck(btn) {
+ var body = {};
+ var cutRaw = (document.getElementById('fc-cutover').value || '').trim();
+ if (cutRaw) {
+ var ms = Date.parse(cutRaw);
+ if (isNaN(ms)) { toast('Could not parse the cutover time — use ISO like 2026-09-18T18:00Z'); return; }
+ body.cutoverMs = ms;
+ }
+ btn.disabled = true;
+ try {
+ var res = await fetch('/control/final-check', { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body) });
+ var out = await res.json();
+ if (!out.started) toast('Not started: ' + (out.reason || 'unknown'));
+ else toast('Final check started — the verdict will appear below');
+ } catch (e) { toast('failed: ' + e.message); }
+ btn.disabled = false;
+ pollFinalCheck();
+}
+var fcTimer = null;
+async function pollFinalCheck() {
+ try {
+ var fc = await fetch('/api/final-check').then(function (r) { return r.json(); });
+ renderFinalCheck(fc);
+ if (fc.status === 'running') { clearTimeout(fcTimer); fcTimer = setTimeout(pollFinalCheck, 2000); }
+ } catch (e) { /* engine restarting — next poll or reload recovers */ }
+}
+function fcEsc(s) { var d = document.createElement('div'); d.textContent = s == null ? '' : String(s); return d.innerHTML; }
+function renderFinalCheck(fc) {
+ var el = document.getElementById('finalcheck-out');
+ if (!el || !fc || fc.status === 'not_run') return;
+ if (fc.status === 'running') {
+ var a = fc.audit || {};
+ el.innerHTML = '
cutover applied: source compared only for cd < ' + new Date(fc.cutoverMs).toISOString() + '
';
+ el.innerHTML = html;
+}
+
async function tick() {
try {
const [stats, chunkResp] = await Promise.all([
@@ -760,15 +822,27 @@ async function tick() {
if (changes >= 3 && (last.t - windowStart.t) >= 10000) break;
}
var rspan = (last.t - windowStart.t) / 1000;
- var liveRate = rspan >= 10 ? Math.max(0, (last.d - windowStart.d) / rspan) : null;
+ // the client window is only trustworthy once it has WITNESSED chunk
+ // completions — a freshly opened tab on a huge-chunk run showed "0"
+ // (field report); until then the server's 10-min ledger window is truth
+ var clientReady = changes >= 3 && rspan >= 10;
+ var slowRate = stats.clusterSlow && stats.clusterSlow.docsPerSecond > 0 ? stats.clusterSlow.docsPerSecond : null;
+ var liveRate = clientReady ? Math.max(0, (last.d - windowStart.d) / rspan)
+ : slowRate !== null ? slowRate
+ : null;
var multiPod = stats.cluster && stats.cluster.pods > 1;
var effRate = liveRate !== null ? liveRate
: multiPod && stats.status === 'running' ? stats.cluster.docsPerSecond
: stats.docsPerSecond;
if (stats.status === 'completed') {
- // a pod restarted after completion migrated nothing itself — its
- // local average is 0 and would read as an anomaly
- dpsEl.textContent = stats.docsPerSecond >= 1 ? fmt(stats.docsPerSecond) + ' avg' : '\u2013';
+ // whole-RUN average from ledger docs + run timeline; the pod's own
+ // lifetime counter is only its share of a multi-pod run (field: a
+ // 4-pod run showed 5,060 instead of the run's ~20,300), and a pod
+ // restarted after completion migrated nothing at all
+ var rt = stats.runTimes || {};
+ var runSec = rt.startedAtMs && rt.completedAtMs ? (rt.completedAtMs - rt.startedAtMs) / 1000 : 0;
+ var runAvg = runSec > 0 && sum.docsDone > 0 ? sum.docsDone / runSec : stats.docsPerSecond;
+ dpsEl.textContent = runAvg >= 1 ? fmt(Math.round(runAvg)) + ' avg' : '\u2013';
} else if (liveRate !== null) {
dpsEl.textContent = fmt(Math.round(liveRate)) + (multiPod ? ' \u00b7 ' + stats.cluster.pods + ' pods' : '');
} else if (stats.status === 'running') {
@@ -1048,7 +1122,7 @@ async function applyBound(btn, ms) {
btn.dataset.label = btn.textContent;
btn.textContent = 'Click again to confirm';
btn.classList.add('armed');
- setTimeout(function() { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 4000);
+ setTimeout(function() { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 8000);
return;
}
armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); btn.disabled = true;
@@ -1223,7 +1297,7 @@ async function slowTick() {
} catch { /* engine restarting */ }
}
-tick(); slowTick();
+tick(); slowTick(); pollFinalCheck();
setInterval(tick, 2000);
setInterval(slowTick, 5000);
diff --git a/src/runtime/dedupe-overlap.ts b/src/runtime/dedupe-overlap.ts
new file mode 100644
index 0000000..465e0be
--- /dev/null
+++ b/src/runtime/dedupe-overlap.ts
@@ -0,0 +1,140 @@
+/**
+ * Tee-overlap dedupe: remove the duplicates a missing cd bound created.
+ *
+ * On a mirrored cutover (new cluster primary, nginx mirroring to the old
+ * stack — or the reverse) a migration run WITHOUT LEDGER_CD_UPPER_BOUND
+ * copies the mirror's re-ingested docs on top of the rows the new cluster
+ * already ingested natively: every event in the overlap window exists twice
+ * in ClickHouse, under two different _ids.
+ *
+ * The two copies are cleanly separable: the migrated copy carries an _id
+ * that exists in the OLD cluster's Mongo; the native row's _id was minted by
+ * the new cluster and does not. And because the mirror only ever re-ingests
+ * requests the new cluster served first, every migrated row in the overlap
+ * window duplicates a native row — deleting all id-matched rows in the
+ * window removes exactly the duplicates, never data.
+ *
+ * Safety: dry-run by default (counts only); execute is refused until a dry
+ * run over the SAME window has completed in this process, and the old
+ * cluster's Mongo must still be reachable (it is the separator — this is
+ * why cleanup must happen BEFORE the old stack is decommissioned).
+ */
+
+import type { Logger } from 'pino';
+import { MongoClient } from 'mongodb';
+import type { Config } from '../config/schema.ts';
+import { StagingManager } from '../target/staging-manager.ts';
+import { discoverCollections } from '../source/discover-collections.ts';
+
+export interface DedupeOverlapState {
+ status: 'not_run' | 'running' | 'completed' | 'failed';
+ phase: string;
+ execute: boolean;
+ fromMs: number | null;
+ toMs: number | null;
+ collections: Array<{ collection: string; mongoDocsInWindow: number; chMatched: number; deleted: number }>;
+ totals: { mongoDocsInWindow: number; chMatched: number; deleted: number };
+ /** Window of the last COMPLETED dry run — the license to execute. */
+ lastDryRun: { fromMs: number; toMs: number; chMatched: number; at: number } | null;
+ error: string | null;
+ startedAt: number | null;
+ finishedAt: number | null;
+}
+
+export function newDedupeOverlapState(): DedupeOverlapState {
+ return {
+ status: 'not_run', phase: '', execute: false, fromMs: null, toMs: null,
+ collections: [], totals: { mongoDocsInWindow: 0, chMatched: 0, deleted: 0 },
+ lastDryRun: null, error: null, startedAt: null, finishedAt: null,
+ };
+}
+
+const ID_BATCH = 200_000;
+
+export async function runDedupeOverlap(
+ deps: { config: Config; logger: Logger },
+ state: DedupeOverlapState,
+ opts: { fromMs: number; toMs: number; execute: boolean },
+): Promise {
+ const { config } = deps;
+ const logger = deps.logger.child({ component: 'DedupeOverlap' });
+ const lastDry = state.lastDryRun;
+
+ Object.assign(state, newDedupeOverlapState(), {
+ status: 'running', startedAt: Date.now(), phase: 'starting',
+ execute: opts.execute, fromMs: opts.fromMs, toMs: opts.toMs, lastDryRun: lastDry,
+ });
+
+ // Own connections, like the rebuild — never disturbs the orchestrator's.
+ const mongo = new MongoClient(config.source.uri);
+ const staging = new StagingManager(
+ {
+ url: config.target.url, database: config.target.db, table: config.target.table,
+ username: config.target.username, password: config.target.password,
+ queryTimeoutMs: config.target.queryTimeoutMs,
+ },
+ logger,
+ );
+ try {
+ await mongo.connect();
+ await staging.connect();
+ const db = mongo.db(config.source.db);
+
+ state.phase = 'discovering collections';
+ const collections = await discoverCollections(db, config.source.collectionPrefix, logger);
+ const from = new Date(opts.fromMs);
+ const to = new Date(opts.toMs);
+
+ for (const collection of collections) {
+ state.phase = `scanning ${collection}`;
+ const coll = db.collection(collection);
+ const row = { collection, mongoDocsInWindow: 0, chMatched: 0, deleted: 0 };
+
+ // Old-Mongo ids in the window = the mirror's re-ingested docs — the
+ // exact set whose migrated copies are duplicates.
+ let batch: string[] = [];
+ const flush = async (): Promise => {
+ if (batch.length === 0) return;
+ const matched = await staging.countMatchingIdsInWindow(batch, opts.fromMs, opts.toMs);
+ row.chMatched += matched;
+ if (opts.execute && matched > 0) {
+ await staging.deleteMatchingIdsInWindow(batch, opts.fromMs, opts.toMs);
+ row.deleted += matched;
+ }
+ batch = [];
+ };
+ const cursor = coll.find({ cd: { $gte: from, $lt: to } }, { projection: { _id: 1 } }).batchSize(10_000);
+ for await (const doc of cursor) {
+ row.mongoDocsInWindow++;
+ batch.push(String(doc._id));
+ if (batch.length >= ID_BATCH) await flush();
+ }
+ await flush();
+
+ if (row.mongoDocsInWindow > 0 || row.chMatched > 0) state.collections.push(row);
+ state.totals.mongoDocsInWindow += row.mongoDocsInWindow;
+ state.totals.chMatched += row.chMatched;
+ state.totals.deleted += row.deleted;
+ }
+
+ state.status = 'completed';
+ state.phase = 'done';
+ state.finishedAt = Date.now();
+ if (!opts.execute) {
+ state.lastDryRun = { fromMs: opts.fromMs, toMs: opts.toMs, chMatched: state.totals.chMatched, at: Date.now() };
+ }
+ logger.info(
+ { execute: opts.execute, ...state.totals, collections: state.collections.length },
+ opts.execute ? 'Tee-overlap duplicates deleted' : 'Tee-overlap dedupe dry run complete — nothing deleted',
+ );
+ } catch (err) {
+ state.status = 'failed';
+ state.error = (err as Error).message;
+ state.phase = 'failed';
+ state.finishedAt = Date.now();
+ logger.error({ err }, 'Tee-overlap dedupe failed');
+ } finally {
+ await mongo.close().catch(() => {});
+ await staging.close().catch(() => {});
+ }
+}
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
new file mode 100644
index 0000000..182847d
--- /dev/null
+++ b/src/runtime/final-check.ts
@@ -0,0 +1,199 @@
+/**
+ * Final check: the whole sign-off, interpreted.
+ *
+ * Operators kept having to understand four audit buckets, tee semantics and
+ * DLQ states to answer the only question they actually have at the end of a
+ * migration: "is it safe to decommission the old cluster?" This module runs
+ * every validation the tool has — chunk states, DLQ, the full source recount
+ * with cd-checksum fingerprints, sampled content comparison — applies the
+ * interpretation rules itself (including the tee cutover: windows past the
+ * cutover diverge BY DESIGN and must not read as data loss), and emits one
+ * verdict in plain sentences:
+ *
+ * PASS — safe to decommission.
+ * PASS WITH NOTES — safe, but read the amber lines first (waived DLQ,
+ * source retention drift, excluded post-cutover tail).
+ * FAIL — do not decommission; each red line names the action.
+ */
+
+import type { Logger } from 'pino';
+import type { Config } from '../config/schema.ts';
+import type { HashResolver } from '../transform/hash-resolver.ts';
+import type { LedgerStore } from '../state/ledger-store.ts';
+import type { DlqStore } from '../state/dlq-store.ts';
+import { rebuildLedger, newRebuildProgress, type RebuildProgress } from './ledger-rebuild.ts';
+
+export interface FinalCheckResult {
+ status: 'not_run' | 'running' | 'completed' | 'failed';
+ verdict: 'PASS' | 'PASS_WITH_NOTES' | 'FAIL' | null;
+ /** One sentence answering "can I decommission the old cluster?" */
+ headline: string | null;
+ /** Green lines — what was verified and held. */
+ passes: string[];
+ /** Amber lines — true, explained, and safe; read before sign-off. */
+ notes: string[];
+ /** Red lines — each names the problem AND the action. */
+ problems: string[];
+ /** The cutover used to scope the source recount (null = full range). */
+ cutoverMs: number | null;
+ phase: string;
+ /** Drill-down: the raw source-audit report backing the verdict. */
+ audit: RebuildProgress | null;
+ content: { sampled: number; matched: number; missing: number; different: number } | null;
+ error: string | null;
+ startedAt: number | null;
+ finishedAt: number | null;
+}
+
+export function newFinalCheckResult(): FinalCheckResult {
+ return {
+ status: 'not_run', verdict: null, headline: null,
+ passes: [], notes: [], problems: [],
+ cutoverMs: null, phase: '', audit: null, content: null,
+ error: null, startedAt: null, finishedAt: null,
+ };
+}
+
+const fmt = (n: number): string => n.toLocaleString('en-US');
+const iso = (ms: number): string => new Date(ms).toISOString().slice(0, 16).replace('T', ' ') + ' UTC';
+
+interface ContentAuditRunner {
+ contentAudit(samplesPerCollection?: number): Promise<{
+ sampled: number; matched: number; missing: number; different: number;
+ mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
+ }>;
+}
+
+export async function runFinalCheck(
+ deps: {
+ config: Config;
+ logger: Logger;
+ ledger: LedgerStore;
+ dlq: DlqStore;
+ hashResolver: HashResolver;
+ orchestrator: ContentAuditRunner;
+ },
+ out: FinalCheckResult,
+ opts: { cutoverMs: number | null; samples: number },
+): Promise {
+ const { config, ledger, dlq, hashResolver } = deps;
+ const logger = deps.logger.child({ component: 'FinalCheck' });
+ const runId = config.ledger.runId;
+
+ Object.assign(out, newFinalCheckResult(), { status: 'running', startedAt: Date.now(), phase: 'starting' });
+ try {
+ // ── Cutover: explicit param > stored bound > env bound > none ─────────
+ const stored = await ledger.getStoredBound(runId).catch(() => null);
+ const cutoverMs = opts.cutoverMs ?? stored ?? config.ledger.cdUpperBoundMs ?? null;
+ out.cutoverMs = cutoverMs;
+
+ // ── 1. Chunk ledger states ─────────────────────────────────────────────
+ out.phase = 'checking chunk states';
+ const counts = await ledger.statusCounts(runId);
+ const done = counts.done ?? 0;
+ const failed = counts.failed ?? 0;
+ const superseded = counts.superseded ?? 0;
+ const total = Object.values(counts).reduce((a, b) => a + b, 0);
+ const notDone = total - done - superseded - failed;
+ if (failed > 0) {
+ out.problems.push(`${fmt(failed)} chunk(s) FAILED — click "Retry failed chunks" (or POST /control/retry-failed), wait for them to finish, then run this check again.`);
+ }
+ if (notDone > 0) {
+ out.problems.push(`${fmt(notDone)} chunk(s) are not migrated yet — the run is not complete. Let it finish (or press Start/Resume), then run this check again.`);
+ }
+ if (failed === 0 && notDone === 0 && total > 0) {
+ out.passes.push(`All ${fmt(done)} chunks migrated and verified (per-chunk count + id checks passed before every attach).`);
+ }
+ if (total === 0) {
+ out.problems.push('The ledger holds no chunks — nothing has been migrated under this run id.');
+ }
+
+ // ── 2. DLQ ─────────────────────────────────────────────────────────────
+ out.phase = 'checking dead-letter queue';
+ const dlqCounts = await dlq.countByStatus(runId).catch(() => ({} as Record));
+ const dlqPending = dlqCounts.pending ?? 0;
+ const dlqWaived = dlqCounts.waived ?? 0;
+ if (dlqPending > 0) {
+ const top = await dlq.topErrors(runId, 3).catch(() => []);
+ const reasons = top.map((t) => `${t.error} ×${fmt(t.n)}`).join(', ');
+ out.notes.push(`${fmt(dlqPending)} skipped docs wait in the DLQ (${reasons}) — they are NOT in ClickHouse. Review a few in the DLQ panel, then Waive them (accepted as unmigratable) or Replay after a fix. Sign-off is complete once the DLQ shows 0 pending.`);
+ }
+ if (dlqWaived > 0) {
+ out.notes.push(`${fmt(dlqWaived)} docs were waived earlier — deliberately accepted as not migrated (their raw copies stay in the DLQ collection as the record).`);
+ }
+ if (dlqPending === 0 && dlqWaived === 0) out.passes.push('Dead-letter queue is empty — no document was skipped.');
+
+ // ── 3. Full source recount + cd-checksum fingerprint (the heavy one) ──
+ out.phase = 'recounting every window against the source';
+ const audit = newRebuildProgress();
+ out.audit = audit;
+ await rebuildLedger({ config, logger, ledger, dlq, hashResolver, progress: audit, checkOnly: true, upToMs: cutoverMs });
+ const windows = audit.summary.reduce((a, s) => a + s.chunks, 0);
+ if (audit.mismatchedWindows.length > 0) {
+ out.problems.push(`${fmt(audit.mismatchedWindows.length)} window(s) hold FEWER docs in ClickHouse than the source — data is missing from the target. Click "Retry failed chunks" after a rebuild, or escalate; do NOT decommission the old cluster.`);
+ }
+ if (audit.checksumMismatchWindows.length > 0) {
+ out.problems.push(`${fmt(audit.checksumMismatchWindows.length)} window(s) hold the right COUNT of the WRONG documents (checksum fingerprint differs) — escalate; do NOT decommission the old cluster.`);
+ }
+ if (audit.deletionDriftWindows.length > 0) {
+ out.notes.push(`${fmt(audit.deletionDriftWindows.length)} window(s) now hold MORE docs in ClickHouse than the source — the source shrank after migration (retention TTL / deletions). Expected on deployments with retention; the migrated copy is the complete one.`);
+ }
+ if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0) {
+ out.passes.push(`Recounted ${fmt(windows)} window(s) directly against the source: every count matches, every checksum fingerprint matches.`);
+ }
+ if (cutoverMs !== null) {
+ const excluded = audit.excludedBeyondCutover ?? 0;
+ out.notes.push(`Source docs after the cutover (${iso(cutoverMs)}) were excluded from the comparison${excluded > 0 ? ` (${fmt(excluded)} docs)` : ''} — after that moment the old side receives mirrored/live traffic that was never meant to be migrated, so divergence there is expected and is NOT data loss.`);
+ }
+
+ // ── 4. Sampled content comparison ──────────────────────────────────────
+ out.phase = 'comparing sampled documents field-by-field';
+ const content = await deps.orchestrator.contentAudit(opts.samples);
+ out.content = { sampled: content.sampled, matched: content.matched, missing: content.missing, different: content.different };
+ if (content.missing > 0 || content.different > 0) {
+ out.problems.push(`Content sampling found ${fmt(content.missing)} missing and ${fmt(content.different)} differing doc(s) out of ${fmt(content.sampled)} sampled — the migrated content does not match the source; escalate before decommissioning.`);
+ } else if (content.sampled > 0) {
+ out.passes.push(`Sampled ${fmt(content.sampled)} random docs field-by-field — all identical between source and ClickHouse.`);
+ }
+
+ // ── Verdict ────────────────────────────────────────────────────────────
+ out.verdict = out.problems.length > 0 ? 'FAIL' : out.notes.length > 0 ? 'PASS_WITH_NOTES' : 'PASS';
+ out.headline = out.verdict === 'FAIL'
+ ? `DO NOT decommission the old cluster yet — ${out.problems.length} problem(s) below need action first.`
+ : out.verdict === 'PASS_WITH_NOTES'
+ ? 'Safe to decommission the old cluster after reading the notes below.'
+ : 'ClickHouse verifiably holds everything the source holds — safe to decommission the old cluster.';
+ out.status = 'completed';
+ out.phase = 'done';
+ out.finishedAt = Date.now();
+ logger.info({ verdict: out.verdict, problems: out.problems.length, notes: out.notes.length }, 'Final check complete');
+ } catch (err) {
+ out.status = 'failed';
+ out.error = (err as Error).message;
+ out.phase = 'failed';
+ out.finishedAt = Date.now();
+ logger.error({ err }, 'Final check failed to complete');
+ }
+}
+
+/** Plain-text rendering for SSH-only operation (GET /final-check.txt). */
+export function renderFinalCheckText(fc: FinalCheckResult, runId: string): string {
+ const lines: string[] = [`FINAL CHECK - run ${runId}`];
+ if (fc.status === 'not_run') {
+ lines.push('Not run yet. Start it with: curl -X POST localhost:PORT/control/final-check');
+ } else if (fc.status === 'running') {
+ const a = fc.audit;
+ lines.push(`RUNNING - ${fc.phase}${a && a.collectionsTotal > 0 ? ` (${a.collectionsDone}/${a.collectionsTotal} collections)` : ''}`);
+ } else if (fc.status === 'failed') {
+ lines.push(`CHECK FAILED TO COMPLETE: ${fc.error} - fix and re-run; this is a tooling error, not a data verdict.`);
+ } else {
+ const badge = fc.verdict === 'PASS' ? 'PASS' : fc.verdict === 'PASS_WITH_NOTES' ? 'PASS WITH NOTES' : 'FAIL';
+ lines.push(`Verdict: ${badge} - ${fc.headline}`);
+ for (const p of fc.problems) lines.push(` [X] ${p}`);
+ for (const n of fc.notes) lines.push(` [!] ${n}`);
+ for (const g of fc.passes) lines.push(` [ok] ${g}`);
+ if (fc.cutoverMs !== null) lines.push(` cutover used: ${new Date(fc.cutoverMs).toISOString()}`);
+ if (fc.finishedAt) lines.push(` finished: ${new Date(fc.finishedAt).toISOString()}`);
+ }
+ return lines.join('\n') + '\n';
+}
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 2972ec4..a62f4a6 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -21,6 +21,8 @@ import { ClickHousePressure } from '../target/clickhouse-pressure.ts';
import { ChunkOrchestrator } from './chunk-orchestrator.ts';
import { wireExitOnComplete } from './exit-on-complete.ts';
import { rebuildLedger, newRebuildProgress, type RebuildProgress } from './ledger-rebuild.ts';
+import { runFinalCheck, newFinalCheckResult, renderFinalCheckText, type FinalCheckResult } from './final-check.ts';
+import { runDedupeOverlap, newDedupeOverlapState, type DedupeOverlapState } from './dedupe-overlap.ts';
export async function runLedgerEngine(config: Config, logger: Logger): Promise {
logger.info({ engine: 'ledger', runId: config.ledger.runId }, 'Starting ledger engine (no Redis)');
@@ -294,11 +296,15 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise {
const stats = orchestrator.getStats();
const runId = config.ledger.dryRun ? `${config.ledger.runId}-dry` : config.ledger.runId;
- const [cluster, runTimes] = await Promise.all([
+ const [cluster, clusterSlow, runTimes] = await Promise.all([
ledger.clusterRate(runId, 120).catch(() => null),
+ // 10-min window: with huge chunks completions land ~once a minute, so
+ // the 2-min window strobes and a freshly opened dashboard tab has no
+ // client-side history yet — this one is real the moment the page loads
+ ledger.clusterRate(runId, 600).catch(() => null),
ledger.getRunTimes(config.ledger.runId).catch(() => ({ startedAtMs: null, completedAtMs: null })),
]);
- return { ...stats, cluster, runTimes };
+ return { ...stats, cluster, clusterSlow, runTimes };
});
app.get('/report', async () => orchestrator.getReport());
app.post('/control/pause', async () => { orchestrator.pause(); return { status: orchestrator.getStatus() }; });
@@ -374,6 +380,48 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise ({ ...auditContentState, progress: orchestrator.contentAuditProgress }));
+
+ // ── Final check: the whole sign-off, interpreted (chunks + DLQ + source
+ // recount + checksums + content samples → one PASS/NOTES/FAIL verdict) ──
+ const finalCheckState: FinalCheckResult = newFinalCheckResult();
+ app.post<{ Body: { cutoverMs?: number; samples?: number } }>('/control/final-check', async (req) => {
+ if (finalCheckState.status === 'running') return { started: false, reason: 'final check already running' };
+ if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — run the final check after completion (or while paused)' };
+ const busyFc = await ledger.activeClaims(config.ledger.runId, config.worker.podId);
+ if (busyFc.length > 0) return { started: false, reason: `other pods are actively migrating (${busyFc.map((row) => row.pod).join(', ')}) — run the final check after completion` };
+ const cutoverMs = typeof req.body?.cutoverMs === 'number' && Number.isFinite(req.body.cutoverMs) ? req.body.cutoverMs : null;
+ const samples = Math.min(10_000, Math.max(50, req.body?.samples ?? 500));
+ void runFinalCheck({ config, logger, ledger, dlq, hashResolver, orchestrator }, finalCheckState, { cutoverMs, samples });
+ return { started: true, cutoverMs, samples };
+ });
+ app.get('/api/final-check', async () => finalCheckState);
+ app.get('/final-check.txt', async (_req, reply) => {
+ reply.type('text/plain; charset=utf-8').send(renderFinalCheckText(finalCheckState, config.ledger.runId));
+ });
+
+ // ── Tee-overlap dedupe: remove duplicates a missing cd bound created ────
+ // Dry-run by default; execute is licensed by a completed dry run over the
+ // SAME window in this process — measure first, delete second.
+ const dedupeState: DedupeOverlapState = newDedupeOverlapState();
+ app.post<{ Body: { fromMs?: number; toMs?: number; execute?: boolean } }>('/control/dedupe-overlap', async (req) => {
+ if (dedupeState.status === 'running') return { started: false, reason: 'dedupe already running' };
+ if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — dedupe only applies after completion' };
+ const fromMs = req.body?.fromMs;
+ const toMs = req.body?.toMs;
+ if (typeof fromMs !== 'number' || typeof toMs !== 'number' || !(fromMs < toMs)) {
+ return { started: false, reason: 'pass the overlap window as {fromMs, toMs} (epoch ms): fromMs = the tee flip / IP swap, toMs = migration completion' };
+ }
+ const execute = req.body?.execute === true;
+ if (execute) {
+ const dry = dedupeState.lastDryRun;
+ if (!dry || dry.fromMs !== fromMs || dry.toMs !== toMs) {
+ return { started: false, reason: 'execute refused: run a DRY RUN over this exact window first (same call without "execute") and review the matched counts' };
+ }
+ }
+ void runDedupeOverlap({ config, logger }, dedupeState, { fromMs, toMs, execute });
+ return { started: true, execute, fromMs, toMs };
+ });
+ app.get('/api/dedupe-overlap', async () => dedupeState);
app.get('/api/dryrun', async () => dryState);
app.get('/api/config', async () => ({
knobs: [
diff --git a/src/runtime/ledger-rebuild.ts b/src/runtime/ledger-rebuild.ts
index 80b8391..576cd87 100644
--- a/src/runtime/ledger-rebuild.ts
+++ b/src/runtime/ledger-rebuild.ts
@@ -62,6 +62,9 @@ export interface RebuildProgress {
deletionDriftWindows: Array<{ collection: string; lowerCd: string; upperCd: string; source: number; live: number }>;
/** counts MATCH but the cd-sum fingerprint differs: same number of docs, WRONG docs (identity swap). */
checksumMismatchWindows: Array<{ collection: string; lowerCd: string; upperCd: string; count: number; sumDeltaMs: number }>;
+ /** When a cutover clamp was applied: the clamp and how many source docs sit beyond it (out of scope). */
+ cutoverMs?: number | null;
+ excludedBeyondCutover?: number;
error: string | null;
startedAt: number | null;
finishedAt: number | null;
@@ -92,10 +95,20 @@ export async function rebuildLedger(opts: {
* the truth, not the tally.
*/
checkOnly?: boolean;
+ /**
+ * Tee/mirror cutover clamp: windows are only built for cd < upToMs and
+ * source docs at/after it are counted but excluded. Past the cutover the
+ * old side receives mirrored/live traffic that was never meant to be
+ * migrated, so comparing there reports divergence BY DESIGN — clamping is
+ * what turns the audit into a yes/no answer on tee deployments.
+ */
+ upToMs?: number | null;
}): Promise {
- const { config, ledger, dlq, hashResolver, progress, checkOnly = false } = opts;
+ const { config, ledger, dlq, hashResolver, progress, checkOnly = false, upToMs = null } = opts;
const logger = opts.logger.child({ component: 'LedgerRebuild' });
const runId = config.ledger.runId;
+ progress.cutoverMs = upToMs;
+ progress.excludedBeyondCutover = 0;
// Own connections — never disturbs the main orchestrator's bindings.
const mongo = new MongoClient(config.source.uri);
@@ -169,9 +182,16 @@ export async function rebuildLedger(opts: {
let bounds: Array<{ lowerCd: number; upperCd: number }> = [];
if (lowDoc && highDoc) {
const lowerCd = (lowDoc.cd as Date).getTime();
- const upperCd = (highDoc.cd as Date).getTime();
- const estimated = await coll.estimatedDocumentCount();
- bounds = computeChunkBounds(lowerCd, upperCd, estimated, config.ledger.chunkDocsTarget, config.ledger.maxChunkDays);
+ let upperCd = (highDoc.cd as Date).getTime();
+ if (upToMs !== null) {
+ progress.excludedBeyondCutover = (progress.excludedBeyondCutover ?? 0)
+ + await coll.countDocuments({ cd: { $gte: new Date(upToMs) } });
+ upperCd = Math.min(upperCd, upToMs - 1);
+ }
+ if (upperCd >= lowerCd) {
+ const estimated = await coll.estimatedDocumentCount();
+ bounds = computeChunkBounds(lowerCd, upperCd, estimated, config.ledger.chunkDocsTarget, config.ledger.maxChunkDays);
+ }
}
let idx = 0;
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index a80ffe5..4d0f85d 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -572,4 +572,40 @@ export class StagingManager {
}
return out;
}
+
+ /** Live rows in [fromMs, toMs) whose _id is one of the given ids. */
+ async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ let total = 0;
+ for (let i = 0; i < ids.length; i += 50_000) {
+ const page = ids.slice(i, i + 50_000);
+ const res = await this.ch().query({
+ query: `SELECT count() AS n FROM ${this.fq(this.config.table)}
+ WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
+ AND _id IN {ids:Array(String)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs },
+ format: 'JSONEachRow',
+ });
+ const rows = await res.json<{ n: string }>();
+ total += Number(rows[0]?.n ?? 0);
+ }
+ return total;
+ }
+
+ /**
+ * Lightweight-delete live rows in [fromMs, toMs) whose _id is one of the
+ * given ids. Tee-overlap cleanup: rows the migration copied from the old
+ * cluster that the mirror had already re-ingested natively. The cd window
+ * keeps each DELETE partition-prunable on multi-billion-row tables.
+ */
+ async deleteMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ for (let i = 0; i < ids.length; i += 50_000) {
+ const page = ids.slice(i, i + 50_000);
+ await this.ch().command({
+ query: `DELETE FROM ${this.fq(this.config.table)}
+ WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
+ AND _id IN {ids:Array(String)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs },
+ });
+ }
+ }
}
diff --git a/tests/integration/dedupe-overlap.test.ts b/tests/integration/dedupe-overlap.test.ts
new file mode 100644
index 0000000..0214636
--- /dev/null
+++ b/tests/integration/dedupe-overlap.test.ts
@@ -0,0 +1,123 @@
+/**
+ * Tee-overlap dedupe: a run without the cd bound migrated the mirror's
+ * re-ingested docs on top of natively ingested rows. Pinned here:
+ *
+ * - dry run counts the duplicates exactly and deletes NOTHING
+ * - execute deletes precisely the id-matched rows inside the window:
+ * native rows and pre-window (legitimately migrated) rows survive
+ */
+import { describe, it, expect, beforeAll, afterAll } from 'vitest';
+import pino from 'pino';
+import { createHash } from 'node:crypto';
+import { MongoClient } from 'mongodb';
+import { createClient, type ClickHouseClient } from '@clickhouse/client';
+
+import { runDedupeOverlap, newDedupeOverlapState } from '../../src/runtime/dedupe-overlap.ts';
+import { loadConfig } from '../../src/config/loader.ts';
+import type { Config } from '../../src/config/schema.ts';
+
+const MONGO_URI = 'mongodb://localhost:27017/?directConnection=true';
+const CH_URL = process.env.TEST_CLICKHOUSE_URL ?? 'http://localhost:8123';
+const CH_PASSWORD = process.env.TEST_CLICKHOUSE_PASSWORD ?? '';
+const DB = 'test_mig_dedupe';
+const logger = pino({ level: 'silent' });
+
+const APP = 'app_dd';
+const COLL = `drill_events${createHash('sha1').update('views' + APP).digest('hex')}`;
+
+const FLIP = Math.floor(Date.now() / 60_000) * 60_000 - 2 * 3_600_000; // tee flip 2h ago
+const DONE = FLIP + 3_600_000; // migration completed 1h later
+
+const chRow = (id: string, cdMs: number): Record => ({
+ a: APP, e: '[CLY]_custom', n: 'views', uid: 'u', did: 'd', _id: id,
+ ts: new Date(cdMs).toISOString().replace('T', ' ').replace('Z', ''),
+ cd: new Date(cdMs).toISOString().replace('T', ' ').replace('Z', ''),
+ up: {}, sg: {}, c: 1, s: 0, dur: 0,
+});
+
+describe('tee-overlap dedupe', () => {
+ let ch: ClickHouseClient;
+ let mc: MongoClient;
+ let config: Config;
+
+ const chCount = async (where = '1'): Promise => {
+ const res = await ch.query({ query: `SELECT count() AS n FROM ${DB}.drill_events WHERE ${where}`, format: 'JSONEachRow' });
+ return Number((await res.json<{ n: string }>())[0].n);
+ };
+
+ beforeAll(async () => {
+ mc = new MongoClient(MONGO_URI);
+ await mc.connect();
+ await mc.db(DB).dropDatabase();
+
+ ch = createClient({ url: CH_URL, password: CH_PASSWORD });
+ await ch.command({ query: `CREATE DATABASE IF NOT EXISTS ${DB}` });
+ await ch.command({ query: `DROP TABLE IF EXISTS ${DB}.drill_events` });
+ await ch.command({
+ query: `CREATE TABLE ${DB}.drill_events (
+ \`a\` LowCardinality(String), \`e\` LowCardinality(String), \`n\` String,
+ \`uid\` String, \`uid_canon\` Nullable(String), \`did\` String, \`lsid\` Nullable(String),
+ \`_id\` String, \`ts\` DateTime64(3), \`up\` JSON(max_dynamic_paths = 32),
+ \`custom\` Nullable(JSON(max_dynamic_paths = 0)), \`cmp\` Nullable(JSON(max_dynamic_paths = 0)),
+ \`sg\` JSON(max_dynamic_paths = 0), \`c\` UInt32, \`s\` Float64, \`dur\` Float64,
+ \`lu\` Nullable(DateTime64(3)), \`cd\` DateTime64(3) DEFAULT now64(3))
+ ENGINE = MergeTree PARTITION BY toYYYYMM(ts, 'UTC') ORDER BY (a, e, n, ts)`,
+ });
+
+ const mongoDocs: Record[] = [];
+ const chRows: Record[] = [];
+ // pre-flip history: migrated once, identical ids both sides — must survive
+ for (let i = 0; i < 200; i++) {
+ const cd = FLIP - 3_600_000 + i * 10_000;
+ mongoDocs.push({ _id: `hist_${i}`, uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ chRows.push(chRow(`hist_${i}`, cd));
+ }
+ // overlap window: each event exists in CH twice — natively (new id) and
+ // as the migrated copy of the mirror's re-ingested doc (old-Mongo id)
+ for (let i = 0; i < 150; i++) {
+ const cd = FLIP + i * 20_000;
+ mongoDocs.push({ _id: `mirror_${i}`, uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ chRows.push(chRow(`mirror_${i}`, cd)); // migrated duplicate
+ chRows.push(chRow(`native_${i}`, cd + 300)); // native original
+ }
+ // extra native rows with no mirror copy (mirror dropped them) — survive
+ for (let i = 0; i < 10; i++) chRows.push(chRow(`native_only_${i}`, FLIP + 500_000 + i * 1_000));
+ await mc.db(DB).collection(COLL).insertMany(mongoDocs as never[]);
+ await mc.db(DB).collection(COLL).createIndex({ cd: 1, _id: 1 });
+ await ch.insert({ table: `${DB}.drill_events`, values: chRows, format: 'JSONEachRow' });
+
+ Object.assign(process.env, {
+ SERVICE_NAME: 'dedupe-test',
+ MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`, MANIFEST_DB: DB,
+ CLICKHOUSE_URL: CH_URL, CLICKHOUSE_PASSWORD: CH_PASSWORD, CLICKHOUSE_DB: DB,
+ LEDGER_RUN_ID: 'dedupe-1', BACKPRESSURE_ENABLED: 'false', MULTI_POD_ENABLED: 'false',
+ });
+ config = loadConfig();
+ }, 120_000);
+
+ afterAll(async () => {
+ await ch.command({ query: `DROP DATABASE IF EXISTS ${DB}` }).catch(() => {});
+ await ch.close();
+ await mc.db(DB).dropDatabase().catch(() => {});
+ await mc.close();
+ });
+
+ it('dry run counts the duplicates exactly and deletes nothing', async () => {
+ const state = newDedupeOverlapState();
+ await runDedupeOverlap({ config, logger }, state, { fromMs: FLIP, toMs: DONE, execute: false });
+ expect(state.status).toBe('completed');
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150, chMatched: 150, deleted: 0 });
+ expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 });
+ expect(await chCount()).toBe(200 + 150 + 150 + 10);
+ });
+
+ it('execute deletes exactly the migrated copies; native and pre-flip rows survive', async () => {
+ const state = newDedupeOverlapState();
+ await runDedupeOverlap({ config, logger }, state, { fromMs: FLIP, toMs: DONE, execute: true });
+ expect(state.status).toBe('completed');
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150, chMatched: 150, deleted: 150 });
+ expect(await chCount("_id LIKE 'mirror_%'")).toBe(0);
+ expect(await chCount("_id LIKE 'native_%'")).toBe(160);
+ expect(await chCount("_id LIKE 'hist_%'")).toBe(200);
+ });
+});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
new file mode 100644
index 0000000..ff710f1
--- /dev/null
+++ b/tests/integration/final-check.test.ts
@@ -0,0 +1,213 @@
+/**
+ * Final check — the interpreted sign-off. Pinned here:
+ *
+ * - a clean, cutover-clamped tee run passes (PASS WITH NOTES: the excluded
+ * post-cutover tail is a note, never a problem)
+ * - the SAME data audited WITHOUT the clamp fails — proving the clamp is
+ * what turns a tee audit into a yes/no answer
+ * - pending DLQ docs surface as an action note and their windows are not
+ * double-flagged (unresolved accounting)
+ * - missing target rows / failed chunks / content mismatches each produce a
+ * FAIL with an action sentence
+ */
+import { describe, it, expect, beforeAll, afterAll } from 'vitest';
+import pino from 'pino';
+import { createHash } from 'node:crypto';
+import { MongoClient } from 'mongodb';
+import { createClient, type ClickHouseClient } from '@clickhouse/client';
+
+import { runFinalCheck, newFinalCheckResult } from '../../src/runtime/final-check.ts';
+import { LedgerStore, type ChunkDoc } from '../../src/state/ledger-store.ts';
+import { DlqStore } from '../../src/state/dlq-store.ts';
+import { HashResolver } from '../../src/transform/hash-resolver.ts';
+import { loadConfig } from '../../src/config/loader.ts';
+import type { Config } from '../../src/config/schema.ts';
+
+const MONGO_URI = 'mongodb://localhost:27017/?directConnection=true';
+const CH_URL = process.env.TEST_CLICKHOUSE_URL ?? 'http://localhost:8123';
+const CH_PASSWORD = process.env.TEST_CLICKHOUSE_PASSWORD ?? '';
+const DB = 'test_mig_finalcheck';
+const RUN = 'fc-1';
+const logger = pino({ level: 'silent' });
+
+const APP = 'app_fc';
+const COLL = `drill_events${createHash('sha1').update('views' + APP).digest('hex')}`;
+const MIN = 60_000;
+
+// Timeline: 600 docs over 2 hours, then the cutover, then a teed tail —
+// 100 mirrored copies on the Mongo side, 80 native rows on the CH side
+// (different identities, different counts: exactly what a tee looks like).
+const CUTOVER = Math.floor((Date.now() - 3_600_000) / MIN) * MIN;
+const START = CUTOVER - 120 * MIN;
+
+const chRow = (id: string, cdMs: number): Record => ({
+ a: APP, e: '[CLY]_custom', n: 'views', uid: 'u', did: 'd', _id: id,
+ ts: new Date(cdMs).toISOString().replace('T', ' ').replace('Z', ''),
+ cd: new Date(cdMs).toISOString().replace('T', ' ').replace('Z', ''),
+ up: {}, sg: {}, c: 1, s: 0, dur: 0,
+});
+
+const contentClean = {
+ contentAudit: async (samples = 500) => ({ sampled: samples, matched: samples, missing: 0, different: 0, mismatches: [] }),
+};
+
+describe('final check: the interpreted sign-off', () => {
+ let ch: ClickHouseClient;
+ let mc: MongoClient;
+ let ledger: LedgerStore;
+ let dlq: DlqStore;
+ let hashResolver: HashResolver;
+ let config: Config;
+
+ const check = async (opts?: { cutoverMs?: number | null; orchestrator?: typeof contentClean }) => {
+ const out = newFinalCheckResult();
+ await runFinalCheck(
+ { config, logger, ledger, dlq, hashResolver, orchestrator: opts?.orchestrator ?? contentClean },
+ out,
+ { cutoverMs: opts?.cutoverMs ?? null, samples: 100 },
+ );
+ expect(out.status).toBe('completed');
+ return out;
+ };
+
+ beforeAll(async () => {
+ mc = new MongoClient(MONGO_URI);
+ await mc.connect();
+ await mc.db(DB).dropDatabase();
+ await mc.db(`${DB}_countly`).dropDatabase();
+ await mc.db(`${DB}_countly`).collection('apps').insertOne({ _id: APP } as never);
+ await mc.db(`${DB}_countly`).collection('events').insertOne({ _id: APP, list: ['views'] } as never);
+
+ ch = createClient({ url: CH_URL, password: CH_PASSWORD });
+ await ch.command({ query: `CREATE DATABASE IF NOT EXISTS ${DB}` });
+ await ch.command({ query: `DROP TABLE IF EXISTS ${DB}.drill_events` });
+ await ch.command({
+ query: `CREATE TABLE ${DB}.drill_events (
+ \`a\` LowCardinality(String), \`e\` LowCardinality(String), \`n\` String,
+ \`uid\` String, \`uid_canon\` Nullable(String), \`did\` String, \`lsid\` Nullable(String),
+ \`_id\` String, \`ts\` DateTime64(3), \`up\` JSON(max_dynamic_paths = 32),
+ \`custom\` Nullable(JSON(max_dynamic_paths = 0)), \`cmp\` Nullable(JSON(max_dynamic_paths = 0)),
+ \`sg\` JSON(max_dynamic_paths = 0), \`c\` UInt32, \`s\` Float64, \`dur\` Float64,
+ \`lu\` Nullable(DateTime64(3)), \`cd\` DateTime64(3) DEFAULT now64(3))
+ ENGINE = MergeTree PARTITION BY toYYYYMM(ts, 'UTC') ORDER BY (a, e, n, ts)`,
+ });
+
+ const mongoDocs: Record[] = [];
+ const chRows: Record[] = [];
+ // migrated body: identical (_id, cd) on both sides
+ for (let i = 0; i < 600; i++) {
+ const cd = START + i * 12_000;
+ mongoDocs.push({ _id: `m_${i}`, uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ chRows.push(chRow(`m_${i}`, cd));
+ }
+ // teed tail: mirrored copies in Mongo, different native rows in CH
+ for (let i = 0; i < 100; i++) {
+ const cd = CUTOVER + i * 500;
+ mongoDocs.push({ _id: `mirror_${i}`, uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ }
+ for (let i = 0; i < 80; i++) chRows.push(chRow(`native_${i}`, CUTOVER + 200 + i * 500));
+ await mc.db(DB).collection(COLL).insertMany(mongoDocs as never[]);
+ await mc.db(DB).collection(COLL).createIndex({ cd: 1, _id: 1 });
+ await ch.insert({ table: `${DB}.drill_events`, values: chRows, format: 'JSONEachRow' });
+
+ Object.assign(process.env, {
+ SERVICE_NAME: 'finalcheck-test',
+ MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`, MANIFEST_DB: DB,
+ CLICKHOUSE_URL: CH_URL, CLICKHOUSE_PASSWORD: CH_PASSWORD, CLICKHOUSE_DB: DB,
+ LEDGER_RUN_ID: RUN, BACKPRESSURE_ENABLED: 'false', MULTI_POD_ENABLED: 'false',
+ });
+ delete process.env.LEDGER_CD_UPPER_BOUND;
+ config = loadConfig();
+
+ ledger = new LedgerStore(MONGO_URI, DB, logger);
+ dlq = new DlqStore(MONGO_URI, DB, logger);
+ hashResolver = new HashResolver({ uri: MONGO_URI, countlyDb: `${DB}_countly` }, logger);
+ await ledger.connect();
+ await dlq.connect();
+ await hashResolver.build();
+
+ // the run's own record: one done chunk covering the migrated body
+ const chunk: ChunkDoc = {
+ _id: `${RUN}:${COLL}:0`, run_id: RUN, collection: COLL,
+ scope_a: APP, scope_e: '[CLY]_custom', scope_n: 'views',
+ idx: 0, lower_cd: START, upper_cd: CUTOVER, status: 'done',
+ pod_id: null, lease_until: null, staging_table: null,
+ docs_read: 600, docs_skipped: 0, rows_expected: 600,
+ partitions: [], attached: [], attach_method: null, attempts: 1,
+ last_error: null, transform_version: config.transform.version, updated_at: new Date(),
+ };
+ await ledger.replaceAllForRun(RUN, [chunk]);
+ }, 120_000);
+
+ afterAll(async () => {
+ await ledger?.close().catch(() => {});
+ await dlq?.close().catch(() => {});
+ await hashResolver?.close?.().catch?.(() => {});
+ await ch.command({ query: `DROP DATABASE IF EXISTS ${DB}` }).catch(() => {});
+ await ch.close();
+ await mc.db(DB).dropDatabase().catch(() => {});
+ await mc.db(`${DB}_countly`).dropDatabase().catch(() => {});
+ await mc.close();
+ });
+
+ it('clean tee run with cutover clamp → PASS WITH NOTES (tail excluded is a note, not a problem)', async () => {
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.problems).toEqual([]);
+ expect(out.verdict).toBe('PASS_WITH_NOTES');
+ expect(out.notes.join(' ')).toContain('excluded');
+ expect(out.audit?.excludedBeyondCutover).toBe(100);
+ expect(out.audit?.mismatchedWindows).toEqual([]);
+ expect(out.headline).toContain('Safe to decommission');
+ });
+
+ it('same data WITHOUT the clamp → FAIL: post-cutover divergence reads as data problems', async () => {
+ const out = await check({ cutoverMs: null });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.length).toBeGreaterThan(0);
+ });
+
+ it('pending DLQ docs → action note, and their window is NOT flagged (unresolved accounting)', async () => {
+ // one doc the run skipped: present in Mongo, absent in CH, recorded in DLQ
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_10'` });
+ await dlq.add([{
+ run_id: RUN, collection: COLL, chunk_id: `${RUN}:${COLL}:0`, source_id: 'm_10',
+ raw_doc: { _id: 'm_10' }, reason: 'skipped', error: 'skip:missing_a',
+ transform_version: config.transform.version, cd_ms: START + 10 * 12_000,
+ }]);
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('PASS_WITH_NOTES');
+ expect(out.problems).toEqual([]);
+ expect(out.notes.join(' ')).toContain('DLQ');
+ // waive → the note softens to the waived form
+ await dlq.waive(RUN);
+ const out2 = await check({ cutoverMs: CUTOVER });
+ expect(out2.verdict).toBe('PASS_WITH_NOTES');
+ expect(out2.notes.join(' ')).toContain('waived');
+ });
+
+ it('missing target rows → FAIL with a do-not-decommission problem', async () => {
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id IN ('m_20','m_21','m_22')` });
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('FEWER');
+ // restore for the next cases
+ await ch.insert({
+ table: `${DB}.drill_events`, format: 'JSONEachRow',
+ values: [20, 21, 22].map((i) => chRow(`m_${i}`, START + i * 12_000)),
+ });
+ });
+
+ it('content mismatch and failed chunks each FAIL with their own action line', async () => {
+ const badContent = {
+ contentAudit: async (samples = 500) => ({ sampled: samples, matched: samples - 2, missing: 1, different: 1, mismatches: [] }),
+ };
+ const out = await check({ cutoverMs: CUTOVER, orchestrator: badContent });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('Content sampling');
+
+ await mc.db(DB).collection('mig_ranges').updateOne({ _id: `${RUN}:${COLL}:0` } as never, { $set: { status: 'failed' } });
+ const out2 = await check({ cutoverMs: CUTOVER });
+ expect(out2.problems.join(' ')).toContain('Retry failed chunks');
+ await mc.db(DB).collection('mig_ranges').updateOne({ _id: `${RUN}:${COLL}:0` } as never, { $set: { status: 'done' } });
+ });
+});
From df7d4a01784c4ea48686ac6a83c02b85c9b890c3 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 11:24:09 +0300
Subject: [PATCH 02/64] feat(ledger): one-call set-boundary + cluster-truth
dashboard tiles
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
- POST /control/set-boundary: the whole boundary flow in one endpoint.
{} detects and auto-applies when the seam is an exact ingestion-pause gap;
an anchor (quantified ambiguity) is never taken without {acceptAnchor:true};
{boundMs} applies a known timestamp directly. The apply receipt lands in
GET /api/boundary under .applied. apply-bound/detect-boundary still work
and now share one apply code path.
- SKIPPED tile reads the ledger's cluster-wide docs_skipped sum instead of
this pod's in-memory counter (field: a 3-pod run showed 100,623 while the
DLQ held 314,125).
- 'N pods' label uses the wider of the 2-min and 10-min activity windows —
the 2-min window strobes to 1 pod on huge-chunk runs.
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 19 +++++++
src/http/ledger-viz-route.ts | 12 +++--
src/runtime/boundary-detector.ts | 22 ++++++++
src/runtime/ledger-engine.ts | 62 ++++++++++++++++++-----
src/state/ledger-store.ts | 10 ++--
tests/integration/boundary-detect.test.ts | 25 ++++++++-
tests/integration/final-check.test.ts | 5 ++
7 files changed, 133 insertions(+), 22 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 8cef6d9..87915c6 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -193,6 +193,25 @@ Caveats:
- Retention TTL keeps deleting on the old side throughout — the source
audit reports that as deletion drift, not as a defect.
+### One-call boundary setting (SSH / API)
+
+The whole detect-and-apply flow is a single endpoint:
+
+```bash
+# detect, and apply automatically when the seam is an exact ingestion-pause gap
+curl -s -X POST localhost:PORT/control/set-boundary -H 'content-type: application/json' -d '{}'
+# read the outcome — the apply receipt lands in .applied
+curl -s localhost:PORT/api/boundary
+# no exact gap? review the report, then accept the anchor explicitly…
+curl -s -X POST localhost:PORT/control/set-boundary -H 'content-type: application/json' -d '{"acceptAnchor": true}'
+# …or set the bound to a known timestamp directly (applies immediately)
+curl -s -X POST localhost:PORT/control/set-boundary -H 'content-type: application/json' -d '{"boundMs": 1789966140000}'
+```
+
+An exact gap applies unattended; an anchor (quantified ambiguity) is never
+auto-applied without `acceptAnchor`. The dashboard flow and the separate
+`/control/detect-boundary` + `/control/apply-bound` endpoints keep working.
+
### Bound is opt-in — pick the mode deliberately
| Situation | LEDGER_CD_UPPER_BOUND | Behavior |
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index 7974ce9..a1d1431 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -830,9 +830,10 @@ async function tick() {
var liveRate = clientReady ? Math.max(0, (last.d - windowStart.d) / rspan)
: slowRate !== null ? slowRate
: null;
- var multiPod = stats.cluster && stats.cluster.pods > 1;
+ var podsSeen = Math.max(stats.cluster ? stats.cluster.pods : 0, stats.clusterSlow ? stats.clusterSlow.pods : 0);
+ var multiPod = podsSeen > 1;
var effRate = liveRate !== null ? liveRate
- : multiPod && stats.status === 'running' ? stats.cluster.docsPerSecond
+ : multiPod && stats.status === 'running' ? (stats.cluster ? stats.cluster.docsPerSecond : 0)
: stats.docsPerSecond;
if (stats.status === 'completed') {
// whole-RUN average from ledger docs + run timeline; the pod's own
@@ -844,13 +845,16 @@ async function tick() {
var runAvg = runSec > 0 && sum.docsDone > 0 ? sum.docsDone / runSec : stats.docsPerSecond;
dpsEl.textContent = runAvg >= 1 ? fmt(Math.round(runAvg)) + ' avg' : '\u2013';
} else if (liveRate !== null) {
- dpsEl.textContent = fmt(Math.round(liveRate)) + (multiPod ? ' \u00b7 ' + stats.cluster.pods + ' pods' : '');
+ dpsEl.textContent = fmt(Math.round(liveRate)) + (multiPod ? ' \u00b7 ' + podsSeen + ' pods' : '');
} else if (stats.status === 'running') {
dpsEl.textContent = 'measuring\u2026';
} else {
dpsEl.textContent = '\u2013';
}
- document.getElementById('s-skipped').textContent = fmt(stats.totalDocsSkipped);
+ // ledger truth — each pod's in-memory counter only knows its own share
+ // (field: a 3-pod run showed 100,623 while the DLQ held 314,125)
+ document.getElementById('s-skipped').textContent =
+ fmt(Math.max(sum.docsSkipped || 0, stats.totalDocsSkipped || 0));
// ledger truth, not this pod's counter — in multi-pod each pod only
// counts its own failures, so the card under-reported cluster-wide
document.getElementById('s-failed').textContent = fmt((sum.byStatus || {}).failed || 0);
diff --git a/src/runtime/boundary-detector.ts b/src/runtime/boundary-detector.ts
index 910c751..06816d2 100644
--- a/src/runtime/boundary-detector.ts
+++ b/src/runtime/boundary-detector.ts
@@ -45,6 +45,28 @@ export interface BoundaryProgress {
report: BoundaryReport | null;
}
+/**
+ * One-call boundary setting: decide whether a detection is safe to apply
+ * unattended. An exact ingestion-pause GAP is; an ANCHOR carries quantified
+ * ambiguity and needs a human (or an explicit acceptAnchor).
+ */
+export function decideAutoApply(
+ report: BoundaryReport | null | undefined,
+ acceptAnchor: boolean,
+): { apply: boolean; boundMs?: number; reason?: string } {
+ const d = report?.detection;
+ if (!d || d.status !== 'ok' || !d.suggestedBoundMs) {
+ return { apply: false, reason: `no boundary detected${d?.reason ? ` — ${d.reason}` : d?.status ? ` (${d.status})` : ''}` };
+ }
+ if (d.method !== 'gap' && !acceptAnchor) {
+ return {
+ apply: false,
+ reason: `detected an ANCHOR, not an exact gap — ${d.ambiguousMongoDocs ?? '?'} old-side docs sit inside the ambiguity band. Review GET /api/boundary, then re-call with {"acceptAnchor": true} to take it, or pass an explicit {"boundMs": ...}.`,
+ };
+ }
+ return { apply: true, boundMs: d.suggestedBoundMs };
+}
+
export interface BoundaryReport {
detection: {
status: 'ok' | 'refused' | 'no_data';
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index a62f4a6..c7c6d41 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -455,36 +455,41 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise('/control/apply-bound', async (req, reply) => {
- const boundMs = Number(req.body?.boundMs);
- if (!Number.isFinite(boundMs) || boundMs <= 0) {
- reply.code(400);
- return { applied: false, reason: 'boundMs (epoch ms) required' };
- }
+ let boundaryApplied: Record | null = null;
+ const applyBoundNow = async (boundMs: number, source: string): Promise> => {
+ if (!Number.isFinite(boundMs) || boundMs <= 0) return { applied: false, reason: 'boundMs (epoch ms) required' };
if (config.ledger.dryRun) return { applied: false, reason: 'dry run — apply on the real run' };
if (envBoundAtBoot !== null) {
return { applied: false, reason: `bound already pinned via LEDGER_CD_UPPER_BOUND=${envBoundAtBoot} — change it in the deployment config, not here` };
}
- if (boundMs >= Date.now() - 60_000) {
- return { applied: false, reason: 'bound must be safely in the past (>60s ago)' };
- }
+ if (boundMs >= Date.now() - 60_000) return { applied: false, reason: 'bound must be safely in the past (>60s ago)' };
try {
const pruned = await ledger.pruneBeyondBound(config.ledger.runId, boundMs);
- await ledger.setStoredBound(config.ledger.runId, boundMs, 'dashboard');
- logger.warn({ boundMs, iso: new Date(boundMs).toISOString(), ...pruned }, 'Run bound applied from dashboard — pods adopt it on their next map pass');
+ await ledger.setStoredBound(config.ledger.runId, boundMs, source);
+ logger.warn({ boundMs, iso: new Date(boundMs).toISOString(), source, ...pruned }, 'Run bound applied — pods adopt it on their next map pass');
return { applied: true, boundMs, iso: new Date(boundMs).toISOString(), ...pruned };
} catch (err) {
- reply.code(409);
return { applied: false, reason: (err as Error).message };
}
+ };
+ app.post<{ Body: { boundMs?: number } }>('/control/apply-bound', async (req, reply) => {
+ const boundMs = Number(req.body?.boundMs);
+ if (!Number.isFinite(boundMs) || boundMs <= 0) {
+ reply.code(400);
+ return { applied: false, reason: 'boundMs (epoch ms) required' };
+ }
+ const res = await applyBoundNow(boundMs, 'dashboard');
+ if (!res.applied) reply.code(409);
+ return res;
});
// Tee-boundary detection + sync parity (background task — the Mongo
// scan across thousands of collections is minutes of work).
- const { detectBoundary, newBoundaryProgress } = await import('./boundary-detector.ts');
+ const { detectBoundary, newBoundaryProgress, decideAutoApply } = await import('./boundary-detector.ts');
const boundaryState = newBoundaryProgress();
app.post<{ Body: { bandMinutes?: number } }>('/control/detect-boundary', async (req) => {
if (boundaryState.status === 'running') return { started: false, reason: 'detection already running' };
+ boundaryApplied = null;
Object.assign(boundaryState, newBoundaryProgress(), { status: 'running', startedAt: Date.now() });
void detectBoundary({
config, logger, db: mongoReader.getDatabase(), staging, ledger,
@@ -494,7 +499,36 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise { boundaryState.status = 'failed'; boundaryState.error = (e as Error).message; boundaryState.finishedAt = Date.now(); });
return { started: true };
});
- app.get('/api/boundary', async () => boundaryState);
+ app.get('/api/boundary', async () => ({ ...boundaryState, applied: boundaryApplied }));
+
+ // ── ONE endpoint for the whole boundary flow ────────────────────────────
+ // {} → detect, and auto-apply when the seam is an exact ingestion-pause
+ // gap; {"acceptAnchor":true} → also take an anchor suggestion; {"boundMs"}
+ // → apply that value directly. The result (incl. the apply receipt) lands
+ // in GET /api/boundary under .applied.
+ app.post<{ Body: { boundMs?: number; acceptAnchor?: boolean; bandMinutes?: number } }>('/control/set-boundary', async (req) => {
+ if (typeof req.body?.boundMs === 'number') {
+ boundaryApplied = await applyBoundNow(req.body.boundMs, 'set-boundary explicit');
+ return boundaryApplied;
+ }
+ if (boundaryState.status === 'running') return { started: false, reason: 'detection already running — poll GET /api/boundary' };
+ const acceptAnchor = req.body?.acceptAnchor === true;
+ boundaryApplied = null;
+ Object.assign(boundaryState, newBoundaryProgress(), { status: 'running', startedAt: Date.now() });
+ void detectBoundary({
+ config, logger, db: mongoReader.getDatabase(), staging, ledger,
+ progress: boundaryState, bandMinutes: req.body?.bandMinutes,
+ })
+ .then(async (report) => {
+ boundaryState.report = report; boundaryState.status = 'completed'; boundaryState.finishedAt = Date.now();
+ const decision = decideAutoApply(report, acceptAnchor);
+ boundaryApplied = decision.apply
+ ? await applyBoundNow(decision.boundMs as number, acceptAnchor ? 'set-boundary anchor accepted' : 'set-boundary exact gap')
+ : { applied: false, reason: decision.reason };
+ })
+ .catch((e) => { boundaryState.status = 'failed'; boundaryState.error = (e as Error).message; boundaryState.finishedAt = Date.now(); });
+ return { started: true, mode: acceptAnchor ? 'detect + apply (anchor accepted)' : 'detect + apply only if the seam is exact', result: 'poll GET /api/boundary — the receipt lands in .applied' };
+ });
app.get('/api/pods', async () => ({
pods: await ledger.podActivity(config.ledger.dryRun ? `${config.ledger.runId}-dry` : config.ledger.runId),
diff --git a/src/state/ledger-store.ts b/src/state/ledger-store.ts
index ce766a7..a4849b3 100644
--- a/src/state/ledger-store.ts
+++ b/src/state/ledger-store.ts
@@ -90,10 +90,12 @@ export class LedgerStore {
total: number;
byStatus: Record;
docsDone: number;
+ /** Cluster truth — each pod's in-memory skip counter only knows its own share. */
+ docsSkipped: number;
perCollection: Array<{ collection: string; byStatus: Record; docsDone: number; doneDocsRead: number; nonDoneRowsExpected: number }>;
}> {
const rows = await this.c().aggregate<{
- _id: { c: string; s: string }; n: number; docsDone: number; docsRead: number; nonDoneExpected: number;
+ _id: { c: string; s: string }; n: number; docsDone: number; docsRead: number; nonDoneExpected: number; docsSkipped: number;
}>([
{ $match: { run_id: runId } },
{ $group: {
@@ -102,11 +104,12 @@ export class LedgerStore {
docsDone: { $sum: { $cond: [{ $eq: ['$status', 'done'] }, '$rows_expected', 0] } },
docsRead: { $sum: { $cond: [{ $eq: ['$status', 'done'] }, '$docs_read', 0] } },
nonDoneExpected: { $sum: { $cond: [{ $in: ['$status', ['pending', 'in_progress', 'written', 'attaching', 'failed']] }, '$rows_expected', 0] } },
+ docsSkipped: { $sum: '$docs_skipped' },
} },
]).toArray();
const perColl = new Map; docsDone: number; doneDocsRead: number; nonDoneRowsExpected: number }>();
const byStatus: Record = {};
- let total = 0, docsDone = 0;
+ let total = 0, docsDone = 0, docsSkipped = 0;
for (const r of rows) {
const e = perColl.get(r._id.c) ?? { collection: r._id.c, byStatus: {}, docsDone: 0, doneDocsRead: 0, nonDoneRowsExpected: 0 };
e.byStatus[r._id.s] = (e.byStatus[r._id.s] ?? 0) + r.n;
@@ -117,8 +120,9 @@ export class LedgerStore {
byStatus[r._id.s] = (byStatus[r._id.s] ?? 0) + r.n;
total += r.n;
docsDone += r.docsDone;
+ docsSkipped += r.docsSkipped;
}
- return { total, byStatus, docsDone, perCollection: [...perColl.values()].sort((a, b) => a.collection.localeCompare(b.collection)) };
+ return { total, byStatus, docsDone, docsSkipped, perCollection: [...perColl.values()].sort((a, b) => a.collection.localeCompare(b.collection)) };
}
/** Non-terminal + failed chunk details, capped — the interesting ones on huge runs. */
diff --git a/tests/integration/boundary-detect.test.ts b/tests/integration/boundary-detect.test.ts
index d2c049a..129b2ff 100644
--- a/tests/integration/boundary-detect.test.ts
+++ b/tests/integration/boundary-detect.test.ts
@@ -17,7 +17,7 @@ import { createHash } from 'node:crypto';
import { MongoClient } from 'mongodb';
import { createClient, type ClickHouseClient } from '@clickhouse/client';
-import { detectBoundary, newBoundaryProgress } from '../../src/runtime/boundary-detector.ts';
+import { detectBoundary, newBoundaryProgress, decideAutoApply } from '../../src/runtime/boundary-detector.ts';
import { LedgerStore } from '../../src/state/ledger-store.ts';
import { StagingManager } from '../../src/target/staging-manager.ts';
import { loadConfig } from '../../src/config/loader.ts';
@@ -281,3 +281,26 @@ describe('tee-boundary detection + sync parity', () => {
await mc.db(DB).collection('mig_ranges').deleteMany({ run_id: RUN } as never);
}, 60_000);
});
+
+describe('set-boundary auto-apply decision', () => {
+ const report = (detection: Record) => ({ detection, sync: { status: 'ok' } }) as never;
+
+ it('an exact gap applies unattended', () => {
+ expect(decideAutoApply(report({ status: 'ok', method: 'gap', suggestedBoundMs: 123 }), false))
+ .toEqual({ apply: true, boundMs: 123 });
+ });
+
+ it('an anchor needs the explicit acceptAnchor', () => {
+ const d = decideAutoApply(report({ status: 'ok', method: 'anchor', suggestedBoundMs: 123, ambiguousMongoDocs: 42 }), false);
+ expect(d.apply).toBe(false);
+ expect(d.reason).toContain('acceptAnchor');
+ expect(d.reason).toContain('42');
+ expect(decideAutoApply(report({ status: 'ok', method: 'anchor', suggestedBoundMs: 123 }), true))
+ .toEqual({ apply: true, boundMs: 123 });
+ });
+
+ it('refused or empty detections never apply', () => {
+ expect(decideAutoApply(report({ status: 'refused', reason: 'run already mapped' }), true).apply).toBe(false);
+ expect(decideAutoApply(null, true).apply).toBe(false);
+ });
+});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index ff710f1..a0c42b9 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -197,6 +197,11 @@ describe('final check: the interpreted sign-off', () => {
});
});
+ it('summarize reports cluster-truth docsSkipped from the ledger', async () => {
+ await mc.db(DB).collection('mig_ranges').updateOne({ _id: `${RUN}:${COLL}:0` } as never, { $set: { docs_skipped: 7 } });
+ expect((await ledger.summarize(RUN)).docsSkipped).toBe(7);
+ });
+
it('content mismatch and failed chunks each FAIL with their own action line', async () => {
const badContent = {
contentAudit: async (samples = 500) => ({ sampled: samples, matched: samples - 2, missing: 1, different: 1, mismatches: [] }),
From d29620b6290bcfdc42f12f394da16b5b58fa4ced Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:05:07 +0300
Subject: [PATCH 03/64] =?UTF-8?q?feat(ledger):=20startup=20guard=20?=
=?UTF-8?q?=E2=80=94=20refuse=20to=20run=20unbounded=20against=20a=20live?=
=?UTF-8?q?=20target?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A FRESH run whose target ClickHouse is already receiving live data, with no
cd bound set, is either a duplication bug about to happen (mirror active —
the mirrored-cutover case) or a deliberate choice (cutover-first / in-place). The
tool cannot tell those apart from data alone, so it now holds before mapping
(pauseReason boundary-unset) and asks once:
- apply a bound (set-boundary / boundary card / LEDGER_CD_UPPER_BOUND), or
- declare no-mirror: POST /control/allow-unbounded (cluster-wide, banner
button with armed confirm) or LEDGER_UNBOUNDED_OK=1.
A plain Resume is deliberately ignored while the question is open. Resumed
runs and runs whose target holds no recent data never trip the guard.
Orchestrator-driving tests declare the ack in their harness env — they ARE
the live-parallel scenario the guard asks about.
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 18 ++
src/config/loader.ts | 1 +
src/config/schema.ts | 2 +
src/http/ledger-viz-route.ts | 48 ++++-
src/runtime/chunk-orchestrator.ts | 52 ++++-
src/runtime/ledger-engine.ts | 8 +
src/state/ledger-store.ts | 15 ++
src/target/staging-manager.ts | 11 +
.../integration/cross-collection-pods.test.ts | 1 +
tests/integration/ledger-engine.test.ts | 1 +
tests/integration/live-parallel.test.ts | 2 +
.../multi-collection-and-rebuild.test.ts | 3 +-
tests/integration/outage-chaos.test.ts | 1 +
tests/integration/pod-chaos.test.ts | 1 +
tests/integration/startup-guard.test.ts | 192 ++++++++++++++++++
15 files changed, 345 insertions(+), 11 deletions(-)
create mode 100644 tests/integration/startup-guard.test.ts
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 87915c6..bb6b135 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -212,6 +212,24 @@ An exact gap applies unattended; an anchor (quantified ambiguity) is never
auto-applied without `acceptAnchor`. The dashboard flow and the separate
`/control/detect-boundary` + `/control/apply-bound` endpoints keep working.
+### The startup guard — the bound mistake, made impossible to miss
+
+A FRESH run that finds its target ClickHouse already receiving live data,
+with no cd bound set, **holds before mapping** (`pauseReason:
+boundary-unset`). That is exactly the setup where an unset bound either
+duplicates the overlap window (mirror active) or is a deliberate choice
+(cutover-first / in-place, where new data must still be migrated). The tool
+cannot tell those apart from data alone, so it asks — once:
+
+- mirror active → apply the bound (`POST /control/set-boundary`, the
+ dashboard card, or `LEDGER_CD_UPPER_BOUND`); the run releases itself, or
+- nothing mirrors traffic → click **Proceed unbounded** in the banner, or
+ `curl -X POST localhost:PORT/control/allow-unbounded` (cluster-wide,
+ releases every held pod), or deploy with `LEDGER_UNBOUNDED_OK=1`.
+
+A plain Resume is deliberately ignored while the question is open. Resumed
+runs and runs whose target holds no recent data never trip the guard.
+
### Bound is opt-in — pick the mode deliberately
| Situation | LEDGER_CD_UPPER_BOUND | Behavior |
diff --git a/src/config/loader.ts b/src/config/loader.ts
index 6b436d9..0dc0fe0 100644
--- a/src/config/loader.ts
+++ b/src/config/loader.ts
@@ -26,6 +26,7 @@ function envToRawConfig(env: NodeJS.ProcessEnv) {
cdUpperBoundMs: env.LEDGER_CD_UPPER_BOUND,
captureTransformErrors: env.LEDGER_CAPTURE_TRANSFORM_ERRORS,
startPaused: env.LEDGER_START_PAUSED,
+ unboundedOk: env.LEDGER_UNBOUNDED_OK,
dryRun: env.DRY_RUN,
dryRunSamplePct: env.DRY_RUN_SAMPLE_PCT,
},
diff --git a/src/config/schema.ts b/src/config/schema.ts
index 9605a2f..672ec70 100644
--- a/src/config/schema.ts
+++ b/src/config/schema.ts
@@ -82,6 +82,8 @@ export const configSchema = z.object({
// click starts the whole fleet, pods that join later start
// immediately, and a pod that restarts after Start stays started.
startPaused: booleanFromEnv.default(false),
+ /** Explicit no-mirror declaration: skips the unbounded-with-live-target startup guard. */
+ unboundedOk: booleanFromEnv.default(false),
// Dry run: sampled rehearsal against a Null-engine clone.
dryRun: booleanFromEnv.default(false),
dryRunSamplePct: numberFromEnv.default(2).pipe(z.number().min(0.1).max(5)),
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index a1d1431..8a6478c 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -729,6 +729,23 @@ function updatePhaseBadges() {
}
updatePhaseBadges();
+async function allowUnbounded(btn) {
+ if (!armed.get(btn)) {
+ armed.set(btn, true);
+ btn.dataset.label = btn.textContent;
+ btn.textContent = 'Click again to confirm: NOTHING mirrors traffic';
+ btn.classList.add('armed');
+ setTimeout(function () { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 8000);
+ return;
+ }
+ armed.delete(btn); btn.disabled = true;
+ try {
+ var res = await fetch('/control/allow-unbounded', { method: 'POST' });
+ var out = await res.json();
+ toast(out.allowed ? '\u2705 no-mirror declared \u2014 held pods release within seconds' : '\u274c ' + (out.reason || res.status));
+ } catch (e) { toast('\u274c ' + e.message); }
+}
+
async function startFinalCheck(btn) {
var body = {};
var cutRaw = (document.getElementById('fc-cutover').value || '').trim();
@@ -936,17 +953,31 @@ async function tick() {
var hint = document.getElementById('pause-hint');
if (isPaused) {
hint.style.display = '';
- hint.textContent = stats.pauseReason === 'not-started'
- ? '\u23f8 NOT STARTED \u2014 deployed and waiting. Nothing has been read, mapped or indexed yet; '
- + 'run preflight, build indexes and rehearse first, then click Start to begin the run (all pods).'
- : '\u23f8 ENGINE PAUSED' +
- (stats.pauseReason === 'breaker-transient' ? ' (backend outage \u2014 auto-resume armed)' :
- stats.pauseReason === 'breaker-data' ? ' (systematic data problem \u2014 needs you)' : ' (by operator)') +
- ' \u2014 Retry / Replay / Waive only QUEUE work; click Resume to process it.';
+ if (stats.pauseReason === 'boundary-unset') {
+ hint.innerHTML = '\u26a0 HELD BY THE BOUNDARY GUARD \u2014 the target ClickHouse is already receiving live data and no cd bound is set. '
+ + 'If a mirror re-ingests the same requests on both sides, running unbounded WILL duplicate the overlap window. '
+ + 'Either apply a bound (Tee boundary card below), or \u2014 if NOTHING mirrors traffic between the stacks \u2014 '
+ + '';
+ } else {
+ hint.textContent = stats.pauseReason === 'not-started'
+ ? '\u23f8 NOT STARTED \u2014 deployed and waiting. Nothing has been read, mapped or indexed yet; '
+ + 'run preflight, build indexes and rehearse first, then click Start to begin the run (all pods).'
+ : '\u23f8 ENGINE PAUSED' +
+ (stats.pauseReason === 'breaker-transient' ? ' (backend outage \u2014 auto-resume armed)' :
+ stats.pauseReason === 'breaker-data' ? ' (systematic data problem \u2014 needs you)' : ' (by operator)') +
+ ' \u2014 Retry / Replay / Waive only QUEUE work; click Resume to process it.';
+ }
} else { hint.style.display = 'none'; }
var prBtn = document.getElementById('btn-pauseresume');
if (prBtn) {
- if (isPaused) {
+ if (isPaused && stats.pauseReason === 'boundary-unset') {
+ // a plain Resume cannot answer the mirror question — the banner
+ // above carries the two real actions (bound / proceed unbounded)
+ prBtn.dataset.action = '';
+ prBtn.innerHTML = '\u25b6 Resume';
+ prBtn.classList.remove('primary');
+ prBtn.disabled = true;
+ } else if (isPaused) {
prBtn.dataset.action = 'resume';
prBtn.innerHTML = stats.pauseReason === 'not-started' ? '\u25b6 Start' : '\u25b6 Resume';
prBtn.classList.add('primary');
@@ -1076,6 +1107,7 @@ var SCENARIOS = [
'
LEDGER_CD_UPPER_BOUND: LEAVE UNSET. The migration must take everything, including data still arriving in the old cluster \u2014 top-up passes chase it until the final drain finds nothing new.
' +
'
Cutover-first: switch SDK ingestion to the new cluster, then run the migration (old drill data is frozen). Bulk-before-cutover: run the bulk first, switch ingestion, then let the final top-up pass drain the tail.
' +
'
Ignore the Tee boundary card \u2014 it is for mirrored setups only. Applying a bound here would ORPHAN newly arrived data.
' +
+ '
If ingestion already switched to the new cluster before the run starts, the startup guard will hold and ask \u2014 Proceed unbounded is the correct answer for this scenario.
' +
'' },
{ id: 'tee-old', name: '2 \u00b7 Mirror old \u2192 new',
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index 0b37edb..aac9c7c 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -138,7 +138,7 @@ export class ChunkOrchestrator {
private consecutiveFailed = 0;
private sourceShrankChunks = 0;
private streakHadPermanent = false;
- private pauseReason: 'operator' | 'not-started' | 'breaker-transient' | 'breaker-data' | null = null;
+ private pauseReason: 'operator' | 'not-started' | 'boundary-unset' | 'breaker-transient' | 'breaker-data' | null = null;
private probeOkStreak = 0;
private autoResuming = false;
private resumeProbeTimer: NodeJS.Timeout | null = null;
@@ -172,7 +172,7 @@ export class ChunkOrchestrator {
// -------------------------------------------------------------------------
stopAfterChunk(): void { this.stopping = true; }
- pause(reason: 'operator' | 'not-started' | 'breaker-transient' | 'breaker-data' = 'operator'): void {
+ pause(reason: 'operator' | 'not-started' | 'boundary-unset' | 'breaker-transient' | 'breaker-data' = 'operator'): void {
this.paused = true;
this.pauseReason = reason;
if (this.status === 'running') this.status = 'paused';
@@ -207,6 +207,42 @@ export class ChunkOrchestrator {
// Main
// -------------------------------------------------------------------------
+ private async boundaryGuard(): Promise {
+ const { config } = this.d;
+ if (config.ledger.cdUpperBoundMs != null || config.ledger.unboundedOk) return;
+ if (await this.d.ledger.getStoredBound(this.runId).catch(() => null)) return;
+ if (await this.d.ledger.getUnboundedAck(this.runId).catch(() => false)) return;
+ // only a FRESH run: a resumed run already made this decision
+ const counts = await this.d.ledger.statusCounts(this.runId).catch(() => null);
+ if (counts === null || Object.values(counts).reduce((a, b) => a + b, 0) > 0) return;
+ const live = await this.d.staging.hasLiveCdSince(Date.now() - 30 * 60_000).catch(() => false);
+ if (!live) return;
+
+ this.pause('boundary-unset');
+ this.logger.warn(
+ { runId: this.runId },
+ 'GUARD: target ClickHouse is receiving live data and no cd upper bound is set — if a mirror re-ingests the same requests on both sides, running unbounded WILL duplicate the overlap window. Apply a bound (POST /control/set-boundary) or declare no-mirror (POST /control/allow-unbounded).',
+ );
+ while (!this.stopping) {
+ if ((await this.d.ledger.getStoredBound(this.runId).catch(() => null)) !== null) {
+ this.logger.info({ runId: this.runId }, 'Boundary guard released: a cd bound was applied');
+ this.resume();
+ return;
+ }
+ if (await this.d.ledger.getUnboundedAck(this.runId).catch(() => false)) {
+ this.logger.warn({ runId: this.runId }, 'Boundary guard released: operator declared no-mirror — running unbounded');
+ this.resume();
+ return;
+ }
+ if (!this.paused) {
+ // a plain Resume does not answer the mirror question — re-hold
+ this.pause('boundary-unset');
+ this.logger.warn('Resume ignored while the boundary question is open — apply a bound or POST /control/allow-unbounded');
+ }
+ await sleep(3_000);
+ }
+ }
+
async run(): Promise {
this.status = 'running';
this.startedAt = Date.now();
@@ -242,6 +278,18 @@ export class ChunkOrchestrator {
}
}
+ // ── UNBOUNDED-WITH-LIVE-TARGET GUARD ──────────────────────────────────
+ // The one mistake the tool cannot detect afterwards (field: Wurth-it): a
+ // mirrored cutover migrated without LEDGER_CD_UPPER_BOUND duplicates the
+ // whole overlap window. The condition IS detectable up front — a fresh
+ // run whose target ClickHouse is already receiving live data — so the
+ // run holds there until the operator answers the mirror question: apply
+ // a bound (set-boundary), or declare no-mirror (allow-unbounded).
+ if (!this.dryRun) {
+ await this.boundaryGuard();
+ if (this.stopping) { this.status = 'stopped'; return; }
+ }
+
// Transient-outage self-healing: only acts while paused with reason
// 'breaker-transient' (backend outage tripped the failure breaker) —
// every other pause stays owned by the operator.
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index c7c6d41..50f17d4 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -501,6 +501,14 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise ({ ...boundaryState, applied: boundaryApplied }));
+ // Startup-guard answer: "nothing mirrors traffic between the stacks" —
+ // cluster-wide (stored in run config), releases every held pod.
+ app.post('/control/allow-unbounded', async () => {
+ await ledger.setUnboundedAck(config.ledger.runId, config.worker.podId);
+ logger.warn('Operator declared no-mirror: unbounded run allowed — held pods release within seconds');
+ return { allowed: true, note: 'held pods release within ~3s; the decision is stored cluster-wide in mig_run_config' };
+ });
+
// ── ONE endpoint for the whole boundary flow ────────────────────────────
// {} → detect, and auto-apply when the seam is an exact ingestion-pause
// gap; {"acceptAnchor":true} → also take an anchor suggestion; {"boundMs"}
diff --git a/src/state/ledger-store.ts b/src/state/ledger-store.ts
index a4849b3..81a8091 100644
--- a/src/state/ledger-store.ts
+++ b/src/state/ledger-store.ts
@@ -482,6 +482,7 @@ export class LedgerStore {
private rc(): Collection<{
_id: string; cd_upper_bound_ms: number; set_at: Date; set_by: string;
start_gate_open?: boolean; start_gate_opened_at?: Date; start_gate_opened_by?: string;
+ unbounded_ok?: boolean; unbounded_ok_by?: string; unbounded_ok_at?: Date;
}> {
if (!this.coll) throw new Error('LedgerStore not connected');
return this.client.db(this.dbName).collection('mig_run_config');
@@ -583,6 +584,20 @@ export class LedgerStore {
return doc?.cd_upper_bound_ms ?? null;
}
+ /** Cluster-wide operator answer to the startup guard: "nothing mirrors traffic — run unbounded". */
+ async getUnboundedAck(runId: string): Promise {
+ const doc = await this.rc().findOne({ _id: runId });
+ return doc?.unbounded_ok === true;
+ }
+
+ async setUnboundedAck(runId: string, by: string): Promise {
+ await this.rc().updateOne(
+ { _id: runId },
+ { $set: { unbounded_ok: true, unbounded_ok_by: by, unbounded_ok_at: new Date() } },
+ { upsert: true },
+ );
+ }
+
async setStoredBound(runId: string, boundMs: number, setBy: string): Promise {
await this.rc().updateOne(
{ _id: runId },
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index 4d0f85d..ed7fd8a 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -573,6 +573,17 @@ export class StagingManager {
return out;
}
+ /** Does the live table hold ANY row with cd at/after fromMs? (partition-pruned, LIMIT 1) */
+ async hasLiveCdSince(fromMs: number): Promise {
+ const res = await this.ch().query({
+ query: `SELECT 1 AS x FROM ${this.fq(this.config.table)}
+ WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) LIMIT 1`,
+ query_params: { lo: fromMs },
+ format: 'JSONEachRow',
+ });
+ return (await res.json<{ x: number }>()).length > 0;
+ }
+
/** Live rows in [fromMs, toMs) whose _id is one of the given ids. */
async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
let total = 0;
diff --git a/tests/integration/cross-collection-pods.test.ts b/tests/integration/cross-collection-pods.test.ts
index 6507853..6452a21 100644
--- a/tests/integration/cross-collection-pods.test.ts
+++ b/tests/integration/cross-collection-pods.test.ts
@@ -87,6 +87,7 @@ describe('cross-collection scheduling with two pods', () => {
process.env.MONGO_PAGE_SIZE = '100'; // slow pods down enough to overlap
process.env.LEDGER_MONITOR_INTERVAL_MS = '0';
process.env.BACKPRESSURE_ENABLED = 'false';
+ process.env.LEDGER_UNBOUNDED_OK = 'true'; // no-mirror declaration — recent-cd rows are this test's own data
process.env.MULTI_POD_ENABLED = 'true';
const config = loadConfig();
diff --git a/tests/integration/ledger-engine.test.ts b/tests/integration/ledger-engine.test.ts
index e92f5cd..e9d1ca3 100644
--- a/tests/integration/ledger-engine.test.ts
+++ b/tests/integration/ledger-engine.test.ts
@@ -357,6 +357,7 @@ describe('ledger engine end-to-end', () => {
process.env.LEDGER_CHUNK_DOCS_TARGET = '500';
process.env.LEDGER_MONITOR_INTERVAL_MS = '0';
process.env.BACKPRESSURE_ENABLED = 'false';
+ process.env.LEDGER_UNBOUNDED_OK = 'true'; // no-mirror declaration for the e2e harness
const config = loadConfig();
const mongoReader = new MongoReader({
diff --git a/tests/integration/live-parallel.test.ts b/tests/integration/live-parallel.test.ts
index 254b63d..ec4f45a 100644
--- a/tests/integration/live-parallel.test.ts
+++ b/tests/integration/live-parallel.test.ts
@@ -103,6 +103,8 @@ describe('migration under concurrent live ingestion', () => {
// would trip it and pause the engine (which the test would catch below).
process.env.LEDGER_MONITOR_INTERVAL_MS = '150';
process.env.BACKPRESSURE_ENABLED = 'false';
+ // live-parallel IS the scenario the startup guard asks about — declare no-mirror, like the operator would
+ process.env.LEDGER_UNBOUNDED_OK = 'true';
const config = loadConfig();
const mongoReader = new MongoReader({
diff --git a/tests/integration/multi-collection-and-rebuild.test.ts b/tests/integration/multi-collection-and-rebuild.test.ts
index bfcad73..366e7f0 100644
--- a/tests/integration/multi-collection-and-rebuild.test.ts
+++ b/tests/integration/multi-collection-and-rebuild.test.ts
@@ -125,6 +125,7 @@ describe('multi-collection scoping + ledger rebuild', () => {
process.env.LEDGER_CHUNK_DOCS_TARGET = '250';
process.env.LEDGER_MONITOR_INTERVAL_MS = '0';
process.env.BACKPRESSURE_ENABLED = 'false';
+ process.env.LEDGER_UNBOUNDED_OK = 'true'; // no-mirror declaration — the top-up tests migrate recent-cd docs
config = loadConfig();
const mongoReader = new MongoReader({
@@ -890,7 +891,7 @@ describe('multi-collection scoping + ledger rebuild', () => {
SERVICE_NAME: 'skiponly', MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`,
MANIFEST_DB: DB, CLICKHOUSE_URL: CH_URL, CLICKHOUSE_PASSWORD: CH_PASSWORD, CLICKHOUSE_DB: DB,
LEDGER_RUN_ID: SK, LEDGER_CHUNK_DOCS_TARGET: '5000', MONGO_PAGE_SIZE: '500',
- LEDGER_MONITOR_INTERVAL_MS: '0', BACKPRESSURE_ENABLED: 'false', MULTI_POD_ENABLED: 'false',
+ LEDGER_MONITOR_INTERVAL_MS: '0', BACKPRESSURE_ENABLED: 'false', MULTI_POD_ENABLED: 'false', LEDGER_UNBOUNDED_OK: 'true',
POD_ID: 'skip-pod',
});
delete process.env.LEDGER_CD_UPPER_BOUND;
diff --git a/tests/integration/outage-chaos.test.ts b/tests/integration/outage-chaos.test.ts
index 4c71a4d..f3fa414 100644
--- a/tests/integration/outage-chaos.test.ts
+++ b/tests/integration/outage-chaos.test.ts
@@ -79,6 +79,7 @@ describe.skipIf(!ENABLED)('backing-service outage chaos (CHAOS_OUTAGE=1, dedicat
LEDGER_LEASE_SEC: '3',
LEDGER_MONITOR_INTERVAL_MS: '0',
BACKPRESSURE_ENABLED: 'false',
+ LEDGER_UNBOUNDED_OK: 'true', // no-mirror declaration for the chaos harness
MULTI_POD_ENABLED: 'true',
LOG_LEVEL: 'fatal',
CHAOS_LOG_LEVEL: 'warn', // worker pino level — stderrTail captures it
diff --git a/tests/integration/pod-chaos.test.ts b/tests/integration/pod-chaos.test.ts
index 5a777b5..bb4fa08 100644
--- a/tests/integration/pod-chaos.test.ts
+++ b/tests/integration/pod-chaos.test.ts
@@ -90,6 +90,7 @@ describe('pod chaos: random SIGKILL across all stages, exact end state', () => {
LEDGER_LEASE_SEC: '2', // dead pods' leases recover in seconds
LEDGER_MONITOR_INTERVAL_MS: '0',
BACKPRESSURE_ENABLED: 'false',
+ LEDGER_UNBOUNDED_OK: 'true', // live writers run alongside the pods — the guard's question is answered
MULTI_POD_ENABLED: 'true',
LOG_LEVEL: 'fatal',
// fast retries: local CH is healthy here; prod-like backoff would only
diff --git a/tests/integration/startup-guard.test.ts b/tests/integration/startup-guard.test.ts
new file mode 100644
index 0000000..ad46416
--- /dev/null
+++ b/tests/integration/startup-guard.test.ts
@@ -0,0 +1,192 @@
+/**
+ * Unbounded-with-live-target startup guard — the "Wurth-it mistake" made
+ * impossible to make silently. Pinned here:
+ *
+ * - a FRESH run against a ClickHouse that is already receiving live data,
+ * with no cd bound set, HOLDS before mapping (pauseReason boundary-unset)
+ * - a plain Resume does not answer the mirror question: the run re-holds
+ * - applying a bound releases the hold and the run respects it
+ * - the explicit no-mirror ack (allow-unbounded) releases the hold and the
+ * run proceeds unbounded — including on a re-run over already-migrated
+ * data (idempotent redo across runs)
+ */
+import { describe, it, expect, beforeAll, afterAll } from 'vitest';
+import pino from 'pino';
+import { createHash } from 'node:crypto';
+import { MongoClient } from 'mongodb';
+import { createClient, type ClickHouseClient } from '@clickhouse/client';
+
+import { LedgerStore } from '../../src/state/ledger-store.ts';
+import { DlqStore } from '../../src/state/dlq-store.ts';
+import { StagingManager } from '../../src/target/staging-manager.ts';
+import { MongoReader } from '../../src/source/mongo-reader.ts';
+import { RetryPolicy } from '../../src/runtime/retry-policy.ts';
+import { HashResolver } from '../../src/transform/hash-resolver.ts';
+import { ChunkOrchestrator } from '../../src/runtime/chunk-orchestrator.ts';
+import { loadConfig } from '../../src/config/loader.ts';
+
+const MONGO_URI = 'mongodb://localhost:27017/?directConnection=true';
+const CH_URL = process.env.TEST_CLICKHOUSE_URL ?? 'http://localhost:8123';
+const CH_PASSWORD = process.env.TEST_CLICKHOUSE_PASSWORD ?? '';
+const DB = 'test_mig_guard';
+const logger = pino({ level: 'silent' });
+
+const APP = 'app_guard';
+const EV = 'views';
+const COLL = `drill_events${createHash('sha1').update(EV + APP).digest('hex')}`;
+const HIST = 800;
+const POST = 50;
+const FLIP = Date.now() - 20 * 60_000; // tee flip 20 min ago
+const sleep = (ms: number): Promise => new Promise((r) => setTimeout(r, ms));
+
+describe('unbounded-with-live-target startup guard', () => {
+ let ch: ClickHouseClient;
+ let mc: MongoClient;
+ let ledger: LedgerStore;
+ let dlqStore: DlqStore;
+ let staging: StagingManager;
+ let hashResolver: HashResolver;
+ const closers: Array<() => Promise> = [];
+
+ const mkOrchestrator = async (runId: string, podId: string): Promise => {
+ Object.assign(process.env, {
+ SERVICE_NAME: 'guard-test',
+ MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`, MANIFEST_DB: DB,
+ CLICKHOUSE_URL: CH_URL, CLICKHOUSE_PASSWORD: CH_PASSWORD, CLICKHOUSE_DB: DB,
+ LEDGER_RUN_ID: runId, LEDGER_CHUNK_DOCS_TARGET: '400', MONGO_PAGE_SIZE: '200',
+ LEDGER_MONITOR_INTERVAL_MS: '0', BACKPRESSURE_ENABLED: 'false',
+ MULTI_POD_ENABLED: 'false', POD_ID: podId,
+ });
+ delete process.env.LEDGER_CD_UPPER_BOUND;
+ delete process.env.LEDGER_UNBOUNDED_OK;
+ delete process.env.LEDGER_START_PAUSED;
+ const config = loadConfig();
+ const mongoReader = new MongoReader({
+ uri: MONGO_URI, database: DB, readPreference: 'primary', readConcern: 'local',
+ retryReads: true, appName: podId, cursorBatchSize: 500, maxTimeMs: 60_000,
+ }, logger);
+ await mongoReader.connect();
+ closers.push(() => mongoReader.close());
+ return new ChunkOrchestrator({
+ config, logger, mongoReader, ledger, dlq: dlqStore, staging,
+ retryPolicy: new RetryPolicy({ maxRetries: 3, baseDelayMs: 100, maxDelayMs: 500 }), hashResolver,
+ });
+ };
+
+ const waitFor = async (cond: () => boolean, ms: number): Promise => {
+ const until = Date.now() + ms;
+ while (!cond() && Date.now() < until) await sleep(300);
+ expect(cond()).toBe(true);
+ };
+
+ beforeAll(async () => {
+ mc = new MongoClient(MONGO_URI);
+ await mc.connect();
+ await mc.db(DB).dropDatabase();
+ await mc.db(`${DB}_countly`).dropDatabase();
+ await mc.db(`${DB}_countly`).collection('apps').insertOne({ _id: APP } as never);
+ await mc.db(`${DB}_countly`).collection('events').insertOne({ _id: APP, list: [EV] } as never);
+
+ ch = createClient({ url: CH_URL, password: CH_PASSWORD });
+ await ch.command({ query: `CREATE DATABASE IF NOT EXISTS ${DB}` });
+ await ch.command({ query: `DROP TABLE IF EXISTS ${DB}.drill_events` });
+ await ch.command({
+ query: `CREATE TABLE ${DB}.drill_events (
+ \`a\` LowCardinality(String), \`e\` LowCardinality(String), \`n\` String,
+ \`uid\` String, \`uid_canon\` Nullable(String), \`did\` String, \`lsid\` Nullable(String),
+ \`_id\` String, \`ts\` DateTime64(3), \`up\` JSON(max_dynamic_paths = 32),
+ \`custom\` Nullable(JSON(max_dynamic_paths = 0)), \`cmp\` Nullable(JSON(max_dynamic_paths = 0)),
+ \`sg\` JSON(max_dynamic_paths = 0), \`c\` UInt32, \`s\` Float64, \`dur\` Float64,
+ \`lu\` Nullable(DateTime64(3)), \`cd\` DateTime64(3) DEFAULT now64(3))
+ ENGINE = MergeTree PARTITION BY toYYYYMM(ts, 'UTC') ORDER BY (a, e, n, ts)`,
+ });
+
+ // source: history before the flip + the mirror's post-flip re-ingested docs
+ const docs: Record[] = [];
+ for (let i = 0; i < HIST; i++) {
+ const t = FLIP - (HIST - i) * 60_000;
+ docs.push({ _id: `h_${i}`, uid: String(i % 20), did: `d${i}`, ts: t, cd: new Date(t), sg: { v: i }, c: 1 });
+ }
+ for (let i = 0; i < POST; i++) {
+ const t = FLIP + i * 1_000;
+ docs.push({ _id: `post_${i}`, uid: 'p', did: 'd', ts: t, cd: new Date(t), sg: {}, c: 1 });
+ }
+ await mc.db(DB).collection(COLL).insertMany(docs as never[]);
+ await mc.db(DB).collection(COLL).createIndex({ cd: 1, _id: 1 });
+
+ // the guard's trigger: the target is ALREADY receiving live traffic
+ const nowIso = (ms: number): string => new Date(ms).toISOString().replace('T', ' ').replace('Z', '');
+ await ch.insert({
+ table: `${DB}.drill_events`, format: 'JSONEachRow',
+ values: Array.from({ length: 5 }, (_, i) => ({
+ a: 'live_app', e: '[CLY]_custom', n: 'live', uid: 'u', did: 'd', _id: `live_${i}`,
+ ts: nowIso(Date.now() - 5 * 60_000 + i), cd: nowIso(Date.now() - 5 * 60_000 + i),
+ up: {}, sg: {}, c: 1, s: 0, dur: 0,
+ })),
+ });
+
+ ledger = new LedgerStore(MONGO_URI, DB, logger);
+ dlqStore = new DlqStore(MONGO_URI, DB, logger);
+ staging = new StagingManager({
+ url: CH_URL, database: DB, table: 'drill_events', username: 'default', password: CH_PASSWORD, queryTimeoutMs: 60_000,
+ }, logger);
+ hashResolver = new HashResolver({ uri: MONGO_URI, countlyDb: `${DB}_countly` }, logger);
+ await ledger.connect();
+ await dlqStore.connect();
+ await staging.connect();
+ await hashResolver.build();
+ closers.push(() => ledger.close(), () => dlqStore.close(), () => staging.close(), () => hashResolver.close());
+ }, 120_000);
+
+ afterAll(async () => {
+ for (const close of closers) await close().catch(() => {});
+ await ch.command({ query: `DROP DATABASE IF EXISTS ${DB}` }).catch(() => {});
+ await ch.close();
+ await mc.db(DB).dropDatabase().catch(() => {});
+ await mc.db(`${DB}_countly`).dropDatabase().catch(() => {});
+ await mc.close();
+ }, 60_000);
+
+ const chCount = async (where: string): Promise => {
+ const res = await ch.query({ query: `SELECT count() AS n FROM ${DB}.drill_events WHERE ${where}`, format: 'JSONEachRow' });
+ return Number((await res.json<{ n: string }>())[0].n);
+ };
+
+ it('holds a fresh unbounded run, ignores plain Resume, and releases when a bound is applied', async () => {
+ const orch = await mkOrchestrator('guard-1', 'guard-pod-1');
+ const done = orch.run();
+
+ await waitFor(() => orch.getStats().status === 'paused' && orch.getStats().pauseReason === 'boundary-unset', 20_000);
+
+ // Resume without answering the mirror question → re-held
+ orch.resume();
+ await sleep(4_500);
+ expect(orch.getStats().status).toBe('paused');
+ expect(orch.getStats().pauseReason).toBe('boundary-unset');
+
+ // applying a bound answers it — the run releases AND respects the bound
+ await ledger.setStoredBound('guard-1', FLIP, 'guard-test');
+ await done;
+ const stats = orch.getStats();
+ expect(stats.status).toBe('completed');
+ expect(stats.cdUpperBoundMs).toBe(FLIP);
+ expect(await chCount("_id LIKE 'h_%'")).toBe(HIST);
+ expect(await chCount("_id LIKE 'post_%'")).toBe(0); // post-flip mirror copies never migrated
+ }, 120_000);
+
+ it('the explicit no-mirror ack releases the hold and the run proceeds unbounded', async () => {
+ const orch = await mkOrchestrator('guard-2', 'guard-pod-2');
+ const done = orch.run();
+ await waitFor(() => orch.getStats().status === 'paused' && orch.getStats().pauseReason === 'boundary-unset', 20_000);
+
+ await ledger.setUnboundedAck('guard-2', 'guard-test');
+ await done;
+ expect(orch.getStats().status).toBe('completed');
+ expect(orch.getStats().cdUpperBoundMs).toBeNull();
+ // unbounded, as declared: the post-flip docs are migrated this time
+ expect(await chCount("_id LIKE 'post_%'")).toBe(POST);
+ expect(await chCount("_id LIKE 'h_%'")).toBe(HIST); // idempotent redo — no duplicates
+ // a completed decision never re-arms: a third fresh-looking pod for the
+ // same run sails through (statusCounts > 0 short-circuits anyway)
+ }, 180_000);
+});
From 26b6d770c46f2ea012a179b4fd2854f391865c7d Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:09:39 +0300
Subject: [PATCH 04/64] docs: keep deployment references generic
Co-Authored-By: Claude Fable 5
---
src/runtime/chunk-orchestrator.ts | 2 +-
tests/integration/startup-guard.test.ts | 2 +-
2 files changed, 2 insertions(+), 2 deletions(-)
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index aac9c7c..db3f100 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -279,7 +279,7 @@ export class ChunkOrchestrator {
}
// ── UNBOUNDED-WITH-LIVE-TARGET GUARD ──────────────────────────────────
- // The one mistake the tool cannot detect afterwards (field: Wurth-it): a
+ // The one mistake the tool cannot detect afterwards (seen in the field): a
// mirrored cutover migrated without LEDGER_CD_UPPER_BOUND duplicates the
// whole overlap window. The condition IS detectable up front — a fresh
// run whose target ClickHouse is already receiving live data — so the
diff --git a/tests/integration/startup-guard.test.ts b/tests/integration/startup-guard.test.ts
index ad46416..4bdf1c0 100644
--- a/tests/integration/startup-guard.test.ts
+++ b/tests/integration/startup-guard.test.ts
@@ -1,5 +1,5 @@
/**
- * Unbounded-with-live-target startup guard — the "Wurth-it mistake" made
+ * Unbounded-with-live-target startup guard — the missing-bound mistake made
* impossible to make silently. Pinned here:
*
* - a FRESH run against a ClickHouse that is already receiving live data,
From ba960a4637d68e363128f836962c83c4c77215f2 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:20:00 +0300
Subject: [PATCH 05/64] =?UTF-8?q?fix(ledger):=20address=20review=20?=
=?UTF-8?q?=E2=80=94=20dedupe=20native-counterpart=20evidence,=20stricter?=
=?UTF-8?q?=20final=20check,=20cutover-clamped=20content=20audit,=20dedupe?=
=?UTF-8?q?=20UI?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
All four review findings were real; each is now pinned by a test:
- dedupe-overlap: an old-Mongo id match alone is not proof of duplication —
when the tee (or new-side ingestion) dropped a request, the migrated row
is the ONLY copy. Every hour bucket now needs count-evidence of native
counterparts (native = live − matched must cover matched; a bucket with
zero natives is the outage signature outright). Falling short → bucket
skipped, reported under 'unsafe', never deleted.
- final check: pending DLQ docs are an undecided absence — now a FAIL with
the action named, not a note; waived stays a note. Windows the audit
classifies 'pending' (live=0: the WHOLE window missing from the target)
now FAIL instead of hiding behind 'every count matches'; unscopable
collections get an explanatory note instead.
- content audit: clamped to the cutover (new upToMs param) — sampling
post-cutover old-side docs reported phantom 'missing' rows on bounded
tee runs.
- dashboard: Tee-overlap dedupe card (window inputs, dry run first, delete
unlocks only after it, unsafe buckets surfaced in red).
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 17 ++-
src/http/ledger-viz-route.ts | 72 ++++++++++++-
src/runtime/chunk-orchestrator.ts | 15 ++-
src/runtime/dedupe-overlap.ts | 132 +++++++++++++++++------
src/runtime/final-check.ts | 25 ++++-
src/runtime/ledger-engine.ts | 2 +-
tests/integration/cd-upper-bound.test.ts | 9 ++
tests/integration/dedupe-overlap.test.ts | 55 ++++++++--
tests/integration/final-check.test.ts | 20 +++-
9 files changed, 290 insertions(+), 57 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index bb6b135..a217bb5 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -86,9 +86,20 @@ A mirrored cutover migrated WITHOUT `LEDGER_CD_UPPER_BOUND` copies the
mirror's re-ingested docs on top of natively ingested rows: every event in
the overlap window (tee flip → migration completion) exists twice in
ClickHouse. The copies are separable — the migrated copy's `_id` exists in
-the old cluster's Mongo; the native one's doesn't — so cleanup is exact and
-loses nothing. **Must run before the old cluster is decommissioned** (old
-Mongo is the separator).
+the old cluster's Mongo; the native one's doesn't. **Must run before the old
+cluster is decommissioned** (old Mongo is the separator).
+
+An id match alone is not proof of duplication: if the tee (or the new
+side's ingestion) dropped a request, the migrated row is the ONLY copy of
+that event. Every hour bucket therefore needs count-evidence of native
+counterparts — `native = live − matched` must roughly cover `matched` —
+before anything in it is deleted. Buckets that fall short are skipped and
+reported (`unsafe` in the result); review those hours (tee outage? wrong
+start time?) instead of forcing them.
+
+There is a dashboard card for this (Overview → **Tee-overlap dedupe**:
+enter the window, *Dry run* first — *Delete duplicates* unlocks only after
+it) as well as the endpoints below.
```bash
# 1. DRY RUN (counts only): fromMs = tee flip / IP swap, toMs = migration completion
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index 8a6478c..c341519 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -342,6 +342,17 @@ const PAGE = `
+
+
Tee-overlap dedupe (fix a mirrored run that migrated WITHOUT the cd bound: remove the migrated copies of events the new cluster already ingested natively)
+
+
+
+
+
+
+
Only for runs that migrated a mirrored setup unbounded. Every hour bucket is checked for count-evidence of native counterparts before anything is deleted — buckets where migrated rows are the ONLY copy are skipped and reported. Old-cluster Mongo must still be up. Start must be AT or AFTER the actual flip: too early deletes real data, too late only leaves a few duplicates.
+
+
Dead-letter queue (unmigratable docs, stored with their full raw source — replay after a fix, or waive)
@@ -746,6 +757,65 @@ async function allowUnbounded(btn) {
} catch (e) { toast('\u274c ' + e.message); }
}
+function ddWindow() {
+ var f = Date.parse((document.getElementById('dd-from').value || '').trim());
+ var t = Date.parse((document.getElementById('dd-to').value || '').trim());
+ if (isNaN(f) || isNaN(t) || !(f < t)) { toast('Enter both times as ISO (e.g. 2026-09-18T18:00Z), start before end'); return null; }
+ return { fromMs: f, toMs: t };
+}
+async function startDedupe(btn, execute) {
+ var w = ddWindow();
+ if (!w) return;
+ if (execute && !armed.get(btn)) {
+ armed.set(btn, true);
+ btn.dataset.label = btn.textContent;
+ btn.textContent = 'Click again to DELETE the counted duplicates';
+ btn.classList.add('armed');
+ setTimeout(function () { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }, 8000);
+ return;
+ }
+ if (execute) { armed.delete(btn); btn.textContent = btn.dataset.label; btn.classList.remove('armed'); }
+ btn.disabled = true;
+ try {
+ var res = await fetch('/control/dedupe-overlap', { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ fromMs: w.fromMs, toMs: w.toMs, execute: !!execute }) });
+ var out = await res.json();
+ if (!out.started) toast('Not started: ' + (out.reason || 'unknown'));
+ else toast(execute ? 'Deleting duplicates\u2026' : 'Dry run started \u2014 counting duplicates');
+ } catch (e) { toast('failed: ' + e.message); }
+ btn.disabled = false;
+ pollDedupe();
+}
+var ddTimer = null;
+async function pollDedupe() {
+ try {
+ var dd = await fetch('/api/dedupe-overlap').then(function (r) { return r.json(); });
+ renderDedupe(dd);
+ if (dd.status === 'running') { clearTimeout(ddTimer); ddTimer = setTimeout(pollDedupe, 2000); }
+ } catch (e) { /* engine restarting */ }
+}
+function renderDedupe(dd) {
+ var el = document.getElementById('dedupe-out');
+ if (!el || !dd || dd.status === 'not_run') return;
+ var execBtn = document.getElementById('btn-dd-exec');
+ if (dd.status === 'running') { el.innerHTML = '
\u26a0 ' + fmt(t.unsafeMatched) + ' matched row(s) in ' + unsafeN + ' hour bucket(s) lack count-evidence of a native counterpart \u2014 there the migrated row may be the ONLY copy. They were ' + (dd.execute ? 'NOT deleted' : 'excluded') + '; review those hours (tee outage / wrong start time?) before touching them.
';
+ }
+ if (!dd.execute && dd.lastDryRun) {
+ html += '
Window measured \u2014 the Delete button is now enabled for this exact window.
';
+ if (execBtn) { execBtn.disabled = false; execBtn.title = ''; }
+ }
+ html += '
window: ' + new Date(dd.fromMs).toISOString() + ' \u2192 ' + new Date(dd.toMs).toISOString() + ' \u00b7 ' + (dd.collections || []).length + ' collection(s) with matches
';
+ el.innerHTML = html;
+}
+
async function startFinalCheck(btn) {
var body = {};
var cutRaw = (document.getElementById('fc-cutover').value || '').trim();
@@ -1333,7 +1403,7 @@ async function slowTick() {
} catch { /* engine restarting */ }
}
-tick(); slowTick(); pollFinalCheck();
+tick(); slowTick(); pollFinalCheck(); pollDedupe();
setInterval(tick, 2000);
setInterval(slowTick, 5000);
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index db3f100..41360ed 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -1626,7 +1626,7 @@ export class ChunkOrchestrator {
* so value-level equality there belongs to the differential harness, which
* pins the transform itself).
*/
- async contentAudit(samplesPerCollection = 500): Promise<{
+ async contentAudit(samplesPerCollection = 500, upToMs: number | null = null): Promise<{
sampled: number; matched: number; missing: number; different: number;
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}> {
@@ -1649,7 +1649,14 @@ export class ChunkOrchestrator {
const [lowDoc] = await coll.find({ cd: { $type: 'date' } }).sort({ cd: 1 }).limit(1).project({ cd: 1 }).toArray();
const [highDoc] = await coll.find({ cd: { $type: 'date' } }).sort({ cd: -1 }).limit(1).project({ cd: 1 }).toArray();
if (!lowDoc || !highDoc) continue;
- const lo = (lowDoc.cd as Date).getTime(), hi = (highDoc.cd as Date).getTime();
+ const lo = (lowDoc.cd as Date).getTime();
+ let hi = (highDoc.cd as Date).getTime();
+ // tee/cutover clamp: post-cutover old-side docs were deliberately
+ // never migrated — sampling them reports phantom "missing" rows
+ if (upToMs !== null) {
+ if (lo >= upToMs) continue;
+ hi = Math.min(hi, upToMs - 1);
+ }
// K random cd probe points, a small run of docs from each — cheap
// index-served sampling without $sample's whole-collection scan.
@@ -1658,7 +1665,9 @@ export class ChunkOrchestrator {
const docs: Record[] = [];
for (let k = 0; k < probes; k++) {
const at = new Date(lo + Math.floor(((k + 0.5) / probes) * (hi - lo)));
- const page = await coll.find({ cd: { $gte: at } }).sort({ cd: 1, _id: 1 }).limit(RUN_LEN).toArray();
+ const cdQ: Record = { $gte: at } as never;
+ if (upToMs !== null) (cdQ as Record).$lt = new Date(upToMs);
+ const page = await coll.find({ cd: cdQ }).sort({ cd: 1, _id: 1 }).limit(RUN_LEN).toArray();
docs.push(...(page as Record[]));
}
diff --git a/src/runtime/dedupe-overlap.ts b/src/runtime/dedupe-overlap.ts
index 465e0be..e5f5fb2 100644
--- a/src/runtime/dedupe-overlap.ts
+++ b/src/runtime/dedupe-overlap.ts
@@ -7,33 +7,62 @@
* already ingested natively: every event in the overlap window exists twice
* in ClickHouse, under two different _ids.
*
- * The two copies are cleanly separable: the migrated copy carries an _id
- * that exists in the OLD cluster's Mongo; the native row's _id was minted by
- * the new cluster and does not. And because the mirror only ever re-ingests
- * requests the new cluster served first, every migrated row in the overlap
- * window duplicates a native row — deleting all id-matched rows in the
- * window removes exactly the duplicates, never data.
+ * The migrated copy is identifiable — its _id exists in the OLD cluster's
+ * Mongo; a native row's _id was minted by the new cluster and does not.
+ * But an id match alone is NOT proof of duplication: when the tee (or the
+ * new cluster's ingestion) dropped a request, the migrated row is the ONLY
+ * copy of that event, and deleting it would lose data. The identities
+ * differ per side, so no per-event pairing exists — instead every hour
+ * bucket must carry COUNT evidence of native counterparts:
+ *
+ * native(bucket) = live rows in bucket − id-matched rows in bucket
+ * safe ⇔ native ≥ matched − slack
+ *
+ * In a healthy tee every matched row duplicates a native one, so native is
+ * at least matched (plus mirror losses only ever shrink matched). A bucket
+ * where native falls short holds migrated rows WITHOUT counterparts —
+ * those are skipped, reported, and never deleted.
*
* Safety: dry-run by default (counts only); execute is refused until a dry
* run over the SAME window has completed in this process, and the old
- * cluster's Mongo must still be reachable (it is the separator — this is
- * why cleanup must happen BEFORE the old stack is decommissioned).
+ * cluster's Mongo must still be reachable (it is the separator — cleanup
+ * must happen BEFORE the old stack is decommissioned).
*/
import type { Logger } from 'pino';
import { MongoClient } from 'mongodb';
import type { Config } from '../config/schema.ts';
+import type { HashResolver } from '../transform/hash-resolver.ts';
+import { chScopeOf } from '../transform/hash-resolver.ts';
import { StagingManager } from '../target/staging-manager.ts';
import { discoverCollections } from '../source/discover-collections.ts';
+export interface DedupeUnsafeBucket {
+ fromMs: number;
+ toMs: number;
+ matched: number;
+ native: number;
+}
+
+export interface DedupeCollectionRow {
+ collection: string;
+ /** Scoped (a,e,n) live counts — exact safety evidence. Unscoped rows use table-wide counts (weaker). */
+ scoped: boolean;
+ mongoDocsInWindow: number;
+ chMatched: number;
+ deleted: number;
+ /** Buckets whose migrated rows lack count-evidence of native counterparts — never deleted. */
+ unsafe: DedupeUnsafeBucket[];
+}
+
export interface DedupeOverlapState {
status: 'not_run' | 'running' | 'completed' | 'failed';
phase: string;
execute: boolean;
fromMs: number | null;
toMs: number | null;
- collections: Array<{ collection: string; mongoDocsInWindow: number; chMatched: number; deleted: number }>;
- totals: { mongoDocsInWindow: number; chMatched: number; deleted: number };
+ collections: DedupeCollectionRow[];
+ totals: { mongoDocsInWindow: number; chMatched: number; deleted: number; unsafeMatched: number };
/** Window of the last COMPLETED dry run — the license to execute. */
lastDryRun: { fromMs: number; toMs: number; chMatched: number; at: number } | null;
error: string | null;
@@ -44,19 +73,22 @@ export interface DedupeOverlapState {
export function newDedupeOverlapState(): DedupeOverlapState {
return {
status: 'not_run', phase: '', execute: false, fromMs: null, toMs: null,
- collections: [], totals: { mongoDocsInWindow: 0, chMatched: 0, deleted: 0 },
+ collections: [], totals: { mongoDocsInWindow: 0, chMatched: 0, deleted: 0, unsafeMatched: 0 },
lastDryRun: null, error: null, startedAt: null, finishedAt: null,
};
}
-const ID_BATCH = 200_000;
+const ID_BATCH = 50_000;
+const BUCKET_MS = 3_600_000;
+/** Hard ceiling on one bucket's ids held in memory — pick a smaller window if hit. */
+const MAX_BUCKET_IDS = 3_000_000;
export async function runDedupeOverlap(
- deps: { config: Config; logger: Logger },
+ deps: { config: Config; logger: Logger; hashResolver: HashResolver },
state: DedupeOverlapState,
opts: { fromMs: number; toMs: number; execute: boolean },
): Promise {
- const { config } = deps;
+ const { config, hashResolver } = deps;
const logger = deps.logger.child({ component: 'DedupeOverlap' });
const lastDry = state.lastDryRun;
@@ -88,33 +120,65 @@ export async function runDedupeOverlap(
for (const collection of collections) {
state.phase = `scanning ${collection}`;
const coll = db.collection(collection);
- const row = { collection, mongoDocsInWindow: 0, chMatched: 0, deleted: 0 };
+ const defaults = hashResolver.resolveCollectionName(collection, config.source.collectionPrefix);
+ const scope = defaults ? chScopeOf(defaults) : null;
+ const row: DedupeCollectionRow = { collection, scoped: !!scope, mongoDocsInWindow: 0, chMatched: 0, deleted: 0, unsafe: [] };
- // Old-Mongo ids in the window = the mirror's re-ingested docs — the
- // exact set whose migrated copies are duplicates.
- let batch: string[] = [];
- const flush = async (): Promise => {
- if (batch.length === 0) return;
- const matched = await staging.countMatchingIdsInWindow(batch, opts.fromMs, opts.toMs);
+ const processBucket = async (ids: string[], loMs: number, hiMs: number): Promise => {
+ if (ids.length === 0) return;
+ let matched = 0;
+ for (let i = 0; i < ids.length; i += ID_BATCH) {
+ matched += await staging.countMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs);
+ }
row.chMatched += matched;
- if (opts.execute && matched > 0) {
- await staging.deleteMatchingIdsInWindow(batch, opts.fromMs, opts.toMs);
+ state.totals.chMatched += matched;
+ if (matched === 0) return;
+ // Count evidence of native counterparts: what remains in this bucket
+ // after the matched rows is the native side. Falling short means some
+ // migrated rows are the ONLY copy of their event — never delete those.
+ const liveTotal = await staging.countLiveInCdRange(loMs, hiMs, scope);
+ const native = liveTotal - matched;
+ // slack absorbs ingest-timing straddle at bucket edges, but a bucket
+ // with NO native rows at all is the outage signature outright — the
+ // slack floor must never wave those through
+ const slack = Math.max(10, Math.ceil(matched * 0.02));
+ if (native < matched - slack || native <= 0) {
+ row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native });
+ state.totals.unsafeMatched += matched;
+ return;
+ }
+ if (opts.execute) {
+ for (let i = 0; i < ids.length; i += ID_BATCH) {
+ await staging.deleteMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs);
+ }
row.deleted += matched;
+ state.totals.deleted += matched;
}
- batch = [];
};
- const cursor = coll.find({ cd: { $gte: from, $lt: to } }, { projection: { _id: 1 } }).batchSize(10_000);
+
+ // Old-Mongo ids in the window (cd order → contiguous hour buckets)
+ let bucketStart = -1;
+ let ids: string[] = [];
+ const cursor = coll.find({ cd: { $gte: from, $lt: to } }, { projection: { _id: 1, cd: 1 } })
+ .sort({ cd: 1 }).batchSize(10_000);
for await (const doc of cursor) {
row.mongoDocsInWindow++;
- batch.push(String(doc._id));
- if (batch.length >= ID_BATCH) await flush();
+ state.totals.mongoDocsInWindow++;
+ const cdMs = (doc.cd as Date).getTime();
+ const bucket = Math.floor(cdMs / BUCKET_MS) * BUCKET_MS;
+ if (bucket !== bucketStart) {
+ await processBucket(ids, Math.max(bucketStart, opts.fromMs), Math.min(bucketStart + BUCKET_MS, opts.toMs));
+ bucketStart = bucket;
+ ids = [];
+ }
+ ids.push(String(doc._id));
+ if (ids.length > MAX_BUCKET_IDS) {
+ throw new Error(`${collection}: more than ${MAX_BUCKET_IDS.toLocaleString('en-US')} docs in one hour bucket — run the dedupe over a smaller {fromMs, toMs} window`);
+ }
}
- await flush();
+ await processBucket(ids, Math.max(bucketStart, opts.fromMs), Math.min(bucketStart + BUCKET_MS, opts.toMs));
if (row.mongoDocsInWindow > 0 || row.chMatched > 0) state.collections.push(row);
- state.totals.mongoDocsInWindow += row.mongoDocsInWindow;
- state.totals.chMatched += row.chMatched;
- state.totals.deleted += row.deleted;
}
state.status = 'completed';
@@ -125,7 +189,11 @@ export async function runDedupeOverlap(
}
logger.info(
{ execute: opts.execute, ...state.totals, collections: state.collections.length },
- opts.execute ? 'Tee-overlap duplicates deleted' : 'Tee-overlap dedupe dry run complete — nothing deleted',
+ opts.execute
+ ? (state.totals.unsafeMatched > 0
+ ? 'Tee-overlap duplicates deleted — SOME BUCKETS SKIPPED: migrated rows there lack native counterparts (see unsafe buckets)'
+ : 'Tee-overlap duplicates deleted')
+ : 'Tee-overlap dedupe dry run complete — nothing deleted',
);
} catch (err) {
state.status = 'failed';
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 182847d..16932e2 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -58,7 +58,7 @@ const fmt = (n: number): string => n.toLocaleString('en-US');
const iso = (ms: number): string => new Date(ms).toISOString().slice(0, 16).replace('T', ' ') + ' UTC';
interface ContentAuditRunner {
- contentAudit(samplesPerCollection?: number): Promise<{
+ contentAudit(samplesPerCollection?: number, upToMs?: number | null): Promise<{
sampled: number; matched: number; missing: number; different: number;
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}>;
@@ -116,7 +116,10 @@ export async function runFinalCheck(
if (dlqPending > 0) {
const top = await dlq.topErrors(runId, 3).catch(() => []);
const reasons = top.map((t) => `${t.error} ×${fmt(t.n)}`).join(', ');
- out.notes.push(`${fmt(dlqPending)} skipped docs wait in the DLQ (${reasons}) — they are NOT in ClickHouse. Review a few in the DLQ panel, then Waive them (accepted as unmigratable) or Replay after a fix. Sign-off is complete once the DLQ shows 0 pending.`);
+ // unresolved = undecided: these docs are NOT in ClickHouse and nobody
+ // has accepted that yet — a sign-off cannot authorize teardown over
+ // an open decision, so this is a problem, not a note
+ out.problems.push(`${fmt(dlqPending)} skipped docs wait UNRESOLVED in the DLQ (${reasons}) — they are NOT in ClickHouse. Review a few in the DLQ panel, then Waive them (accepted as unmigratable) or Replay after a fix, and run this check again.`);
}
if (dlqWaived > 0) {
out.notes.push(`${fmt(dlqWaived)} docs were waived earlier — deliberately accepted as not migrated (their raw copies stay in the DLQ collection as the record).`);
@@ -129,6 +132,20 @@ export async function runFinalCheck(
out.audit = audit;
await rebuildLedger({ config, logger, ledger, dlq, hashResolver, progress: audit, checkOnly: true, upToMs: cutoverMs });
const windows = audit.summary.reduce((a, s) => a + s.chunks, 0);
+ // a window with live === 0 is classified 'pending' by the audit, not
+ // mismatched — on a run that claims completion it means the WHOLE
+ // window is missing from the target (e.g. rows removed after attach)
+ const scopedPendingWindows = audit.summary.filter((s) => s.scoped || audit.summary.length === 1)
+ .reduce((a, s) => a + s.pending, 0);
+ const unscopedWindows = audit.summary.length > 1
+ ? audit.summary.filter((s) => !s.scoped).reduce((a, s) => a + s.chunks, 0)
+ : 0;
+ if (scopedPendingWindows > 0) {
+ out.problems.push(`${fmt(scopedPendingWindows)} window(s) hold ZERO rows in ClickHouse for data the source has — whole windows are missing from the target. Rebuild the ledger from data, Retry failed chunks, and run this check again; do NOT decommission the old cluster.`);
+ }
+ if (unscopedWindows > 0) {
+ out.notes.push(`${fmt(unscopedWindows)} window(s) belong to collection(s) without their own (a,e,n) scope and cannot be recounted against the source individually — for those, trust rests on the per-chunk verify at attach time plus the content samples below.`);
+ }
if (audit.mismatchedWindows.length > 0) {
out.problems.push(`${fmt(audit.mismatchedWindows.length)} window(s) hold FEWER docs in ClickHouse than the source — data is missing from the target. Click "Retry failed chunks" after a rebuild, or escalate; do NOT decommission the old cluster.`);
}
@@ -138,7 +155,7 @@ export async function runFinalCheck(
if (audit.deletionDriftWindows.length > 0) {
out.notes.push(`${fmt(audit.deletionDriftWindows.length)} window(s) now hold MORE docs in ClickHouse than the source — the source shrank after migration (retention TTL / deletions). Expected on deployments with retention; the migrated copy is the complete one.`);
}
- if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0) {
+ if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0 && scopedPendingWindows === 0) {
out.passes.push(`Recounted ${fmt(windows)} window(s) directly against the source: every count matches, every checksum fingerprint matches.`);
}
if (cutoverMs !== null) {
@@ -148,7 +165,7 @@ export async function runFinalCheck(
// ── 4. Sampled content comparison ──────────────────────────────────────
out.phase = 'comparing sampled documents field-by-field';
- const content = await deps.orchestrator.contentAudit(opts.samples);
+ const content = await deps.orchestrator.contentAudit(opts.samples, cutoverMs);
out.content = { sampled: content.sampled, matched: content.matched, missing: content.missing, different: content.different };
if (content.missing > 0 || content.different > 0) {
out.problems.push(`Content sampling found ${fmt(content.missing)} missing and ${fmt(content.different)} differing doc(s) out of ${fmt(content.sampled)} sampled — the migrated content does not match the source; escalate before decommissioning.`);
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 50f17d4..fd8f8ea 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -418,7 +418,7 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise dedupeState);
diff --git a/tests/integration/cd-upper-bound.test.ts b/tests/integration/cd-upper-bound.test.ts
index ac54262..a3c12ec 100644
--- a/tests/integration/cd-upper-bound.test.ts
+++ b/tests/integration/cd-upper-bound.test.ts
@@ -205,6 +205,15 @@ describe('cd upper bound (tee-mirror duplication guard)', () => {
// second run (which appended nothing) did not disturb it
expect(await ledger.sumEstimates(RUN)).toBe(HIST);
+ // content audit: unclamped sampling reaches the post-flip old-side docs
+ // (never migrated by design) and reports phantom "missing" rows; the
+ // clamped call must stay clean — this is what the Final check passes in
+ const unclamped = await orchestrator.contentAudit(100);
+ expect(unclamped.missing).toBeGreaterThan(0);
+ const clamped = await orchestrator.contentAudit(100, BOUND);
+ expect(clamped.missing).toBe(0);
+ expect(clamped.different).toBe(0);
+
// verify + source audit stay green with a growing source: post-bound
// windows derived from the source show live=0 (pending), never defects
const verify = await orchestrator.verifyMigration();
diff --git a/tests/integration/dedupe-overlap.test.ts b/tests/integration/dedupe-overlap.test.ts
index 0214636..80a1f3a 100644
--- a/tests/integration/dedupe-overlap.test.ts
+++ b/tests/integration/dedupe-overlap.test.ts
@@ -5,6 +5,9 @@
* - dry run counts the duplicates exactly and deletes NOTHING
* - execute deletes precisely the id-matched rows inside the window:
* native rows and pre-window (legitimately migrated) rows survive
+ * - a bucket whose migrated rows lack count-evidence of native
+ * counterparts (tee outage: the migrated row is the ONLY copy) is
+ * skipped, reported, and NEVER deleted — even under execute
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import pino from 'pino';
@@ -13,6 +16,7 @@ import { MongoClient } from 'mongodb';
import { createClient, type ClickHouseClient } from '@clickhouse/client';
import { runDedupeOverlap, newDedupeOverlapState } from '../../src/runtime/dedupe-overlap.ts';
+import { HashResolver } from '../../src/transform/hash-resolver.ts';
import { loadConfig } from '../../src/config/loader.ts';
import type { Config } from '../../src/config/schema.ts';
@@ -24,6 +28,11 @@ const logger = pino({ level: 'silent' });
const APP = 'app_dd';
const COLL = `drill_events${createHash('sha1').update('views' + APP).digest('hex')}`;
+// second app: its mirrored docs were migrated but the native side is GONE
+// (tee outage during the overlap) — the safety check must protect them
+const APP2 = 'app_dd_outage';
+const COLL2 = `drill_events${createHash('sha1').update('views' + APP2).digest('hex')}`;
+const OUTAGE = 40;
const FLIP = Math.floor(Date.now() / 60_000) * 60_000 - 2 * 3_600_000; // tee flip 2h ago
const DONE = FLIP + 3_600_000; // migration completed 1h later
@@ -39,6 +48,7 @@ describe('tee-overlap dedupe', () => {
let ch: ClickHouseClient;
let mc: MongoClient;
let config: Config;
+ let hashResolver: HashResolver;
const chCount = async (where = '1'): Promise => {
const res = await ch.query({ query: `SELECT count() AS n FROM ${DB}.drill_events WHERE ${where}`, format: 'JSONEachRow' });
@@ -49,6 +59,11 @@ describe('tee-overlap dedupe', () => {
mc = new MongoClient(MONGO_URI);
await mc.connect();
await mc.db(DB).dropDatabase();
+ await mc.db(`${DB}_countly`).dropDatabase();
+ await mc.db(`${DB}_countly`).collection('apps').insertMany([{ _id: APP }, { _id: APP2 }] as never[]);
+ await mc.db(`${DB}_countly`).collection('events').insertMany([
+ { _id: APP, list: ['views'] }, { _id: APP2, list: ['views'] },
+ ] as never[]);
ch = createClient({ url: CH_URL, password: CH_PASSWORD });
await ch.command({ query: `CREATE DATABASE IF NOT EXISTS ${DB}` });
@@ -86,6 +101,21 @@ describe('tee-overlap dedupe', () => {
await mc.db(DB).collection(COLL).createIndex({ cd: 1, _id: 1 });
await ch.insert({ table: `${DB}.drill_events`, values: chRows, format: 'JSONEachRow' });
+ // outage app: mirrored docs migrated, native side never landed —
+ // deleting these would remove the only copy
+ const outageDocs: Record[] = [];
+ const outageRows: Record[] = [];
+ for (let i = 0; i < OUTAGE; i++) {
+ const cd = FLIP + i * 10_000;
+ outageDocs.push({ _id: `only_${i}`, uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ const r = chRow(`only_${i}`, cd);
+ (r as Record).a = APP2;
+ outageRows.push(r);
+ }
+ await mc.db(DB).collection(COLL2).insertMany(outageDocs as never[]);
+ await mc.db(DB).collection(COLL2).createIndex({ cd: 1, _id: 1 });
+ await ch.insert({ table: `${DB}.drill_events`, values: outageRows, format: 'JSONEachRow' });
+
Object.assign(process.env, {
SERVICE_NAME: 'dedupe-test',
MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`, MANIFEST_DB: DB,
@@ -93,31 +123,40 @@ describe('tee-overlap dedupe', () => {
LEDGER_RUN_ID: 'dedupe-1', BACKPRESSURE_ENABLED: 'false', MULTI_POD_ENABLED: 'false',
});
config = loadConfig();
+ hashResolver = new HashResolver({ uri: MONGO_URI, countlyDb: `${DB}_countly` }, logger);
+ await hashResolver.build();
}, 120_000);
afterAll(async () => {
+ await hashResolver?.close().catch(() => {});
await ch.command({ query: `DROP DATABASE IF EXISTS ${DB}` }).catch(() => {});
await ch.close();
await mc.db(DB).dropDatabase().catch(() => {});
+ await mc.db(`${DB}_countly`).dropDatabase().catch(() => {});
await mc.close();
});
- it('dry run counts the duplicates exactly and deletes nothing', async () => {
+ it('dry run counts the duplicates exactly, flags the outage buckets, and deletes nothing', async () => {
const state = newDedupeOverlapState();
- await runDedupeOverlap({ config, logger }, state, { fromMs: FLIP, toMs: DONE, execute: false });
+ await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: false });
expect(state.status).toBe('completed');
- expect(state.totals).toEqual({ mongoDocsInWindow: 150, chMatched: 150, deleted: 0 });
- expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 });
- expect(await chCount()).toBe(200 + 150 + 150 + 10);
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE, chMatched: 150 + OUTAGE, deleted: 0, unsafeMatched: OUTAGE });
+ expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 + OUTAGE });
+ const outageRow = state.collections.find((c) => c.collection === COLL2);
+ expect(outageRow?.unsafe.length).toBeGreaterThan(0);
+ expect(outageRow?.unsafe.reduce((a, u) => a + u.matched, 0)).toBe(OUTAGE);
+ expect(await chCount()).toBe(200 + 150 + 150 + 10 + OUTAGE);
});
- it('execute deletes exactly the migrated copies; native and pre-flip rows survive', async () => {
+ it('execute deletes exactly the evidenced duplicates; unsafe buckets, native and pre-flip rows survive', async () => {
const state = newDedupeOverlapState();
- await runDedupeOverlap({ config, logger }, state, { fromMs: FLIP, toMs: DONE, execute: true });
+ await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: true });
expect(state.status).toBe('completed');
- expect(state.totals).toEqual({ mongoDocsInWindow: 150, chMatched: 150, deleted: 150 });
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE, chMatched: 150 + OUTAGE, deleted: 150, unsafeMatched: OUTAGE });
expect(await chCount("_id LIKE 'mirror_%'")).toBe(0);
expect(await chCount("_id LIKE 'native_%'")).toBe(160);
expect(await chCount("_id LIKE 'hist_%'")).toBe(200);
+ // the only-copy rows are untouched — the safety check protected them
+ expect(await chCount("_id LIKE 'only_%'")).toBe(OUTAGE);
});
});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index a0c42b9..22d97ef 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -166,7 +166,7 @@ describe('final check: the interpreted sign-off', () => {
expect(out.problems.length).toBeGreaterThan(0);
});
- it('pending DLQ docs → action note, and their window is NOT flagged (unresolved accounting)', async () => {
+ it('pending DLQ docs → FAIL (undecided = no sign-off); waiving turns it into a note', async () => {
// one doc the run skipped: present in Mongo, absent in CH, recorded in DLQ
await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_10'` });
await dlq.add([{
@@ -175,13 +175,15 @@ describe('final check: the interpreted sign-off', () => {
transform_version: config.transform.version, cd_ms: START + 10 * 12_000,
}]);
const out = await check({ cutoverMs: CUTOVER });
- expect(out.verdict).toBe('PASS_WITH_NOTES');
- expect(out.problems).toEqual([]);
- expect(out.notes.join(' ')).toContain('DLQ');
- // waive → the note softens to the waived form
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('UNRESOLVED');
+ // …but the DLQ'd doc's window is NOT double-flagged (unresolved accounting)
+ expect(out.audit?.mismatchedWindows).toEqual([]);
+ // waive = the decision was made → note, sign-off possible
await dlq.waive(RUN);
const out2 = await check({ cutoverMs: CUTOVER });
expect(out2.verdict).toBe('PASS_WITH_NOTES');
+ expect(out2.problems).toEqual([]);
expect(out2.notes.join(' ')).toContain('waived');
});
@@ -215,4 +217,12 @@ describe('final check: the interpreted sign-off', () => {
expect(out2.problems.join(' ')).toContain('Retry failed chunks');
await mc.db(DB).collection('mig_ranges').updateOne({ _id: `${RUN}:${COLL}:0` } as never, { $set: { status: 'done' } });
});
+
+ it('a WHOLE window missing from the target → FAIL (the audit calls it pending, the check must not)', async () => {
+ // stale ledger says done, but every row of the window is gone from CH
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id LIKE 'm\\_%'` });
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('ZERO rows');
+ });
});
From 712cb2dd08873873f7e5bd2db4143289bcb9006e Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:38:21 +0300
Subject: [PATCH 06/64] fix(ledger): harden guard evidence, strict dedupe
gates, corroborated auto-apply, drift spot-check
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Second review round — all findings addressed:
- boundary guard: probe failures now HOLD instead of proceeding (absence of
evidence is not evidence of a mirror-free topology); liveness lookback
widened from 30 min to 24 h so quiet spells on low-volume targets cannot
slip a mirrored setup past the probe; the guard loop re-evaluates all
evidence, so it also releases when the stores come back and say 'quiet'.
- dedupe + final check: refuse while ANY pod (the serving one included)
holds an active chunk claim — dry runs included; the migration must be
fully stopped or complete before either runs.
- dedupe: strict count evidence by default (zero slack — one uncovered
matched row marks the bucket unsafe); operator-chosen slackPct (≤5%) for
edge straddle; documented limit: losses exactly offset by mirror-dropped
natives in the same hour are invisible to counting.
- set-boundary auto-apply: a gap needs corroborating volume on both flanks
(≥25 docs in the 10 min before and after) — a quiet minute on a sparse
install is never taken as the seam unattended.
- source audit: drift windows (live > source) get an id spot-check (≤5k
sampled source ids per window, ≤50 windows) — surplus retained rows can
mask missing current docs, and the Final check now FAILs on that instead
of calling the migrated copy complete; clamped audits size their window
grid from the clamped population.
- endpoints: epoch-ms validation on cutoverMs/fromMs/toMs (epoch-seconds
mistake named outright; future timestamps refused; cutover below the
stored bound refused).
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 8 ++++-
src/runtime/boundary-detector.ts | 18 ++++++++++
src/runtime/chunk-orchestrator.ts | 42 ++++++++++++++---------
src/runtime/dedupe-overlap.ts | 19 ++++++----
src/runtime/final-check.ts | 8 +++--
src/runtime/ledger-engine.ts | 42 +++++++++++++++++++----
src/runtime/ledger-rebuild.ts | 36 ++++++++++++++++---
tests/integration/boundary-detect.test.ts | 25 ++++++++++++--
tests/integration/final-check.test.ts | 15 ++++++++
9 files changed, 173 insertions(+), 40 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index a217bb5..1f0158c 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -95,7 +95,13 @@ that event. Every hour bucket therefore needs count-evidence of native
counterparts — `native = live − matched` must roughly cover `matched` —
before anything in it is deleted. Buckets that fall short are skipped and
reported (`unsafe` in the result); review those hours (tee outage? wrong
-start time?) instead of forcing them.
+start time?) instead of forcing them. The check is strict (zero slack) by
+default; `slackPct` (≤5) may be passed consciously to absorb ingest-timing
+straddle at bucket edges. Known limit: a loss exactly offset by
+mirror-dropped natives in the same hour is invisible to count evidence —
+an EMPTY dry run means no duplicates (skip the step; never widen the window
+to make it match something). Both dedupe (dry run included) and the Final
+check refuse while any pod still holds an active chunk claim.
There is a dashboard card for this (Overview → **Tee-overlap dedupe**:
enter the window, *Dry run* first — *Delete duplicates* unlocks only after
diff --git a/src/runtime/boundary-detector.ts b/src/runtime/boundary-detector.ts
index 06816d2..c28dbf6 100644
--- a/src/runtime/boundary-detector.ts
+++ b/src/runtime/boundary-detector.ts
@@ -64,6 +64,24 @@ export function decideAutoApply(
reason: `detected an ANCHOR, not an exact gap — ${d.ambiguousMongoDocs ?? '?'} old-side docs sit inside the ambiguity band. Review GET /api/boundary, then re-call with {"acceptAnchor": true} to take it, or pass an explicit {"boundMs": ...}.`,
};
}
+ // A quiet minute only proves a seam when there was traffic to go quiet
+ // FROM: on low-volume installs every other minute is silent, and the
+ // first lull would be taken as the flip. Require corroborating volume on
+ // both flanks before applying a gap unattended.
+ if (d.method === 'gap' && !acceptAnchor) {
+ const gap = d.gap;
+ const mins = d.minutes ?? [];
+ const FLANK_MS = 10 * 60_000;
+ const MIN_FLANK_DOCS = 25;
+ const before = gap ? mins.filter((m) => m.minuteMs >= gap.fromMs - FLANK_MS && m.minuteMs < gap.fromMs).reduce((a, m) => a + m.mongo, 0) : 0;
+ const after = gap ? mins.filter((m) => m.minuteMs >= gap.toMs && m.minuteMs < gap.toMs + FLANK_MS).reduce((a, m) => a + m.ch, 0) : 0;
+ if (!gap || before < MIN_FLANK_DOCS || after < MIN_FLANK_DOCS) {
+ return {
+ apply: false,
+ reason: `a gap was found but traffic around it is too sparse to trust a quiet minute as the seam (${before} old-side docs in the 10 min before, ${after} new-side docs in the 10 min after — need ${MIN_FLANK_DOCS} each). Review GET /api/boundary, then re-call with {"acceptAnchor": true} or pass an explicit {"boundMs": ...}.`,
+ };
+ }
+ }
return { apply: true, boundMs: d.suggestedBoundMs };
}
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index 41360ed..ebefc4f 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -109,6 +109,8 @@ class ClaimLostError extends Error {
}
const MAX_CHUNK_ATTEMPTS = 3;
+/** Liveness lookback for the boundary guard: a full day, so quiet spells on low-volume deployments cannot slip a mirrored target past the probe. */
+const GUARD_LIVE_LOOKBACK_MS = 24 * 3_600_000;
const BISECT_LOG_THRESHOLD = 1;
function shortHash(s: string): string {
@@ -210,27 +212,36 @@ export class ChunkOrchestrator {
private async boundaryGuard(): Promise {
const { config } = this.d;
if (config.ledger.cdUpperBoundMs != null || config.ledger.unboundedOk) return;
- if (await this.d.ledger.getStoredBound(this.runId).catch(() => null)) return;
- if (await this.d.ledger.getUnboundedAck(this.runId).catch(() => false)) return;
- // only a FRESH run: a resumed run already made this decision
- const counts = await this.d.ledger.statusCounts(this.runId).catch(() => null);
- if (counts === null || Object.values(counts).reduce((a, b) => a + b, 0) > 0) return;
- const live = await this.d.staging.hasLiveCdSince(Date.now() - 30 * 60_000).catch(() => false);
- if (!live) return;
+ // 'proceed' | 'hold'. Evidence failures HOLD: absence of evidence is not
+ // evidence of a mirror-free topology — proceeding unbounded on a probe
+ // error is exactly the silent-duplication path this guard closes. The
+ // liveness lookback is a full day so a low-volume deployment's quiet
+ // spells cannot slip a mirrored target past the probe.
+ const evaluate = async (): Promise<'proceed' | 'hold'> => {
+ try {
+ if ((await this.d.ledger.getStoredBound(this.runId)) !== null) return 'proceed';
+ if (await this.d.ledger.getUnboundedAck(this.runId)) return 'proceed';
+ const counts = await this.d.ledger.statusCounts(this.runId);
+ if (Object.values(counts).reduce((a, b) => a + b, 0) > 0) return 'proceed'; // resumed run: decided already
+ const live = await this.d.staging.hasLiveCdSince(Date.now() - GUARD_LIVE_LOOKBACK_MS);
+ return live ? 'hold' : 'proceed';
+ } catch (err) {
+ this.logger.warn({ err: (err as Error).message }, 'Boundary guard: evidence probe failed — holding until the stores answer');
+ return 'hold';
+ }
+ };
+
+ if ((await evaluate()) === 'proceed') return;
this.pause('boundary-unset');
this.logger.warn(
{ runId: this.runId },
- 'GUARD: target ClickHouse is receiving live data and no cd upper bound is set — if a mirror re-ingests the same requests on both sides, running unbounded WILL duplicate the overlap window. Apply a bound (POST /control/set-boundary) or declare no-mirror (POST /control/allow-unbounded).',
+ 'GUARD: target ClickHouse holds recent live data and no cd upper bound is set — if a mirror re-ingests the same requests on both sides, running unbounded WILL duplicate the overlap window. Apply a bound (POST /control/set-boundary) or declare no-mirror (POST /control/allow-unbounded).',
);
while (!this.stopping) {
- if ((await this.d.ledger.getStoredBound(this.runId).catch(() => null)) !== null) {
- this.logger.info({ runId: this.runId }, 'Boundary guard released: a cd bound was applied');
- this.resume();
- return;
- }
- if (await this.d.ledger.getUnboundedAck(this.runId).catch(() => false)) {
- this.logger.warn({ runId: this.runId }, 'Boundary guard released: operator declared no-mirror — running unbounded');
+ await sleep(3_000);
+ if ((await evaluate()) === 'proceed') {
+ this.logger.warn({ runId: this.runId }, 'Boundary guard released — a bound was applied, no-mirror was declared, or the run already has mapped state');
this.resume();
return;
}
@@ -239,7 +250,6 @@ export class ChunkOrchestrator {
this.pause('boundary-unset');
this.logger.warn('Resume ignored while the boundary question is open — apply a bound or POST /control/allow-unbounded');
}
- await sleep(3_000);
}
}
diff --git a/src/runtime/dedupe-overlap.ts b/src/runtime/dedupe-overlap.ts
index e5f5fb2..3435916 100644
--- a/src/runtime/dedupe-overlap.ts
+++ b/src/runtime/dedupe-overlap.ts
@@ -21,7 +21,13 @@
* In a healthy tee every matched row duplicates a native one, so native is
* at least matched (plus mirror losses only ever shrink matched). A bucket
* where native falls short holds migrated rows WITHOUT counterparts —
- * those are skipped, reported, and never deleted.
+ * those are skipped, reported, and never deleted. Strictness is default:
+ * ZERO slack, so even one uncovered matched row marks the bucket unsafe;
+ * ingest-timing straddle at bucket edges can flag a few healthy buckets,
+ * and the operator may consciously allow it with slackPct (≤5%). Known
+ * limit of count evidence: a loss exactly offset by mirror-dropped natives
+ * in the SAME hour is invisible — which is why unsafe hours must be taken
+ * seriously, not overridden casually.
*
* Safety: dry-run by default (counts only); execute is refused until a dry
* run over the SAME window has completed in this process, and the old
@@ -86,7 +92,7 @@ const MAX_BUCKET_IDS = 3_000_000;
export async function runDedupeOverlap(
deps: { config: Config; logger: Logger; hashResolver: HashResolver },
state: DedupeOverlapState,
- opts: { fromMs: number; toMs: number; execute: boolean },
+ opts: { fromMs: number; toMs: number; execute: boolean; slackPct?: number },
): Promise {
const { config, hashResolver } = deps;
const logger = deps.logger.child({ component: 'DedupeOverlap' });
@@ -138,10 +144,11 @@ export async function runDedupeOverlap(
// migrated rows are the ONLY copy of their event — never delete those.
const liveTotal = await staging.countLiveInCdRange(loMs, hiMs, scope);
const native = liveTotal - matched;
- // slack absorbs ingest-timing straddle at bucket edges, but a bucket
- // with NO native rows at all is the outage signature outright — the
- // slack floor must never wave those through
- const slack = Math.max(10, Math.ceil(matched * 0.02));
+ // strict by default: every matched row needs a native counterpart in
+ // its bucket. slackPct (operator-chosen, ≤5%) only absorbs
+ // ingest-timing straddle at bucket edges; zero natives is the outage
+ // signature outright and no slack ever waves it through
+ const slack = Math.ceil(matched * (Math.min(5, Math.max(0, opts.slackPct ?? 0)) / 100));
if (native < matched - slack || native <= 0) {
row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native });
state.totals.unsafeMatched += matched;
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 16932e2..195c308 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -152,8 +152,12 @@ export async function runFinalCheck(
if (audit.checksumMismatchWindows.length > 0) {
out.problems.push(`${fmt(audit.checksumMismatchWindows.length)} window(s) hold the right COUNT of the WRONG documents (checksum fingerprint differs) — escalate; do NOT decommission the old cluster.`);
}
- if (audit.deletionDriftWindows.length > 0) {
- out.notes.push(`${fmt(audit.deletionDriftWindows.length)} window(s) now hold MORE docs in ClickHouse than the source — the source shrank after migration (retention TTL / deletions). Expected on deployments with retention; the migrated copy is the complete one.`);
+ if ((audit.driftSubsetMissing ?? []).length > 0) {
+ const missingN = (audit.driftSubsetMissing ?? []).reduce((a, w) => a + w.missing, 0);
+ out.problems.push(`${fmt((audit.driftSubsetMissing ?? []).length)} retention-drift window(s) are MISSING current source docs behind their surplus counts (${fmt(missingN)} sampled ids not found live) — surplus rows were masking gaps; do NOT decommission the old cluster.`);
+ }
+ if (audit.deletionDriftWindows.length > 0 && (audit.driftSubsetMissing ?? []).length === 0) {
+ out.notes.push(`${fmt(audit.deletionDriftWindows.length)} window(s) now hold MORE docs in ClickHouse than the source — the source shrank after migration (retention TTL / deletions). Sampled source ids in those windows were all found live, so the surplus is retained history, not masked gaps.`);
}
if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0 && scopedPendingWindows === 0) {
out.passes.push(`Recounted ${fmt(windows)} window(s) directly against the source: every count matches, every checksum fingerprint matches.`);
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index fd8f8ea..4a47360 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -383,13 +383,32 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise {
+ if (typeof v !== 'number' || !Number.isFinite(v)) return `${name} (epoch ms) required`;
+ if (v < 1_000_000_000_000) return `${name}=${v} looks like epoch SECONDS — pass milliseconds (×1000)`;
+ if (v > Date.now() + 60_000) return `${name} is in the future`;
+ return null;
+ };
const finalCheckState: FinalCheckResult = newFinalCheckResult();
app.post<{ Body: { cutoverMs?: number; samples?: number } }>('/control/final-check', async (req) => {
if (finalCheckState.status === 'running') return { started: false, reason: 'final check already running' };
if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — run the final check after completion (or while paused)' };
- const busyFc = await ledger.activeClaims(config.ledger.runId, config.worker.podId);
- if (busyFc.length > 0) return { started: false, reason: `other pods are actively migrating (${busyFc.map((row) => row.pod).join(', ')}) — run the final check after completion` };
- const cutoverMs = typeof req.body?.cutoverMs === 'number' && Number.isFinite(req.body.cutoverMs) ? req.body.cutoverMs : null;
+ // no exclusion: the SERVING pod's own live claims block the check too —
+ // a paused pod mid-chunk still owns half-written state
+ const busyFc = await ledger.activeClaims(config.ledger.runId);
+ if (busyFc.length > 0) return { started: false, reason: `pods still hold active chunk claims (${busyFc.map((row) => `${row.pod}×${row.count}`).join(', ')}) — the migration must be fully stopped/complete before the final check` };
+ let cutoverMs: number | null = null;
+ if (req.body?.cutoverMs !== undefined) {
+ const err = epochMsError(req.body.cutoverMs, 'cutoverMs');
+ if (err) return { started: false, reason: err };
+ cutoverMs = req.body.cutoverMs as number;
+ const storedFc = await ledger.getStoredBound(config.ledger.runId).catch(() => null);
+ if (storedFc !== null && cutoverMs < storedFc) {
+ return { started: false, reason: `cutoverMs is EARLIER than the run's stored bound (${new Date(storedFc).toISOString()}) — that would silently exclude migrated data from the audit; pass the bound or later` };
+ }
+ }
const samples = Math.min(10_000, Math.max(50, req.body?.samples ?? 500));
void runFinalCheck({ config, logger, ledger, dlq, hashResolver, orchestrator }, finalCheckState, { cutoverMs, samples });
return { started: true, cutoverMs, samples };
@@ -403,12 +422,20 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise('/control/dedupe-overlap', async (req) => {
+ app.post<{ Body: { fromMs?: number; toMs?: number; execute?: boolean; slackPct?: number } }>('/control/dedupe-overlap', async (req) => {
if (dedupeState.status === 'running') return { started: false, reason: 'dedupe already running' };
- if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — dedupe only applies after completion' };
+ // destructive against the live table: the migration must be fully
+ // stopped — no pod (this one included) may hold an active chunk claim,
+ // dry run included, so the counts it licenses execute with are stable
+ if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — dedupe (even a dry run) requires the migration stopped or complete' };
+ const busyDd = await ledger.activeClaims(config.ledger.runId);
+ if (busyDd.length > 0) return { started: false, reason: `pods still hold active chunk claims (${busyDd.map((row) => `${row.pod}×${row.count}`).join(', ')}) — stop the migration everywhere before dedupe, even for a dry run` };
const fromMs = req.body?.fromMs;
const toMs = req.body?.toMs;
- if (typeof fromMs !== 'number' || typeof toMs !== 'number' || !(fromMs < toMs)) {
+ const fromErr = epochMsError(fromMs, 'fromMs');
+ const toErr = fromErr ? null : epochMsError(toMs, 'toMs');
+ if (fromErr || toErr) return { started: false, reason: (fromErr ?? toErr) as string };
+ if (!((fromMs as number) < (toMs as number))) {
return { started: false, reason: 'pass the overlap window as {fromMs, toMs} (epoch ms): fromMs = the tee flip / IP swap, toMs = migration completion' };
}
const execute = req.body?.execute === true;
@@ -418,7 +445,8 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise dedupeState);
diff --git a/src/runtime/ledger-rebuild.ts b/src/runtime/ledger-rebuild.ts
index 576cd87..3394a60 100644
--- a/src/runtime/ledger-rebuild.ts
+++ b/src/runtime/ledger-rebuild.ts
@@ -65,6 +65,8 @@ export interface RebuildProgress {
/** When a cutover clamp was applied: the clamp and how many source docs sit beyond it (out of scope). */
cutoverMs?: number | null;
excludedBeyondCutover?: number;
+ /** Drift windows (live > source) whose sampled source ids were NOT all found in the target — surplus rows were masking missing ones. */
+ driftSubsetMissing?: Array<{ collection: string; lowerCd: string; upperCd: string; sampled: number; missing: number }>;
error: string | null;
startedAt: number | null;
finishedAt: number | null;
@@ -73,7 +75,7 @@ export interface RebuildProgress {
export function newRebuildProgress(): RebuildProgress {
return {
status: 'not_run', phase: '', collectionsDone: 0, collectionsTotal: 0,
- summary: [], mismatchedWindows: [], deletionDriftWindows: [], checksumMismatchWindows: [], error: null, startedAt: null, finishedAt: null,
+ summary: [], mismatchedWindows: [], deletionDriftWindows: [], checksumMismatchWindows: [], driftSubsetMissing: [], error: null, startedAt: null, finishedAt: null,
};
}
@@ -126,6 +128,7 @@ export async function rebuildLedger(opts: {
await staging.connect();
const db = mongo.db(config.source.db);
+ let driftChecks = 0;
progress.phase = 'discovering collections';
let collections = await discoverCollections(db, config.source.collectionPrefix, logger);
const skipEventNames = new Set(['[CLY]_apm_device', '[CLY]_apm_network']);
@@ -183,13 +186,16 @@ export async function rebuildLedger(opts: {
if (lowDoc && highDoc) {
const lowerCd = (lowDoc.cd as Date).getTime();
let upperCd = (highDoc.cd as Date).getTime();
+ let excludedHere = 0;
if (upToMs !== null) {
- progress.excludedBeyondCutover = (progress.excludedBeyondCutover ?? 0)
- + await coll.countDocuments({ cd: { $gte: new Date(upToMs) } });
+ excludedHere = await coll.countDocuments({ cd: { $gte: new Date(upToMs) } });
+ progress.excludedBeyondCutover = (progress.excludedBeyondCutover ?? 0) + excludedHere;
upperCd = Math.min(upperCd, upToMs - 1);
}
if (upperCd >= lowerCd) {
- const estimated = await coll.estimatedDocumentCount();
+ // size the grid from the CLAMPED population — the excluded tail
+ // would otherwise inflate the window count for the remaining span
+ const estimated = Math.max(1, (await coll.estimatedDocumentCount()) - excludedHere);
bounds = computeChunkBounds(lowerCd, upperCd, estimated, config.ledger.chunkDocsTarget, config.ledger.maxChunkDays);
}
}
@@ -235,13 +241,33 @@ export async function rebuildLedger(opts: {
// live > source = the SOURCE shrank after migration (retention
// TTL, GDPR purges) — report as drift, not as a defect; only
// live < source means data is missing from the target.
- const bucket = live + unresolved > mongoCount ? progress.deletionDriftWindows : progress.mismatchedWindows;
+ const isDrift = live + unresolved > mongoCount;
+ const bucket = isDrift ? progress.deletionDriftWindows : progress.mismatchedWindows;
if (bucket.length < 200) {
bucket.push({
collection, lowerCd: new Date(b.lowerCd).toISOString(), upperCd: new Date(b.upperCd).toISOString(),
source: mongoCount, live,
});
}
+ // A surplus count proves nothing about coverage: expired rows the
+ // target kept can MASK current source docs it is missing. Spot-check
+ // drift windows by id — sampled source ids must all exist live.
+ if (isDrift && driftChecks < 50 && mongoCount > 0) {
+ driftChecks++;
+ const sampleIds = (await coll
+ .find({ cd: { $gte: new Date(b.lowerCd), $lt: new Date(b.upperCd) } }, { projection: { _id: 1 } })
+ .limit(5_000).toArray()).map((d) => String(d._id));
+ const present = await staging.countMatchingIdsInWindow(sampleIds, b.lowerCd, b.upperCd);
+ // DLQ'd docs are legitimately absent — only a shortfall beyond
+ // the window's unresolved count is a real coverage gap
+ const missing = Math.max(0, sampleIds.length - present - unresolved);
+ if (missing > 0 && (progress.driftSubsetMissing ?? []).length < 200) {
+ (progress.driftSubsetMissing ?? (progress.driftSubsetMissing = [])).push({
+ collection, lowerCd: new Date(b.lowerCd).toISOString(), upperCd: new Date(b.upperCd).toISOString(),
+ sampled: sampleIds.length, missing,
+ });
+ }
+ }
}
// Checksum: only meaningful on windows that are count-exact with no
// DLQ residue — equal counts hiding DIFFERENT docs is the one error
diff --git a/tests/integration/boundary-detect.test.ts b/tests/integration/boundary-detect.test.ts
index 129b2ff..6f8b355 100644
--- a/tests/integration/boundary-detect.test.ts
+++ b/tests/integration/boundary-detect.test.ts
@@ -284,10 +284,29 @@ describe('tee-boundary detection + sync parity', () => {
describe('set-boundary auto-apply decision', () => {
const report = (detection: Record) => ({ detection, sync: { status: 'ok' } }) as never;
+ const M = 60_000;
+ const gapMinutes = (mongoPerMin: number, chPerMin: number) => {
+ const gap = { fromMs: 20 * M, toMs: 22 * M };
+ const minutes: Array<{ minuteMs: number; mongo: number; ch: number }> = [];
+ for (let m = 5; m < 20; m++) minutes.push({ minuteMs: m * M, mongo: mongoPerMin, ch: 0 });
+ for (let m = 22; m < 40; m++) minutes.push({ minuteMs: m * M, mongo: 0, ch: chPerMin });
+ return { gap, minutes, suggestedBoundMs: 21 * M };
+ };
- it('an exact gap applies unattended', () => {
- expect(decideAutoApply(report({ status: 'ok', method: 'gap', suggestedBoundMs: 123 }), false))
- .toEqual({ apply: true, boundMs: 123 });
+ it('a corroborated gap applies unattended', () => {
+ const g = gapMinutes(5, 4);
+ expect(decideAutoApply(report({ status: 'ok', method: 'gap', ...g }), false))
+ .toEqual({ apply: true, boundMs: 21 * M });
+ });
+
+ it('a quiet minute on a sparse install is NOT taken as the seam', () => {
+ const g = gapMinutes(1, 1); // 10 docs per flank — any lull looks like this
+ const d = decideAutoApply(report({ status: 'ok', method: 'gap', ...g }), false);
+ expect(d.apply).toBe(false);
+ expect(d.reason).toContain('sparse');
+ // …unless the operator explicitly accepts imperfect evidence
+ expect(decideAutoApply(report({ status: 'ok', method: 'gap', ...g }), true))
+ .toEqual({ apply: true, boundMs: 21 * M });
});
it('an anchor needs the explicit acceptAnchor', () => {
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index 22d97ef..abe33f5 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -218,6 +218,21 @@ describe('final check: the interpreted sign-off', () => {
await mc.db(DB).collection('mig_ranges').updateOne({ _id: `${RUN}:${COLL}:0` } as never, { $set: { status: 'done' } });
});
+ it('retention drift with masked missing docs → FAIL from the id spot-check', async () => {
+ // retention deleted 8 source docs (live > source = drift) while one
+ // MIGRATED row also vanished — the surplus count hides it from counting
+ await mc.db(DB).collection(COLL).deleteMany({ _id: { $in: ['m_50', 'm_51', 'm_52', 'm_53', 'm_54', 'm_55', 'm_56', 'm_57'] } } as never);
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_60'` });
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('masking');
+ // clean drift (no masked gaps) stays a note
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('m_60', START + 60 * 12_000)] });
+ const out2 = await check({ cutoverMs: CUTOVER });
+ expect(out2.verdict).toBe('PASS_WITH_NOTES');
+ expect(out2.notes.join(' ')).toContain('retained history');
+ });
+
it('a WHOLE window missing from the target → FAIL (the audit calls it pending, the check must not)', async () => {
// stale ledger says done, but every row of the window is gone from CH
await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id LIKE 'm\\_%'` });
From 91a55abba07709bbff9981aae322b27a7e616e62 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:42:01 +0300
Subject: [PATCH 07/64] =?UTF-8?q?docs(runbook):=20clone-source=20migration?=
=?UTF-8?q?=20variant=20=E2=80=94=20frozen-copy=20semantics,=20dedupe=20de?=
=?UTF-8?q?cision=20by=20T-swap=20vs=20T-clone?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 21 +++++++++++++++++++++
1 file changed, 21 insertions(+)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 1f0158c..9ecc618 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -165,6 +165,27 @@ For 2 and 3: use **Detect boundary** + **Apply this bound to the run**
(one click covers all pods), verify the `bounded · cd < …` badge on every
pod, and keep re-running sync parity during the validation window.
+## Clone-source variant (migrate from a frozen copy)
+
+A deployment may clone the old-arch MongoDB onto the new box and migrate
+from THAT clone while live ingestion moves to the new arch (optionally
+mirroring back to the old stack as the rollback net). Seen in the field;
+properties worth knowing:
+
+- The source is frozen at the clone moment, so no bound is needed and top-up
+ finds nothing — the startup guard will still ask (the target ingests live
+ while the run starts): **Proceed unbounded is correct** here.
+- Parity/audit tables compare against the CLONE: zeros after the clone
+ moment mean "clone taken here", not a dead mirror. The live old-arch Mongo
+ is invisible to the tool.
+- Duplicates exist ONLY if the clone was taken AFTER ingestion switched
+ (its tail then holds mirrored copies of natively-ingested events). Get the
+ two timestamps — T-swap and T-clone. T-clone ≤ T-swap → no duplicates,
+ skip dedupe. T-clone > T-swap → dedupe with exactly [T-swap, T-clone].
+- Any doc-count comparison against the live old-arch Mongo will drift by
+ everything ingested after T-clone — compare against the clone, or scope
+ counts to cd < T-clone.
+
## Tee-mirror cutover (customer keeps the old architecture until sign-off)
For customers who require approval before switching: the old arch stays
From 85c6acac48c80b2befbce4647c7e6941b2ab76a9 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:47:40 +0300
Subject: [PATCH 08/64] docs: make the runbook and README self-contained for
on-premise operators
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The tool ships to third parties running it themselves, so the docs assume
no prior knowledge and no vendor in the room:
- RUNBOOK opens with a Terms table (source/target, cd, chunk/ledger, DLQ,
tee, bound, pod) and speaks to the operator directly — sign-off owner
instead of 'the customer', production run instead of customer run.
- Log-pipeline guidance generalized (any collector of container stdout).
- Clone-source variant rewritten as the recommended recipe: pause old
ingestion → clone → resume on the new stack; a clone taken inside the
pause can never hold a mirrored twin, so no bound and no dedupe at all.
- README points to .env.example as the config reference and to the RUNBOOK
as the from-scratch operations manual; dangling heading removed.
- Dashboard copy: two vendor-voice phrases neutralized.
Co-Authored-By: Claude Fable 5
---
README.md | 15 +++++----
docs/RUNBOOK.md | 64 ++++++++++++++++++++++--------------
src/http/ledger-viz-route.ts | 4 +--
3 files changed, 50 insertions(+), 33 deletions(-)
diff --git a/README.md b/README.md
index eb2fcf7..5462987 100644
--- a/README.md
+++ b/README.md
@@ -64,10 +64,11 @@ both stacks, in the same partition), checks match `(_id, cd)` pairs — the
retry copy's cd can never equal the migrated copy's. Preflight verifies the
boundary is trustworthy (source frozen, clocks sane) before anything runs.
-This README covers what you need BEFORE the dashboard exists (installing,
-env vars, starting the service, automation reference). Everything after —
-running, monitoring, troubleshooting, verifying — lives in the dashboard,
-with `docs/RUNBOOK.md` as the cross-system procedure (cutover choreography,
-Kafka retention, incident tables) for operators.
-
-## Architecture
\ No newline at end of file
+This README covers what you need BEFORE the dashboard exists: installing
+and starting the service. `.env.example` is the commented configuration
+reference (the two required variables and every optional one). Everything
+after — running, monitoring, troubleshooting, verifying — lives in the
+dashboard's **Migration Guide** and **Help & Recovery** tabs, with
+`docs/RUNBOOK.md` as the standalone operations manual (terms, cutover
+scenarios, incident table, sign-off procedure, curl reference) — start
+there if you are planning a migration from scratch.
\ No newline at end of file
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 9ecc618..ec9555e 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -1,17 +1,32 @@
# Migration Runbook
-Operational procedure for migrating a customer's `drill_events` from MongoDB
-to ClickHouse with this service. The guiding property: **after cutover, no
+Operational procedure for migrating a deployment's `drill_events` data from
+MongoDB to ClickHouse with this service. It assumes no prior knowledge of the
+tool — terms are defined below, and every action is available both in the
+dashboard and as a `curl` command. The guiding property: **after cutover, no
failure anywhere in this flow can touch live data** — every incident response
is *restart or resume*, never clean up or restore. Ingestion pauses exactly
once, for minutes, at cutover — never for the migration.
+## Terms used throughout
+
+| Term | Meaning |
+|---|---|
+| **Old cluster / source** | The MongoDB holding the `drill_events*` collections being migrated (sometimes a frozen clone of it — see the clone-source variant). |
+| **New stack / target** | The new Countly architecture whose ClickHouse holds the `drill_events` table this service fills. |
+| **cd** | Each document's server-side creation timestamp. The migration chunks, verifies and audits by cd; migrated rows keep their historical cd, live-ingested rows get post-cutover cds. |
+| **Chunk** | One cd range of one collection — the unit of work, retry and verification. Chunk state lives in `mig_ranges` (the *ledger*) in `MANIFEST_DB`. |
+| **DLQ** | Dead-letter queue (`mig_dlq_docs`): documents that could not or should not be migrated, stored with their full raw source so nothing is silently dropped. |
+| **Tee / mirror** | A reverse-proxy (e.g. nginx) duplicating incoming SDK requests to both stacks; each side re-ingests independently, so the same event gets DIFFERENT `_id`/`cd` on each side. |
+| **Bound** | `LEDGER_CD_UPPER_BOUND`: a cd ceiling — documents at/after it are never migrated. Required exactly when a tee is active (see the scenario table). |
+| **Pod** | One instance of this service. Pods coordinate through chunk leases in MongoDB; any pod's dashboard shows the whole run. |
+
## The flow
-1. **Prepare** (old cluster still live, no customer impact)
+1. **Prepare** (old cluster still live, no user-facing impact)
- Deploy the new stack alongside the old.
- Set Kafka `drill-events` retention to cover the migration window
- (14 days default). Replication factor is the customer's redundancy
+ (14 days default). Replication factor is a redundancy
choice — RF≥2 recommended for large instances; if RF=1, record the
accepted risk (one broker disk loss forfeits the replay guarantee).
- Bulk pre-copy the stateful set: apps & app keys, `app_users`, event
@@ -25,7 +40,7 @@ once, for minutes, at cutover — never for the migration.
3. **Rehearse** — dry run with `DRY_RUN=1` (≤5% stratified sample against a
Null-engine clone; full ClickHouse validation, nothing stored). Review
- `GET /report` (skips, coercions per key, DLQ) with the customer, sign off.
+ `GET /report` (skips, coercions per key, DLQ) with whoever owns sign-off.
4. **Cutover** — stop old ingestion → sync the stateful-set delta since the
pre-copy (changed users via last-seen; aggregated data must land BEFORE
@@ -41,7 +56,7 @@ once, for minutes, at cutover — never for the migration.
live ingestion. Watch `/viz`; the invariant monitor spot-checks
continuously.
-6. **Finish** — all chunks done → final `GET /report` → customer sign-off →
+6. **Finish** — all chunks done → Final check green → sign-off →
revert Kafka retention → decommission old cluster.
## Incident responses
@@ -141,7 +156,7 @@ old per-event collections only after their chunks are done and signed off),
and hard memory limits on the new components — an OOM there is a production
incident.
-## Validation before a customer run
+## Validation before a production run
`bench/README.md`: seed → straight run (counts must be exact) → SIGKILL crash
drill → optionally `bench/seed-failures.ts` for a full failure-scenario drill
@@ -155,7 +170,7 @@ the same requests into both stacks. Everything else is shared machinery.
| # | Topology | LEDGER_CD_UPPER_BOUND | Ingestion switch | New data arriving in old Mongo | Sign-off |
|---|---|---|---|---|---|
| 1 | Two clusters, **no mirroring** (plain switch) | **UNSET** | Before the migration (cutover-first) or after the bulk (bulk-before-cutover + final drain) | **Migrated** — top-up passes chase it until the drain finds nothing | Verify + audits, DLQ = 0 |
-| 2 | Two clusters, **mirror old → new** (old primary) | **SET** = tee flip | At customer sign-off | **Never migrated past the bound** — it is the tee's copy (different _id/cd; duplicates would be undetectable) | Verify + audits for pre-bound; dashboard comparison + sync parity for post-bound |
+| 2 | Two clusters, **mirror old → new** (old primary) | **SET** = tee flip | At sign-off | **Never migrated past the bound** — it is the tee's copy (different _id/cd; duplicates would be undetectable) | Verify + audits for pre-bound; dashboard comparison + sync parity for post-bound |
| 3 | Two clusters, **mirror new → old** (new primary, old = rollback net) | **SET** = the moment new became primary | Already happened at the flip | Same as 2 — post-flip old-side docs are mirror copies | Same as 2 |
| 4 | **Single cluster, in-place upgrade** (drill mongo → ClickHouse in background) | **UNSET** | The upgrade itself is the switch; old drill collections freeze | Transition tail drained by top-up; no tee → nothing to duplicate | Verify + audits, DLQ = 0 (live-parallel path; backpressure protects prod CH) |
@@ -167,28 +182,29 @@ pod, and keep re-running sync parity during the validation window.
## Clone-source variant (migrate from a frozen copy)
-A deployment may clone the old-arch MongoDB onto the new box and migrate
-from THAT clone while live ingestion moves to the new arch (optionally
-mirroring back to the old stack as the rollback net). Seen in the field;
-properties worth knowing:
+A robust pattern: pause old ingestion, clone the source MongoDB onto the
+new machine, then resume ingestion on the NEW stack (optionally mirroring
+back to the old one as the rollback net) and migrate from the clone. SDK
+offline queues absorb the pause. Properties worth knowing:
- The source is frozen at the clone moment, so no bound is needed and top-up
finds nothing — the startup guard will still ask (the target ingests live
while the run starts): **Proceed unbounded is correct** here.
- Parity/audit tables compare against the CLONE: zeros after the clone
- moment mean "clone taken here", not a dead mirror. The live old-arch Mongo
- is invisible to the tool.
-- Duplicates exist ONLY if the clone was taken AFTER ingestion switched
- (its tail then holds mirrored copies of natively-ingested events). Get the
- two timestamps — T-swap and T-clone. T-clone ≤ T-swap → no duplicates,
- skip dedupe. T-clone > T-swap → dedupe with exactly [T-swap, T-clone].
+ moment mean "clone taken here", not a dead mirror. The live old-side
+ MongoDB is invisible to the tool.
+- Cloned INSIDE the ingestion pause (the sequence above) → the clone can
+ never hold a natively-ingested event's mirror copy: **no duplicates, no
+ bound, no dedupe** — the cleanest possible run. Only a clone taken AFTER
+ ingestion resumed has a duplicated tail: dedupe with exactly
+ [ingestion-resume, clone-moment], never earlier.
- Any doc-count comparison against the live old-arch Mongo will drift by
everything ingested after T-clone — compare against the clone, or scope
counts to cd < T-clone.
-## Tee-mirror cutover (customer keeps the old architecture until sign-off)
+## Tee-mirror cutover (keep the old architecture until sign-off)
-For customers who require approval before switching: the old arch stays
+When approval is required before switching: the old stack stays
authoritative, nginx TEES the same SDK requests to the new architecture
(which re-ingests them with its own logic — drill, sessions, aggregations,
profiles all populate natively), and the bulk migration backfills history
@@ -216,7 +232,7 @@ can deduplicate across that seam — the ONLY protection is the time bound.
region; post-bound windows show as pending/uncovered, never as defects.
The post-bound region is the tee's responsibility and is validated by
comparing dashboards between the two systems, not by this tool.
-5. Customer validates side-by-side as long as needed; both systems ingest
+5. Validate side-by-side as long as needed; both systems ingest
the same requests the whole time.
6. On approval: point SDK traffic solely at the new arch, drop the tee,
decommission old ingestion on its own schedule.
@@ -290,7 +306,7 @@ in either the old or the new system.
silently dropped), the waive is recorded and counted, and the source audit
attributes each window's shortfall to its waived docs — sign-off stays
exact. Only consider a sentinel-uid replay instead if the affected volume
-is large enough to distort historical event totals for a customer AND the
+is large enough to distort historical event totals for an app AND the
docs carry usable ts/did (check a few samples in the DLQ panel first).
Chunks that were 100% such docs complete as done (structured skips do not
@@ -315,8 +331,8 @@ failed, DLQ, status + pause reason). No network access needed:
kubectl logs -f deploy/drill-migrator | grep 'progress heartbeat'
docker logs -f drill-migrator-p1 2>&1 | grep 'progress heartbeat'
```
-These lines also flow into the stack's log pipeline (alloy → Loki), so
-Grafana log panels/alerts work with zero extra plumbing.
+Because they go to stdout, they flow into whatever log pipeline collects
+container output (Loki, ELK, CloudWatch, …) with zero extra plumbing.
**3. Actions via curl** (same endpoints the buttons call; POSTs need the
JSON content type):
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index c341519..00a103b 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -447,7 +447,7 @@ const PAGE = `
-
Then: final report (/report), customer sign-off, revert Kafka retention, decommission the old cluster.
+
Then: final report (/report), sign-off, revert Kafka retention, decommission the old cluster.
New cluster is already primary; nginx mirrors back to the old stack as the customer\u2019s rollback safety net during validation.
' +
+ html: '
New cluster is already primary; nginx mirrors back to the old stack as the rollback safety net during validation.
' +
'
' +
'
Everything from scenario 2 applies unchanged \u2014 detection, bound, badge, sync parity. ClickHouse is the store that started cold in both directions, so the detector does not care which side is primary.
' +
'
The bound = the moment the new cluster became primary. Old-cluster docs after it are the mirror\u2019s copies \u2014 never migrate them.
' +
From 17844e492924bf0b12298bbc5ec21a86288c166a Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 12:48:16 +0300
Subject: [PATCH 09/64] docs(runbook): scenario chooser first; clone-source
framed as an optional variant of live-source runs
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 46 ++++++++++++++++++++++++----------------------
1 file changed, 24 insertions(+), 22 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index ec9555e..0702490 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -59,6 +59,24 @@ once, for minutes, at cutover — never for the migration.
6. **Finish** — all chunks done → Final check green → sign-off →
revert Kafka retention → decommission old cluster.
+## Choose your scenario first
+
+The one decision that changes the configuration is whether a TEE mirrors
+the same requests into both stacks. Everything else is shared machinery.
+
+| # | Topology | LEDGER_CD_UPPER_BOUND | Ingestion switch | New data arriving in old Mongo | Sign-off |
+|---|---|---|---|---|---|
+| 1 | Two clusters, **no mirroring** (plain switch) | **UNSET** | Before the migration (cutover-first) or after the bulk (bulk-before-cutover + final drain) | **Migrated** — top-up passes chase it until the drain finds nothing | Verify + audits, DLQ = 0 |
+| 2 | Two clusters, **mirror old → new** (old primary) | **SET** = tee flip | At sign-off | **Never migrated past the bound** — it is the tee's copy (different _id/cd; duplicates would be undetectable) | Verify + audits for pre-bound; dashboard comparison + sync parity for post-bound |
+| 3 | Two clusters, **mirror new → old** (new primary, old = rollback net) | **SET** = the moment new became primary | Already happened at the flip | Same as 2 — post-flip old-side docs are mirror copies | Same as 2 |
+| 4 | **Single cluster, in-place upgrade** (drill mongo → ClickHouse in background) | **UNSET** | The upgrade itself is the switch; old drill collections freeze | Transition tail drained by top-up; no tee → nothing to duplicate | Verify + audits, DLQ = 0 (live-parallel path; backpressure protects prod CH) |
+
+Scenario is also selectable on the dashboard's **Migration Guide** tab —
+it renders the per-scenario checklist and states the bound requirement.
+For 2 and 3: use **Detect boundary** + **Apply this bound to the run**
+(one click covers all pods), verify the `bounded · cd < …` badge on every
+pod, and keep re-running sync parity during the validation window.
+
## Incident responses
| Incident | What happens | Operator action |
@@ -162,30 +180,14 @@ incident.
drill → optionally `bench/seed-failures.ts` for a full failure-scenario drill
(breaker, DLQ, monitor, retry-failed).
-## Choose your scenario first
-
-The one decision that changes the configuration is whether a TEE mirrors
-the same requests into both stacks. Everything else is shared machinery.
-
-| # | Topology | LEDGER_CD_UPPER_BOUND | Ingestion switch | New data arriving in old Mongo | Sign-off |
-|---|---|---|---|---|---|
-| 1 | Two clusters, **no mirroring** (plain switch) | **UNSET** | Before the migration (cutover-first) or after the bulk (bulk-before-cutover + final drain) | **Migrated** — top-up passes chase it until the drain finds nothing | Verify + audits, DLQ = 0 |
-| 2 | Two clusters, **mirror old → new** (old primary) | **SET** = tee flip | At sign-off | **Never migrated past the bound** — it is the tee's copy (different _id/cd; duplicates would be undetectable) | Verify + audits for pre-bound; dashboard comparison + sync parity for post-bound |
-| 3 | Two clusters, **mirror new → old** (new primary, old = rollback net) | **SET** = the moment new became primary | Already happened at the flip | Same as 2 — post-flip old-side docs are mirror copies | Same as 2 |
-| 4 | **Single cluster, in-place upgrade** (drill mongo → ClickHouse in background) | **UNSET** | The upgrade itself is the switch; old drill collections freeze | Transition tail drained by top-up; no tee → nothing to duplicate | Verify + audits, DLQ = 0 (live-parallel path; backpressure protects prod CH) |
-
-Scenario is also selectable on the dashboard's **Migration Guide** tab —
-it renders the per-scenario checklist and states the bound requirement.
-For 2 and 3: use **Detect boundary** + **Apply this bound to the run**
-(one click covers all pods), verify the `bounded · cd < …` badge on every
-pod, and keep re-running sync parity during the validation window.
-
## Clone-source variant (migrate from a frozen copy)
-A robust pattern: pause old ingestion, clone the source MongoDB onto the
-new machine, then resume ingestion on the NEW stack (optionally mirroring
-back to the old one as the rollback net) and migrate from the clone. SDK
-offline queues absorb the pause. Properties worth knowing:
+An OPTIONAL variant of scenarios 1–3 — most migrations run against the
+LIVE source cluster, which is fully supported (in unbounded modes, top-up
+passes keep chasing data that arrives during the run). The variant: pause
+old ingestion, clone the source MongoDB, resume ingestion on the NEW stack
+(optionally mirroring back to the old one as the rollback net), and migrate
+from the clone. SDK offline queues absorb the pause. Properties:
- The source is frozen at the clone moment, so no bound is needed and top-up
finds nothing — the startup guard will still ask (the target ingests live
From 29dbff80347bb392ce4f37e3e7f47376823591ec Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 13:03:06 +0300
Subject: [PATCH 10/64] fix(ledger): distinct-id drift coverage, fail-closed
DLQ/bound reads, no-scope dedupe refusal, env-bound cutover validation
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Third review round:
- drift spot-check counts DISTINCT sampled ids (uniqExact) — duplicate rows
of one id no longer vouch for another id's absence — and discounts only
the sampled ids that are themselves unresolved in the DLQ, not the
window's whole unresolved count.
- final check fails CLOSED: an unreadable DLQ count or stored bound aborts
the check as a tooling error instead of reading as 'empty DLQ' / 'no
bound' and authorizing teardown without evidence.
- dedupe: collections without an (a,e,n) scope have no usable native
evidence (table-wide counts let sibling traffic vouch for their outage
buckets) — their matches are always reported unsafe (reason: no-scope)
and never deleted.
- explicit final-check cutoverMs is validated against the EFFECTIVE bound
(stored OR env) — an env-bounded run can no longer be under-audited.
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 5 +++-
src/http/ledger-viz-route.ts | 2 +-
src/runtime/dedupe-overlap.ts | 12 ++++++++-
src/runtime/final-check.ts | 9 +++++--
src/runtime/ledger-engine.ts | 5 ++--
src/runtime/ledger-rebuild.ts | 12 ++++++---
src/state/dlq-store.ts | 11 +++++++++
src/target/staging-manager.ts | 18 ++++++++++++++
tests/integration/dedupe-overlap.test.ts | 31 +++++++++++++++++++++---
tests/integration/final-check.test.ts | 10 ++++++++
10 files changed, 100 insertions(+), 15 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index 0702490..c41d082 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -128,7 +128,10 @@ that event. Every hour bucket therefore needs count-evidence of native
counterparts — `native = live − matched` must roughly cover `matched` —
before anything in it is deleted. Buckets that fall short are skipped and
reported (`unsafe` in the result); review those hours (tee outage? wrong
-start time?) instead of forcing them. The check is strict (zero slack) by
+start time?) instead of forcing them. Collections without their own
+(a,e,n) scope (e.g. a base `drill_events` collection) have no usable
+native-counterpart evidence — their matches are always reported as unsafe
+and never deleted. The check is strict (zero slack) by
default; `slackPct` (≤5) may be passed consciously to absorb ingest-timing
straddle at bucket edges. Known limit: a loss exactly offset by
mirror-dropped natives in the same hour is invisible to count evidence —
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index 00a103b..628dcf1 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -806,7 +806,7 @@ function renderDedupe(dd) {
? '\u2705 Deleted ' + fmt(t.deleted) + ' duplicate row(s).'
: 'Dry run: ' + fmt(t.chMatched) + ' migrated row(s) match old-cluster ids in the window (' + fmt(t.mongoDocsInWindow) + ' old-side docs scanned). Nothing deleted.') + '
';
if (t.unsafeMatched > 0) {
- html += '
\u26a0 ' + fmt(t.unsafeMatched) + ' matched row(s) in ' + unsafeN + ' hour bucket(s) lack count-evidence of a native counterpart \u2014 there the migrated row may be the ONLY copy. They were ' + (dd.execute ? 'NOT deleted' : 'excluded') + '; review those hours (tee outage / wrong start time?) before touching them.
';
+ html += '
\u26a0 ' + fmt(t.unsafeMatched) + ' matched row(s) in ' + unsafeN + ' bucket(s) were NOT ' + (dd.execute ? 'deleted' : 'counted as deletable') + ': they lack count-evidence of a native counterpart, or belong to a collection without its own (a,e,n) scope \u2014 there the migrated row may be the ONLY copy. Review those buckets (tee outage / wrong start time / base collection?) before touching them.
';
}
if (!dd.execute && dd.lastDryRun) {
html += '
Window measured \u2014 the Delete button is now enabled for this exact window.
';
diff --git a/src/runtime/dedupe-overlap.ts b/src/runtime/dedupe-overlap.ts
index 3435916..167664a 100644
--- a/src/runtime/dedupe-overlap.ts
+++ b/src/runtime/dedupe-overlap.ts
@@ -48,6 +48,8 @@ export interface DedupeUnsafeBucket {
toMs: number;
matched: number;
native: number;
+ /** Why the bucket was skipped: missing native counterpart evidence, or a collection whose live counts cannot be scoped. */
+ reason: 'no-native-evidence' | 'no-scope';
}
export interface DedupeCollectionRow {
@@ -139,6 +141,14 @@ export async function runDedupeOverlap(
row.chMatched += matched;
state.totals.chMatched += matched;
if (matched === 0) return;
+ // No (a,e,n) scope → live counts are TABLE-WIDE and sibling
+ // collections' native traffic would vouch for this one's outage
+ // buckets. No usable evidence — never delete, always report.
+ if (!scope) {
+ row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native: -1, reason: 'no-scope' });
+ state.totals.unsafeMatched += matched;
+ return;
+ }
// Count evidence of native counterparts: what remains in this bucket
// after the matched rows is the native side. Falling short means some
// migrated rows are the ONLY copy of their event — never delete those.
@@ -150,7 +160,7 @@ export async function runDedupeOverlap(
// signature outright and no slack ever waves it through
const slack = Math.ceil(matched * (Math.min(5, Math.max(0, opts.slackPct ?? 0)) / 100));
if (native < matched - slack || native <= 0) {
- row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native });
+ row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native, reason: 'no-native-evidence' });
state.totals.unsafeMatched += matched;
return;
}
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 195c308..2eaa880 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -83,7 +83,9 @@ export async function runFinalCheck(
Object.assign(out, newFinalCheckResult(), { status: 'running', startedAt: Date.now(), phase: 'starting' });
try {
// ── Cutover: explicit param > stored bound > env bound > none ─────────
- const stored = await ledger.getStoredBound(runId).catch(() => null);
+ // fail CLOSED: if the bound cannot be read, the check errors out rather
+ // than silently auditing a different range
+ const stored = await ledger.getStoredBound(runId);
const cutoverMs = opts.cutoverMs ?? stored ?? config.ledger.cdUpperBoundMs ?? null;
out.cutoverMs = cutoverMs;
@@ -110,7 +112,10 @@ export async function runFinalCheck(
// ── 2. DLQ ─────────────────────────────────────────────────────────────
out.phase = 'checking dead-letter queue';
- const dlqCounts = await dlq.countByStatus(runId).catch(() => ({} as Record));
+ // fail CLOSED: an unreadable DLQ is indistinguishable from an empty one —
+ // a thrown error here fails the whole check as a tooling error instead of
+ // authorizing teardown without DLQ evidence
+ const dlqCounts = await dlq.countByStatus(runId);
const dlqPending = dlqCounts.pending ?? 0;
const dlqWaived = dlqCounts.waived ?? 0;
if (dlqPending > 0) {
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 4a47360..fcf2590 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -405,8 +405,9 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise null);
- if (storedFc !== null && cutoverMs < storedFc) {
- return { started: false, reason: `cutoverMs is EARLIER than the run's stored bound (${new Date(storedFc).toISOString()}) — that would silently exclude migrated data from the audit; pass the bound or later` };
+ const effectiveBound = storedFc ?? config.ledger.cdUpperBoundMs ?? null;
+ if (effectiveBound !== null && cutoverMs < effectiveBound) {
+ return { started: false, reason: `cutoverMs is EARLIER than the run's effective bound (${new Date(effectiveBound).toISOString()}) — that would silently exclude migrated data from the audit; pass the bound or later` };
}
}
const samples = Math.min(10_000, Math.max(50, req.body?.samples ?? 500));
diff --git a/src/runtime/ledger-rebuild.ts b/src/runtime/ledger-rebuild.ts
index 3394a60..e2553de 100644
--- a/src/runtime/ledger-rebuild.ts
+++ b/src/runtime/ledger-rebuild.ts
@@ -257,10 +257,14 @@ export async function rebuildLedger(opts: {
const sampleIds = (await coll
.find({ cd: { $gte: new Date(b.lowerCd), $lt: new Date(b.upperCd) } }, { projection: { _id: 1 } })
.limit(5_000).toArray()).map((d) => String(d._id));
- const present = await staging.countMatchingIdsInWindow(sampleIds, b.lowerCd, b.upperCd);
- // DLQ'd docs are legitimately absent — only a shortfall beyond
- // the window's unresolved count is a real coverage gap
- const missing = Math.max(0, sampleIds.length - present - unresolved);
+ // DISTINCT coverage: a duplicate row of one sampled id must not
+ // vouch for another sampled id being absent
+ const present = await staging.countDistinctMatchingIdsInWindow(sampleIds, b.lowerCd, b.upperCd);
+ // DLQ'd docs are legitimately absent — but only the SAMPLED ids
+ // that are themselves in the DLQ may be discounted; unrelated
+ // unresolved docs elsewhere in the window explain nothing
+ const unresolvedInSample = await dlq.countUnresolvedMatchingIds(runId, collection, sampleIds, b.lowerCd, b.upperCd);
+ const missing = Math.max(0, sampleIds.length - present - unresolvedInSample);
if (missing > 0 && (progress.driftSubsetMissing ?? []).length < 200) {
(progress.driftSubsetMissing ?? (progress.driftSubsetMissing = [])).push({
collection, lowerCd: new Date(b.lowerCd).toISOString(), upperCd: new Date(b.upperCd).toISOString(),
diff --git a/src/state/dlq-store.ts b/src/state/dlq-store.ts
index dfe2bc8..26039b0 100644
--- a/src/state/dlq-store.ts
+++ b/src/state/dlq-store.ts
@@ -126,6 +126,17 @@ export class DlqStore {
* table is accounted for, not a disagreement. Entries written before the
* cd_ms field (or with unparseable cd/ts) can't be attributed and count 0.
*/
+ /** How many of the GIVEN source ids sit unresolved (pending/waived) in the window — exact per-sample DLQ discount. */
+ async countUnresolvedMatchingIds(runId: string, collection: string, ids: string[], lowerCdMs: number, upperCdMs: number): Promise {
+ if (ids.length === 0) return 0;
+ return this.c().countDocuments({
+ run_id: runId, collection,
+ source_id: { $in: ids },
+ status: { $in: ['pending', 'waived'] },
+ cd_ms: { $gte: lowerCdMs, $lt: upperCdMs },
+ });
+ }
+
async countUnresolvedInWindow(runId: string, collection: string, lowerCdMs: number, upperCdMs: number): Promise {
return this.c().countDocuments({
run_id: runId, collection,
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index ed7fd8a..9d951ed 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -584,6 +584,24 @@ export class StagingManager {
return (await res.json<{ x: number }>()).length > 0;
}
+ /** DISTINCT given ids present live in [fromMs, toMs) — duplicate rows of one id never vouch for another id's absence. */
+ async countDistinctMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ let total = 0;
+ for (let i = 0; i < ids.length; i += 50_000) {
+ const page = ids.slice(i, i + 50_000);
+ const res = await this.ch().query({
+ query: `SELECT uniqExact(_id) AS n FROM ${this.fq(this.config.table)}
+ WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
+ AND _id IN {ids:Array(String)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs },
+ format: 'JSONEachRow',
+ });
+ const rows = await res.json<{ n: string }>();
+ total += Number(rows[0]?.n ?? 0);
+ }
+ return total;
+ }
+
/** Live rows in [fromMs, toMs) whose _id is one of the given ids. */
async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
let total = 0;
diff --git a/tests/integration/dedupe-overlap.test.ts b/tests/integration/dedupe-overlap.test.ts
index 80a1f3a..0ff11de 100644
--- a/tests/integration/dedupe-overlap.test.ts
+++ b/tests/integration/dedupe-overlap.test.ts
@@ -33,6 +33,9 @@ const COLL = `drill_events${createHash('sha1').update('views' + APP).digest('hex
const APP2 = 'app_dd_outage';
const COLL2 = `drill_events${createHash('sha1').update('views' + APP2).digest('hex')}`;
const OUTAGE = 40;
+// base collection: no per-collection (a,e,n) scope resolvable — its matches
+// must never be deleted, even though sibling native traffic fills the table
+const BASE = 20;
const FLIP = Math.floor(Date.now() / 60_000) * 60_000 - 2 * 3_600_000; // tee flip 2h ago
const DONE = FLIP + 3_600_000; // migration completed 1h later
@@ -116,6 +119,19 @@ describe('tee-overlap dedupe', () => {
await mc.db(DB).collection(COLL2).createIndex({ cd: 1, _id: 1 });
await ch.insert({ table: `${DB}.drill_events`, values: outageRows, format: 'JSONEachRow' });
+ // unscoped base collection: migrated copies whose only "native cover" is
+ // SIBLING collections' traffic — no usable evidence, never deletable
+ const baseDocs: Record[] = [];
+ const baseRows: Record[] = [];
+ for (let i = 0; i < BASE; i++) {
+ const cd = FLIP + i * 15_000;
+ baseDocs.push({ _id: `base_${i}`, a: APP, e: '[CLY]_custom', n: 'views', uid: 'u', did: 'd', ts: cd, cd: new Date(cd), sg: {}, c: 1 });
+ baseRows.push(chRow(`base_${i}`, cd));
+ }
+ await mc.db(DB).collection('drill_events').insertMany(baseDocs as never[]);
+ await mc.db(DB).collection('drill_events').createIndex({ cd: 1, _id: 1 });
+ await ch.insert({ table: `${DB}.drill_events`, values: baseRows, format: 'JSONEachRow' });
+
Object.assign(process.env, {
SERVICE_NAME: 'dedupe-test',
MONGO_URI, MONGO_DB: DB, MONGO_COUNTLY_DB: `${DB}_countly`, MANIFEST_DB: DB,
@@ -140,23 +156,30 @@ describe('tee-overlap dedupe', () => {
const state = newDedupeOverlapState();
await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: false });
expect(state.status).toBe('completed');
- expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE, chMatched: 150 + OUTAGE, deleted: 0, unsafeMatched: OUTAGE });
- expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 + OUTAGE });
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE + BASE, chMatched: 150 + OUTAGE + BASE, deleted: 0, unsafeMatched: OUTAGE + BASE });
+ expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 + OUTAGE + BASE });
const outageRow = state.collections.find((c) => c.collection === COLL2);
expect(outageRow?.unsafe.length).toBeGreaterThan(0);
expect(outageRow?.unsafe.reduce((a, u) => a + u.matched, 0)).toBe(OUTAGE);
- expect(await chCount()).toBe(200 + 150 + 150 + 10 + OUTAGE);
+ expect(outageRow?.unsafe.every((u) => u.reason === 'no-native-evidence')).toBe(true);
+ const baseRow = state.collections.find((c) => c.collection === 'drill_events');
+ expect(baseRow?.scoped).toBe(false);
+ expect(baseRow?.unsafe.every((u) => u.reason === 'no-scope')).toBe(true);
+ expect(baseRow?.unsafe.reduce((a, u) => a + u.matched, 0)).toBe(BASE);
+ expect(await chCount()).toBe(200 + 150 + 150 + 10 + OUTAGE + BASE);
});
it('execute deletes exactly the evidenced duplicates; unsafe buckets, native and pre-flip rows survive', async () => {
const state = newDedupeOverlapState();
await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: true });
expect(state.status).toBe('completed');
- expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE, chMatched: 150 + OUTAGE, deleted: 150, unsafeMatched: OUTAGE });
+ expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE + BASE, chMatched: 150 + OUTAGE + BASE, deleted: 150, unsafeMatched: OUTAGE + BASE });
expect(await chCount("_id LIKE 'mirror_%'")).toBe(0);
expect(await chCount("_id LIKE 'native_%'")).toBe(160);
expect(await chCount("_id LIKE 'hist_%'")).toBe(200);
// the only-copy rows are untouched — the safety check protected them
expect(await chCount("_id LIKE 'only_%'")).toBe(OUTAGE);
+ // unscoped base-collection rows: sibling traffic is not evidence
+ expect(await chCount("_id LIKE 'base_%'")).toBe(BASE);
});
});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index abe33f5..e489f51 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -231,6 +231,16 @@ describe('final check: the interpreted sign-off', () => {
const out2 = await check({ cutoverMs: CUTOVER });
expect(out2.verdict).toBe('PASS_WITH_NOTES');
expect(out2.notes.join(' ')).toContain('retained history');
+
+ // a DUPLICATE row of one id must not vouch for another id's absence:
+ // same total row count, one id missing — distinct coverage catches it
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('m_70', START + 70 * 12_000)] });
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_71'` });
+ const out3 = await check({ cutoverMs: CUTOVER });
+ expect(out3.verdict).toBe('FAIL');
+ expect(out3.problems.join(' ')).toContain('masking');
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_70'` });
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('m_70', START + 70 * 12_000), chRow('m_71', START + 71 * 12_000)] });
});
it('a WHOLE window missing from the target → FAIL (the audit calls it pending, the check must not)', async () => {
From d8e0fc149a91ab8c9d7cb0da50a305f57f551c52 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 13:12:19 +0300
Subject: [PATCH 11/64] =?UTF-8?q?feat(ledger):=20tiered=20Final=20check=20?=
=?UTF-8?q?=E2=80=94=20quick=20(ledger=20verify=20+=20samples)=20by=20defa?=
=?UTF-8?q?ult,=20deep=20source=20recount=20opt-in?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The full source recount is the right gate before DELETING the source, but
it is hours of per-window queries on large runs, and routine sign-off
confidence does not need it: everything that can happen AFTER reading is
caught by verifying the target against the run ledger (grouped window
counts + duplicate attribution, minutes) plus random content samples
against the source.
- quick (default): chunks + DLQ (fail-closed) + verifyMigration + sampled
content. Never returns a plain PASS — a note names exactly what was not
re-proven and points at deep.
- deep ({"deep": true} / dashboard checkbox): adds the full recount with
cd-checksums, drift spot-checks and the cutover clamp — unchanged.
- contentAudit gains a total budget so 2,500-collection deployments sample
~tens of thousands of docs, not 500 per collection.
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 22 ++++++++++--
src/http/ledger-viz-route.ts | 8 +++--
src/runtime/chunk-orchestrator.ts | 7 +++-
src/runtime/final-check.ts | 50 ++++++++++++++++++++++-----
src/runtime/ledger-engine.ts | 7 ++--
tests/integration/final-check.test.ts | 22 ++++++++++--
6 files changed, 97 insertions(+), 19 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index c41d082..f74f5eb 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -97,13 +97,29 @@ only question that matters — *is it safe to decommission the old cluster?* —
as **PASS / PASS WITH NOTES / FAIL** in plain sentences with the action named
on every red line.
-- Dashboard: the **Final check** card → *Run final check*. On tee/mirror runs
- without a stored bound, type the cutover time into the field first.
+Two tiers:
+
+- **Quick** (default — minutes): chunk states + DLQ + target-vs-ledger
+ verification (every migrated window's live count against the recorded
+ count, plus duplicate attribution) + random content samples against the
+ source. Catches everything that can happen AFTER reading. Capped at
+ PASS WITH NOTES — the note names what it did not re-prove.
+- **Deep** (opt-in — hours on large runs): additionally recounts EVERY
+ window against the source with cd-checksum fingerprints. This is the one
+ check that would catch a self-consistently under-reading reader, so run
+ it once before the source is deleted; while the source still exists,
+ quick is enough for routine confidence.
+
+- Dashboard: the **Final check** card → *Run final check* (tick *deep source
+ recount* for the pre-teardown gate). On tee/mirror runs without a stored
+ bound, type the cutover time into the field first.
- SSH-only:
```bash
-# start (add {"cutoverMs": } for mirror runs without a stored bound)
+# quick (add {"cutoverMs": } for mirror runs without a stored bound)
curl -s -X POST localhost:PORT/control/final-check -H 'content-type: application/json' -d '{}'
+# deep — before deleting the source
+curl -s -X POST localhost:PORT/control/final-check -H 'content-type: application/json' -d '{"deep": true}'
# read the verdict (re-run until it says PASS/FAIL; shows progress while running)
curl -s localhost:PORT/final-check.txt
```
diff --git a/src/http/ledger-viz-route.ts b/src/http/ledger-viz-route.ts
index 628dcf1..ee9d693 100644
--- a/src/http/ledger-viz-route.ts
+++ b/src/http/ledger-viz-route.ts
@@ -304,7 +304,8 @@ const PAGE = `
Final check — one click answers: is it safe to decommission the old source? (chunks + DLQ + full source recount + checksums + content samples — interpreted for you)
-
+
+
Not run. Run it after the migration completes — it recounts every window against the source, so give it time on big runs; progress shows here. SSH-only: curl -X POST :PORT/control/final-check then curl :PORT/final-check.txt
@@ -818,6 +819,8 @@ function renderDedupe(dd) {
async function startFinalCheck(btn) {
var body = {};
+ var deepEl = document.getElementById('fc-deep');
+ if (deepEl && deepEl.checked) body.deep = true;
var cutRaw = (document.getElementById('fc-cutover').value || '').trim();
if (cutRaw) {
var ms = Date.parse(cutRaw);
@@ -856,7 +859,8 @@ function renderFinalCheck(fc) {
return;
}
var pal = fc.verdict === 'PASS' ? ['#E4F6EC', '#157A45'] : fc.verdict === 'PASS_WITH_NOTES' ? ['#FDEEDD', '#A05A16'] : ['#FDECEC', '#B3261E'];
- var badge = fc.verdict === 'PASS' ? 'PASS' : fc.verdict === 'PASS_WITH_NOTES' ? 'PASS WITH NOTES' : 'FAIL';
+ var badge = (fc.verdict === 'PASS' ? 'PASS' : fc.verdict === 'PASS_WITH_NOTES' ? 'PASS WITH NOTES' : 'FAIL')
+ + (fc.mode === 'deep' ? ' \u00b7 deep (full source recount)' : ' \u00b7 quick (ledger verify + samples)');
var html = '
'
+ '
' + badge + '
'
+ '
' + fcEsc(fc.headline) + '
'
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index ebefc4f..11f04cb 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -1636,7 +1636,7 @@ export class ChunkOrchestrator {
* so value-level equality there belongs to the differential harness, which
* pins the transform itself).
*/
- async contentAudit(samplesPerCollection = 500, upToMs: number | null = null): Promise<{
+ async contentAudit(samplesPerCollection = 500, upToMs: number | null = null, totalBudget: number | null = null): Promise<{
sampled: number; matched: number; missing: number; different: number;
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}> {
@@ -1651,6 +1651,11 @@ export class ChunkOrchestrator {
const defaults = this.d.hashResolver.resolveCollectionName(name, config.source.collectionPrefix);
return !(defaults && skipEventNames.has(defaults.e));
});
+ // a TOTAL budget keeps many-collection deployments sane: 2,500
+ // collections × 500 samples each is a million-doc audit nobody asked for
+ if (totalBudget !== null && collections.length > 0) {
+ samplesPerCollection = Math.min(samplesPerCollection, Math.max(10, Math.ceil(totalBudget / collections.length)));
+ }
let missing = 0, different = 0;
for (const collection of collections) {
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 2eaa880..7397534 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -25,6 +25,8 @@ import { rebuildLedger, newRebuildProgress, type RebuildProgress } from './ledge
export interface FinalCheckResult {
status: 'not_run' | 'running' | 'completed' | 'failed';
+ /** quick = ledger-verify + sampled source checks (minutes); deep = full source recount + checksums (the pre-teardown gate). */
+ mode: 'quick' | 'deep' | null;
verdict: 'PASS' | 'PASS_WITH_NOTES' | 'FAIL' | null;
/** One sentence answering "can I decommission the old cluster?" */
headline: string | null;
@@ -47,7 +49,7 @@ export interface FinalCheckResult {
export function newFinalCheckResult(): FinalCheckResult {
return {
- status: 'not_run', verdict: null, headline: null,
+ status: 'not_run', mode: null, verdict: null, headline: null,
passes: [], notes: [], problems: [],
cutoverMs: null, phase: '', audit: null, content: null,
error: null, startedAt: null, finishedAt: null,
@@ -58,10 +60,11 @@ const fmt = (n: number): string => n.toLocaleString('en-US');
const iso = (ms: number): string => new Date(ms).toISOString().slice(0, 16).replace('T', ' ') + ' UTC';
interface ContentAuditRunner {
- contentAudit(samplesPerCollection?: number, upToMs?: number | null): Promise<{
+ contentAudit(samplesPerCollection?: number, upToMs?: number | null, totalBudget?: number | null): Promise<{
sampled: number; matched: number; missing: number; different: number;
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}>;
+ verifyMigration(): Promise>;
}
export async function runFinalCheck(
@@ -74,13 +77,14 @@ export async function runFinalCheck(
orchestrator: ContentAuditRunner;
},
out: FinalCheckResult,
- opts: { cutoverMs: number | null; samples: number },
+ opts: { cutoverMs: number | null; samples: number; deep?: boolean },
): Promise {
const { config, ledger, dlq, hashResolver } = deps;
const logger = deps.logger.child({ component: 'FinalCheck' });
const runId = config.ledger.runId;
+ const deep = opts.deep === true;
- Object.assign(out, newFinalCheckResult(), { status: 'running', startedAt: Date.now(), phase: 'starting' });
+ Object.assign(out, newFinalCheckResult(), { status: 'running', mode: deep ? 'deep' : 'quick', startedAt: Date.now(), phase: 'starting' });
try {
// ── Cutover: explicit param > stored bound > env bound > none ─────────
// fail CLOSED: if the bound cannot be read, the check errors out rather
@@ -131,9 +135,38 @@ export async function runFinalCheck(
}
if (dlqPending === 0 && dlqWaived === 0) out.passes.push('Dead-letter queue is empty — no document was skipped.');
- // ── 3. Full source recount + cd-checksum fingerprint (the heavy one) ──
- out.phase = 'recounting every window against the source';
+ // ── 3a. QUICK tier: target vs the run's own ledger (minutes) ──────────
+ // Catches everything that happened AFTER reading: lost partitions, rows
+ // deleted from the live table, duplicate attribution. What it cannot see
+ // is a self-consistently under-reading reader — the ledger agreeing with
+ // itself while the source held more. That class is covered
+ // probabilistically by the content samples below, and exactly by the
+ // deep recount — which is why quick mode never returns a plain PASS.
+ if (!deep) {
+ out.phase = 'verifying the target against the run ledger';
+ const verify = await deps.orchestrator.verifyMigration();
+ const vMism = (verify.mismatches as Array> | undefined) ?? [];
+ const vDup = Number((verify as Record).migrationDuplicates ?? 0);
+ if (verify.ok !== true) {
+ if (vMism.length > 0) {
+ out.problems.push(`${fmt(vMism.length)} chunk window(s) hold a different live row count than the ledger recorded — rows were lost or duplicated after migration. Run "Retry failed chunks" after a rebuild, or escalate; do NOT decommission the old cluster.`);
+ }
+ if (vDup > 0) {
+ out.problems.push(`${fmt(vDup)} document(s) exist more than once below the migration boundary — a migration-side duplicate class; escalate before decommissioning.`);
+ }
+ if (vMism.length === 0 && vDup === 0) {
+ out.problems.push('Ledger verification reported a failure — inspect GET /api/verify before decommissioning.');
+ }
+ } else {
+ out.passes.push('Target verified against the run ledger: every migrated chunk window holds exactly the recorded row count, with no migration-side duplicates.');
+ }
+ out.notes.push(`Quick mode: the ledger itself was not re-proven against the source. Per-chunk verification at attach time plus the random content samples below cover that class probabilistically — run the DEEP check ({"deep": true}, or the checkbox in the dashboard) before deleting the source if you want the full recount + checksum fingerprints.`);
+ }
+
+ // ── 3b. DEEP tier: full source recount + cd-checksum fingerprint ──────
const audit = newRebuildProgress();
+ if (deep) {
+ out.phase = 'recounting every window against the source';
out.audit = audit;
await rebuildLedger({ config, logger, ledger, dlq, hashResolver, progress: audit, checkOnly: true, upToMs: cutoverMs });
const windows = audit.summary.reduce((a, s) => a + s.chunks, 0);
@@ -171,10 +204,11 @@ export async function runFinalCheck(
const excluded = audit.excludedBeyondCutover ?? 0;
out.notes.push(`Source docs after the cutover (${iso(cutoverMs)}) were excluded from the comparison${excluded > 0 ? ` (${fmt(excluded)} docs)` : ''} — after that moment the old side receives mirrored/live traffic that was never meant to be migrated, so divergence there is expected and is NOT data loss.`);
}
+ }
// ── 4. Sampled content comparison ──────────────────────────────────────
out.phase = 'comparing sampled documents field-by-field';
- const content = await deps.orchestrator.contentAudit(opts.samples, cutoverMs);
+ const content = await deps.orchestrator.contentAudit(opts.samples, cutoverMs, Math.max(2_000, opts.samples));
out.content = { sampled: content.sampled, matched: content.matched, missing: content.missing, different: content.different };
if (content.missing > 0 || content.different > 0) {
out.problems.push(`Content sampling found ${fmt(content.missing)} missing and ${fmt(content.different)} differing doc(s) out of ${fmt(content.sampled)} sampled — the migrated content does not match the source; escalate before decommissioning.`);
@@ -214,7 +248,7 @@ export function renderFinalCheckText(fc: FinalCheckResult, runId: string): strin
lines.push(`CHECK FAILED TO COMPLETE: ${fc.error} - fix and re-run; this is a tooling error, not a data verdict.`);
} else {
const badge = fc.verdict === 'PASS' ? 'PASS' : fc.verdict === 'PASS_WITH_NOTES' ? 'PASS WITH NOTES' : 'FAIL';
- lines.push(`Verdict: ${badge} - ${fc.headline}`);
+ lines.push(`Verdict: ${badge} (${fc.mode === 'deep' ? 'deep: full source recount' : 'quick: ledger verify + samples'}) - ${fc.headline}`);
for (const p of fc.problems) lines.push(` [X] ${p}`);
for (const n of fc.notes) lines.push(` [!] ${n}`);
for (const g of fc.passes) lines.push(` [ok] ${g}`);
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index fcf2590..578c55d 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -392,7 +392,7 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise('/control/final-check', async (req) => {
+ app.post<{ Body: { cutoverMs?: number; samples?: number; deep?: boolean } }>('/control/final-check', async (req) => {
if (finalCheckState.status === 'running') return { started: false, reason: 'final check already running' };
if (orchestrator.getStatus() === 'running') return { started: false, reason: 'main migration is running — run the final check after completion (or while paused)' };
// no exclusion: the SERVING pod's own live claims block the check too —
@@ -411,8 +411,9 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise finalCheckState);
app.get('/final-check.txt', async (_req, reply) => {
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index e489f51..ade7db9 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -49,6 +49,7 @@ const chRow = (id: string, cdMs: number): Record => ({
const contentClean = {
contentAudit: async (samples = 500) => ({ sampled: samples, matched: samples, missing: 0, different: 0, mismatches: [] }),
+ verifyMigration: async () => ({ ok: true, mismatches: [], migrationDuplicates: 0 }),
};
describe('final check: the interpreted sign-off', () => {
@@ -59,12 +60,12 @@ describe('final check: the interpreted sign-off', () => {
let hashResolver: HashResolver;
let config: Config;
- const check = async (opts?: { cutoverMs?: number | null; orchestrator?: typeof contentClean }) => {
+ const check = async (opts?: { cutoverMs?: number | null; orchestrator?: typeof contentClean; deep?: boolean }) => {
const out = newFinalCheckResult();
await runFinalCheck(
{ config, logger, ledger, dlq, hashResolver, orchestrator: opts?.orchestrator ?? contentClean },
out,
- { cutoverMs: opts?.cutoverMs ?? null, samples: 100 },
+ { cutoverMs: opts?.cutoverMs ?? null, samples: 100, deep: opts?.deep ?? true },
);
expect(out.status).toBe('completed');
return out;
@@ -204,6 +205,23 @@ describe('final check: the interpreted sign-off', () => {
expect((await ledger.summarize(RUN)).docsSkipped).toBe(7);
});
+ it('quick mode: ledger verify + samples, capped at PASS WITH NOTES, never plain PASS', async () => {
+ const out = await check({ cutoverMs: CUTOVER, deep: false });
+ expect(out.mode).toBe('quick');
+ expect(out.problems).toEqual([]);
+ expect(out.verdict).toBe('PASS_WITH_NOTES');
+ expect(out.notes.join(' ')).toContain('DEEP');
+ expect(out.audit).toBeNull(); // no source recount ran
+
+ const badVerify = {
+ ...contentClean,
+ verifyMigration: async () => ({ ok: false, mismatches: [{ chunk: 'x', expected: 10, live: 7 }], migrationDuplicates: 0 }),
+ };
+ const out2 = await check({ cutoverMs: CUTOVER, deep: false, orchestrator: badVerify });
+ expect(out2.verdict).toBe('FAIL');
+ expect(out2.problems.join(' ')).toContain('different live row count');
+ });
+
it('content mismatch and failed chunks each FAIL with their own action line', async () => {
const badContent = {
contentAudit: async (samples = 500) => ({ sampled: samples, matched: samples - 2, missing: 1, different: 1, mismatches: [] }),
From 93845d55cf02c37c5a3d5ccd3b047e76e7e64d18 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 13:21:35 +0300
Subject: [PATCH 12/64] fix(ledger): scope dedupe id-matching end to end,
slack-aware execute license, failed null-cd sweeps surface as mismatches
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Fourth review round:
- countMatchingIdsInWindow / countDistinctMatchingIdsInWindow /
deleteMatchingIdsInWindow take the collection's (a,e,n) scope: a sibling
collection's row sharing an _id can neither vouch for coverage nor be
deleted by another collection's dedupe pass (pinned by a sibling-row
survival test). Alias renamed cnt — 'AS n' collided with the scope
filter's real column n (ILLEGAL_AGGREGATION).
- the dedupe execute license records the dry run's effective slackPct and
refuses an execute with a different slack: execute may only delete what a
reviewed dry run counted.
- a PARTIALLY swept null-cd sentinel (some of a collection's cd:null docs
missing live) now lands in mismatchedWindows instead of hiding in a
summary counter, so the deep final check FAILs on it.
Co-Authored-By: Claude Fable 5
---
src/runtime/dedupe-overlap.ts | 16 +++++++----
src/runtime/ledger-engine.ts | 8 +++---
src/runtime/ledger-rebuild.ts | 10 ++++++-
src/target/staging-manager.ts | 35 ++++++++++++------------
tests/integration/dedupe-overlap.test.ts | 12 ++++++--
tests/integration/final-check.test.ts | 14 ++++++++++
6 files changed, 64 insertions(+), 31 deletions(-)
diff --git a/src/runtime/dedupe-overlap.ts b/src/runtime/dedupe-overlap.ts
index 167664a..28fbd38 100644
--- a/src/runtime/dedupe-overlap.ts
+++ b/src/runtime/dedupe-overlap.ts
@@ -71,8 +71,8 @@ export interface DedupeOverlapState {
toMs: number | null;
collections: DedupeCollectionRow[];
totals: { mongoDocsInWindow: number; chMatched: number; deleted: number; unsafeMatched: number };
- /** Window of the last COMPLETED dry run — the license to execute. */
- lastDryRun: { fromMs: number; toMs: number; chMatched: number; at: number } | null;
+ /** Window + slack of the last COMPLETED dry run — the license to execute (same window AND same slack). */
+ lastDryRun: { fromMs: number; toMs: number; slackPct: number; chMatched: number; at: number } | null;
error: string | null;
startedAt: number | null;
finishedAt: number | null;
@@ -86,6 +86,8 @@ export function newDedupeOverlapState(): DedupeOverlapState {
};
}
+export const effectiveSlackPct = (v: number | undefined): number => Math.min(5, Math.max(0, v ?? 0));
+
const ID_BATCH = 50_000;
const BUCKET_MS = 3_600_000;
/** Hard ceiling on one bucket's ids held in memory — pick a smaller window if hit. */
@@ -136,7 +138,9 @@ export async function runDedupeOverlap(
if (ids.length === 0) return;
let matched = 0;
for (let i = 0; i < ids.length; i += ID_BATCH) {
- matched += await staging.countMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs);
+ // scoped: a same-_id row in a SIBLING collection must neither count
+ // as this collection's match nor be touched by its delete
+ matched += await staging.countMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs, scope);
}
row.chMatched += matched;
state.totals.chMatched += matched;
@@ -158,7 +162,7 @@ export async function runDedupeOverlap(
// its bucket. slackPct (operator-chosen, ≤5%) only absorbs
// ingest-timing straddle at bucket edges; zero natives is the outage
// signature outright and no slack ever waves it through
- const slack = Math.ceil(matched * (Math.min(5, Math.max(0, opts.slackPct ?? 0)) / 100));
+ const slack = Math.ceil(matched * (effectiveSlackPct(opts.slackPct) / 100));
if (native < matched - slack || native <= 0) {
row.unsafe.push({ fromMs: loMs, toMs: hiMs, matched, native, reason: 'no-native-evidence' });
state.totals.unsafeMatched += matched;
@@ -166,7 +170,7 @@ export async function runDedupeOverlap(
}
if (opts.execute) {
for (let i = 0; i < ids.length; i += ID_BATCH) {
- await staging.deleteMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs);
+ await staging.deleteMatchingIdsInWindow(ids.slice(i, i + ID_BATCH), loMs, hiMs, scope);
}
row.deleted += matched;
state.totals.deleted += matched;
@@ -202,7 +206,7 @@ export async function runDedupeOverlap(
state.phase = 'done';
state.finishedAt = Date.now();
if (!opts.execute) {
- state.lastDryRun = { fromMs: opts.fromMs, toMs: opts.toMs, chMatched: state.totals.chMatched, at: Date.now() };
+ state.lastDryRun = { fromMs: opts.fromMs, toMs: opts.toMs, slackPct: effectiveSlackPct(opts.slackPct), chMatched: state.totals.chMatched, at: Date.now() };
}
logger.info(
{ execute: opts.execute, ...state.totals, collections: state.collections.length },
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 578c55d..3ab1948 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -22,7 +22,7 @@ import { ChunkOrchestrator } from './chunk-orchestrator.ts';
import { wireExitOnComplete } from './exit-on-complete.ts';
import { rebuildLedger, newRebuildProgress, type RebuildProgress } from './ledger-rebuild.ts';
import { runFinalCheck, newFinalCheckResult, renderFinalCheckText, type FinalCheckResult } from './final-check.ts';
-import { runDedupeOverlap, newDedupeOverlapState, type DedupeOverlapState } from './dedupe-overlap.ts';
+import { runDedupeOverlap, newDedupeOverlapState, effectiveSlackPct, type DedupeOverlapState } from './dedupe-overlap.ts';
export async function runLedgerEngine(config: Config, logger: Logger): Promise {
logger.info({ engine: 'ledger', runId: config.ledger.runId }, 'Starting ledger engine (no Redis)');
@@ -441,13 +441,13 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise String(d._id));
// DISTINCT coverage: a duplicate row of one sampled id must not
// vouch for another sampled id being absent
- const present = await staging.countDistinctMatchingIdsInWindow(sampleIds, b.lowerCd, b.upperCd);
+ const present = await staging.countDistinctMatchingIdsInWindow(sampleIds, b.lowerCd, b.upperCd, scope);
// DLQ'd docs are legitimately absent — but only the SAMPLED ids
// that are themselves in the DLQ may be discounted; unrelated
// unresolved docs elsewhere in the window explain nothing
@@ -312,6 +312,14 @@ export async function rebuildLedger(opts: {
const status: ChunkDoc['status'] =
swept === nullCdIds.length ? 'done' : swept === 0 ? 'pending' : 'failed';
summary[status === 'done' ? 'done' : status === 'pending' ? 'pending' : 'failed']++;
+ // a PARTIALLY swept sentinel means rows are missing from the target —
+ // it must surface as a mismatch, not hide in a summary counter
+ if (checkOnly && status === 'failed' && progress.mismatchedWindows.length < 200) {
+ progress.mismatchedWindows.push({
+ collection, lowerCd: 'null-cd sweep', upperCd: 'null-cd sweep',
+ source: nullCdIds.length, live: swept,
+ });
+ }
allDocs.push({
_id: `${runId}:${collection}:${idx}`,
run_id: runId, collection,
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index 9d951ed..865999b 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -584,38 +584,39 @@ export class StagingManager {
return (await res.json<{ x: number }>()).length > 0;
}
- /** DISTINCT given ids present live in [fromMs, toMs) — duplicate rows of one id never vouch for another id's absence. */
- async countDistinctMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ /** DISTINCT given ids present live in [fromMs, toMs) — duplicate rows of one id never vouch for another id's absence. Scope keeps a same-_id row in a SIBLING collection from vouching either. */
+ async countDistinctMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
let total = 0;
for (let i = 0; i < ids.length; i += 50_000) {
const page = ids.slice(i, i + 50_000);
const res = await this.ch().query({
- query: `SELECT uniqExact(_id) AS n FROM ${this.fq(this.config.table)}
+ // alias must not be 'n' — the scope filter references the real column n
+ query: `SELECT uniqExact(_id) AS cnt FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
- AND _id IN {ids:Array(String)}`,
- query_params: { ids: page, lo: fromMs, hi: toMs },
+ AND _id IN {ids:Array(String)} ${this.scopeSql(scope)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs, ...this.scopeParams(scope) },
format: 'JSONEachRow',
});
- const rows = await res.json<{ n: string }>();
- total += Number(rows[0]?.n ?? 0);
+ const rows = await res.json<{ cnt: string }>();
+ total += Number(rows[0]?.cnt ?? 0);
}
return total;
}
- /** Live rows in [fromMs, toMs) whose _id is one of the given ids. */
- async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ /** Live rows in [fromMs, toMs) whose _id is one of the given ids, scoped to a collection's (a,e,n) when known. */
+ async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
let total = 0;
for (let i = 0; i < ids.length; i += 50_000) {
const page = ids.slice(i, i + 50_000);
const res = await this.ch().query({
- query: `SELECT count() AS n FROM ${this.fq(this.config.table)}
+ query: `SELECT count() AS cnt FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
- AND _id IN {ids:Array(String)}`,
- query_params: { ids: page, lo: fromMs, hi: toMs },
+ AND _id IN {ids:Array(String)} ${this.scopeSql(scope)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs, ...this.scopeParams(scope) },
format: 'JSONEachRow',
});
- const rows = await res.json<{ n: string }>();
- total += Number(rows[0]?.n ?? 0);
+ const rows = await res.json<{ cnt: string }>();
+ total += Number(rows[0]?.cnt ?? 0);
}
return total;
}
@@ -626,14 +627,14 @@ export class StagingManager {
* cluster that the mirror had already re-ingested natively. The cd window
* keeps each DELETE partition-prunable on multi-billion-row tables.
*/
- async deleteMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number): Promise {
+ async deleteMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
for (let i = 0; i < ids.length; i += 50_000) {
const page = ids.slice(i, i + 50_000);
await this.ch().command({
query: `DELETE FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
- AND _id IN {ids:Array(String)}`,
- query_params: { ids: page, lo: fromMs, hi: toMs },
+ AND _id IN {ids:Array(String)} ${this.scopeSql(scope)}`,
+ query_params: { ids: page, lo: fromMs, hi: toMs, ...this.scopeParams(scope) },
});
}
}
diff --git a/tests/integration/dedupe-overlap.test.ts b/tests/integration/dedupe-overlap.test.ts
index 0ff11de..eb305fa 100644
--- a/tests/integration/dedupe-overlap.test.ts
+++ b/tests/integration/dedupe-overlap.test.ts
@@ -100,6 +100,9 @@ describe('tee-overlap dedupe', () => {
}
// extra native rows with no mirror copy (mirror dropped them) — survive
for (let i = 0; i < 10; i++) chRows.push(chRow(`native_only_${i}`, FLIP + 500_000 + i * 1_000));
+ // a SIBLING collection's row sharing an _id with a mirrored doc, in the
+ // window — scoped deletes must never touch it
+ chRows.push({ ...chRow('mirror_10', FLIP + 10 * 20_000 + 50), a: 'sibling_app' });
await mc.db(DB).collection(COLL).insertMany(mongoDocs as never[]);
await mc.db(DB).collection(COLL).createIndex({ cd: 1, _id: 1 });
await ch.insert({ table: `${DB}.drill_events`, values: chRows, format: 'JSONEachRow' });
@@ -157,7 +160,7 @@ describe('tee-overlap dedupe', () => {
await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: false });
expect(state.status).toBe('completed');
expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE + BASE, chMatched: 150 + OUTAGE + BASE, deleted: 0, unsafeMatched: OUTAGE + BASE });
- expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, chMatched: 150 + OUTAGE + BASE });
+ expect(state.lastDryRun).toMatchObject({ fromMs: FLIP, toMs: DONE, slackPct: 0, chMatched: 150 + OUTAGE + BASE });
const outageRow = state.collections.find((c) => c.collection === COLL2);
expect(outageRow?.unsafe.length).toBeGreaterThan(0);
expect(outageRow?.unsafe.reduce((a, u) => a + u.matched, 0)).toBe(OUTAGE);
@@ -166,7 +169,7 @@ describe('tee-overlap dedupe', () => {
expect(baseRow?.scoped).toBe(false);
expect(baseRow?.unsafe.every((u) => u.reason === 'no-scope')).toBe(true);
expect(baseRow?.unsafe.reduce((a, u) => a + u.matched, 0)).toBe(BASE);
- expect(await chCount()).toBe(200 + 150 + 150 + 10 + OUTAGE + BASE);
+ expect(await chCount()).toBe(200 + 150 + 150 + 10 + OUTAGE + BASE + 1);
});
it('execute deletes exactly the evidenced duplicates; unsafe buckets, native and pre-flip rows survive', async () => {
@@ -174,12 +177,15 @@ describe('tee-overlap dedupe', () => {
await runDedupeOverlap({ config, logger, hashResolver }, state, { fromMs: FLIP, toMs: DONE, execute: true });
expect(state.status).toBe('completed');
expect(state.totals).toEqual({ mongoDocsInWindow: 150 + OUTAGE + BASE, chMatched: 150 + OUTAGE + BASE, deleted: 150, unsafeMatched: OUTAGE + BASE });
- expect(await chCount("_id LIKE 'mirror_%'")).toBe(0);
+ expect(await chCount("_id LIKE 'mirror_%'")).toBe(1); // only the sibling collection's same-_id row remains
expect(await chCount("_id LIKE 'native_%'")).toBe(160);
expect(await chCount("_id LIKE 'hist_%'")).toBe(200);
// the only-copy rows are untouched — the safety check protected them
expect(await chCount("_id LIKE 'only_%'")).toBe(OUTAGE);
// unscoped base-collection rows: sibling traffic is not evidence
expect(await chCount("_id LIKE 'base_%'")).toBe(BASE);
+ // the sibling collection's same-_id row survives the scoped delete
+ expect(await chCount("_id = 'mirror_10' AND a = 'sibling_app'")).toBe(1);
+ expect(await chCount("_id = 'mirror_10'")).toBe(1);
});
});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index ade7db9..062043f 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -261,6 +261,20 @@ describe('final check: the interpreted sign-off', () => {
await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('m_70', START + 70 * 12_000), chRow('m_71', START + 71 * 12_000)] });
});
+ it('a PARTIALLY swept null-cd sentinel surfaces as a mismatch and FAILS', async () => {
+ const ts = START + 100 * 12_000;
+ await mc.db(DB).collection(COLL).insertMany([
+ { _id: 'n_0', uid: 'u', did: 'd', ts, cd: null, sg: {}, c: 1 },
+ { _id: 'n_1', uid: 'u', did: 'd', ts: ts + 1_000, cd: null, sg: {}, c: 1 },
+ ] as never[]);
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('n_0', ts)] }); // one of two swept
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.audit?.mismatchedWindows.some((w) => w.lowerCd === 'null-cd sweep')).toBe(true);
+ await mc.db(DB).collection(COLL).deleteMany({ _id: { $in: ['n_0', 'n_1'] } } as never);
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'n_0'` });
+ });
+
it('a WHOLE window missing from the target → FAIL (the audit calls it pending, the check must not)', async () => {
// stale ledger says done, but every row of the window is gone from CH
await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id LIKE 'm\\_%'` });
From efe8a8206e49a56513cf93cbffb2efe2f246cf67 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 14:26:32 +0300
Subject: [PATCH 13/64] fix(ledger): claim-fenced bound apply, exact duplicate
verdicts, overflow-safe checksums, identity-coverage spot-check
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Fifth review round, including three findings that only appeared in review
BODIES (no inline thread):
- apply-bound is claim-fenced: refused while any pod holds an active chunk
claim (prune only touches PENDING chunks, so a claim racing the prune
would slip a post-bound chunk through), and a second prune after storing
the bound catches a straggler loudly instead of silently.
- verifyMigration's duplicate verdict uses an EXACT windowed count of ids
with ≥2 pre-boundary copies (per partition), not the 20-group display
sample that benign live duplicates could crowd.
- checksum sums are reduced mod 2^32 SERVER-SIDE on both stores: raw
residue sums pass 2^53 near ~2M rows/window and would round in JS,
turning identical data into false mismatches (or hiding real ones).
- identity-coverage spot-check: a doc swapped for another with the SAME cd
keeps both the count and the cd-sum — strided sampled distinct-id
presence (every 25th clean window, capped, DLQ-aware) now sees WHICH
documents exist; the deep check FAILs with a 'documents were swapped'
problem. Pinned by a same-cd ghost-swap test.
Co-Authored-By: Claude Fable 5
---
src/runtime/chunk-orchestrator.ts | 6 +++--
src/runtime/final-check.ts | 8 ++++--
src/runtime/ledger-engine.ts | 18 +++++++++++--
src/runtime/ledger-rebuild.ts | 34 +++++++++++++++++++++++-
src/target/staging-manager.ts | 15 ++++++-----
tests/integration/dedupe-overlap.test.ts | 23 ++++++++++++++++
tests/integration/final-check.test.ts | 18 +++++++++++++
7 files changed, 109 insertions(+), 13 deletions(-)
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index 11f04cb..807b708 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -2172,9 +2172,11 @@ export class ChunkOrchestrator {
// 2+ copies below → migration defect; verification fails.
const boundaryMs = all.reduce((m, c) => Math.max(m, c.upper_cd), 0);
const dup = await staging.duplicateStats(boundaryMs);
- let migrationDuplicates = 0;
+ // the verdict uses the EXACT per-partition count — the display sample is
+ // capped and early partitions full of benign live dups could crowd a
+ // real migration duplicate out of it
+ const migrationDuplicates = dup.migrationDuplicateGroups;
const duplicateSample = dup.sample.map((d) => {
- if (d.migratedCopies >= 2) migrationDuplicates++;
return {
_id: d._id,
copies: d.copies,
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 7397534..359f12b 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -187,6 +187,10 @@ export async function runFinalCheck(
if (audit.mismatchedWindows.length > 0) {
out.problems.push(`${fmt(audit.mismatchedWindows.length)} window(s) hold FEWER docs in ClickHouse than the source — data is missing from the target. Click "Retry failed chunks" after a rebuild, or escalate; do NOT decommission the old cluster.`);
}
+ if ((audit.idCoverageMissing ?? []).length > 0) {
+ const idMissingN = (audit.idCoverageMissing ?? []).reduce((a, w) => a + w.missing, 0);
+ out.problems.push(`${fmt((audit.idCoverageMissing ?? []).length)} window(s) hold the right COUNT and checksum but ${fmt(idMissingN)} sampled document identit${idMissingN === 1 ? 'y is' : 'ies are'} MISSING live — documents were swapped for others; escalate; do NOT decommission the old cluster.`);
+ }
if (audit.checksumMismatchWindows.length > 0) {
out.problems.push(`${fmt(audit.checksumMismatchWindows.length)} window(s) hold the right COUNT of the WRONG documents (checksum fingerprint differs) — escalate; do NOT decommission the old cluster.`);
}
@@ -197,8 +201,8 @@ export async function runFinalCheck(
if (audit.deletionDriftWindows.length > 0 && (audit.driftSubsetMissing ?? []).length === 0) {
out.notes.push(`${fmt(audit.deletionDriftWindows.length)} window(s) now hold MORE docs in ClickHouse than the source — the source shrank after migration (retention TTL / deletions). Sampled source ids in those windows were all found live, so the surplus is retained history, not masked gaps.`);
}
- if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0 && scopedPendingWindows === 0) {
- out.passes.push(`Recounted ${fmt(windows)} window(s) directly against the source: every count matches, every checksum fingerprint matches.`);
+ if (audit.mismatchedWindows.length === 0 && audit.checksumMismatchWindows.length === 0 && scopedPendingWindows === 0 && (audit.idCoverageMissing ?? []).length === 0) {
+ out.passes.push(`Recounted ${fmt(windows)} window(s) directly against the source: every count matches, every checksum fingerprint matches, and sampled identity coverage is complete.`);
}
if (cutoverMs !== null) {
const excluded = audit.excludedBeyondCutover ?? 0;
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 3ab1948..1576fa2 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -493,11 +493,25 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise= Date.now() - 60_000) return { applied: false, reason: 'bound must be safely in the past (>60s ago)' };
+ // Claim fence: prune deletes/clamps PENDING chunks only, so a pod
+ // claiming a post-bound chunk between the check and the delete would
+ // slip past it and migrate mirror territory. No active claims = no
+ // claiming in flight (pods hold at most their current chunk, and a
+ // paused/held fleet holds none).
+ const claims = await ledger.activeClaims(config.ledger.runId).catch(() => null);
+ if (claims === null) return { applied: false, reason: 'could not read active claims — retry when MongoDB answers' };
+ if (claims.length > 0) {
+ return { applied: false, reason: `pods hold active chunk claims (${claims.map((c) => `${c.pod}×${c.count}`).join(', ')}) — pause the pods, let in-flight chunks finish, then apply the bound` };
+ }
try {
const pruned = await ledger.pruneBeyondBound(config.ledger.runId, boundMs);
await ledger.setStoredBound(config.ledger.runId, boundMs, source);
- logger.warn({ boundMs, iso: new Date(boundMs).toISOString(), source, ...pruned }, 'Run bound applied — pods adopt it on their next map pass');
- return { applied: true, boundMs, iso: new Date(boundMs).toISOString(), ...pruned };
+ // belt: a claim raced in anyway → a second prune either cleans the
+ // still-pending stragglers or names the claimed chunk and fails loudly
+ const pruned2 = await ledger.pruneBeyondBound(config.ledger.runId, boundMs);
+ const total = { deleted: (pruned.deleted + pruned2.deleted), clamped: (pruned.clamped + pruned2.clamped) };
+ logger.warn({ boundMs, iso: new Date(boundMs).toISOString(), source, ...total }, 'Run bound applied — pods adopt it on their next map pass');
+ return { applied: true, boundMs, iso: new Date(boundMs).toISOString(), ...total };
} catch (err) {
return { applied: false, reason: (err as Error).message };
}
diff --git a/src/runtime/ledger-rebuild.ts b/src/runtime/ledger-rebuild.ts
index efb4bac..20af3d7 100644
--- a/src/runtime/ledger-rebuild.ts
+++ b/src/runtime/ledger-rebuild.ts
@@ -67,6 +67,8 @@ export interface RebuildProgress {
excludedBeyondCutover?: number;
/** Drift windows (live > source) whose sampled source ids were NOT all found in the target — surplus rows were masking missing ones. */
driftSubsetMissing?: Array<{ collection: string; lowerCd: string; upperCd: string; sampled: number; missing: number }>;
+ /** Count-exact, checksum-clean windows where sampled source ids are missing live — documents swapped for others. */
+ idCoverageMissing?: Array<{ collection: string; lowerCd: string; upperCd: string; sampled: number; missing: number }>;
error: string | null;
startedAt: number | null;
finishedAt: number | null;
@@ -129,6 +131,7 @@ export async function rebuildLedger(opts: {
const db = mongo.db(config.source.db);
let driftChecks = 0;
+ let idChecks = 0;
progress.phase = 'discovering collections';
let collections = await discoverCollections(db, config.source.collectionPrefix, logger);
const skipEventNames = new Set(['[CLY]_apm_device', '[CLY]_apm_network']);
@@ -205,9 +208,13 @@ export async function rebuildLedger(opts: {
progress.phase = `counting ${collection} chunk ${idx + 1}/${bounds.length}`;
// count + cd-sum in one index-covered pass: the sum is an order-free
// fingerprint of WHICH docs the window holds, not just how many
+ // the SUM is reduced mod 2^32 server-side on BOTH stores: raw sums
+ // of 32-bit residues pass 2^53 near ~2M rows/window and would round
+ // in JS — mod-space comparison stays exact at any window size
const [mongoAgg] = await coll.aggregate<{ n: number; sumCd: number }>([
{ $match: { cd: { $gte: new Date(b.lowerCd), $lt: new Date(b.upperCd) } } },
{ $group: { _id: null, n: { $sum: 1 }, sumCd: { $sum: { $mod: [{ $toLong: '$cd' }, 4294967296] } } } },
+ { $project: { n: 1, sumCd: { $mod: ['$sumCd', 4294967296] } } },
]).toArray();
const mongoCount = mongoAgg?.n ?? 0;
const mongoSumCd = mongoAgg?.sumCd ?? 0;
@@ -220,7 +227,8 @@ export async function rebuildLedger(opts: {
let sweptSum = 0;
for (let i = lo; i < sweptCds.length && sweptCds[i] < b.upperCd; i++) { sweptIn++; sweptSum += sweptCds[i] % 4294967296; } // same mod as both fingerprints
const live = liveRaw - sweptIn;
- const liveSumCd = liveAgg.sumCd - sweptSum;
+ const MOD = 4294967296;
+ const liveSumCd = (((liveAgg.sumCd - (sweptSum % MOD)) % MOD) + MOD) % MOD;
// Docs in this window that are KNOWN unmigrated (pending/waived DLQ)
// legitimately explain source > live — without this, a window whose
@@ -273,6 +281,30 @@ export async function rebuildLedger(opts: {
}
}
}
+ // Identity coverage: a doc swapped for ANOTHER doc with the same cd
+ // keeps the count AND the cd-sum — sampled distinct-id presence is
+ // the axis that sees WHICH documents exist. Strided (every 25th
+ // clean window, capped) so big audits stay affordable; drift windows
+ // are always id-checked above.
+ if (checkOnly && !unscopableInMulti && live + unresolved === mongoCount && live > 0
+ && idChecks < 300 && idx % 25 === 0) {
+ idChecks++;
+ const idSample = (await coll
+ .find({ cd: { $gte: new Date(b.lowerCd), $lt: new Date(b.upperCd) } }, { projection: { _id: 1 } })
+ .limit(5_000).toArray()).map((d) => String(d._id));
+ const idPresent = await staging.countDistinctMatchingIdsInWindow(idSample, b.lowerCd, b.upperCd, scope);
+ // sampled ids that are themselves DLQ'd are legitimately absent
+ const idUnresolved = unresolved > 0
+ ? await dlq.countUnresolvedMatchingIds(runId, collection, idSample, b.lowerCd, b.upperCd)
+ : 0;
+ const idMissing = idSample.length - idPresent - idUnresolved;
+ if (idMissing > 0 && (progress.idCoverageMissing ?? []).length < 200) {
+ (progress.idCoverageMissing ?? (progress.idCoverageMissing = [])).push({
+ collection, lowerCd: new Date(b.lowerCd).toISOString(), upperCd: new Date(b.upperCd).toISOString(),
+ sampled: idSample.length, missing: idMissing,
+ });
+ }
+ }
// Checksum: only meaningful on windows that are count-exact with no
// DLQ residue — equal counts hiding DIFFERENT docs is the one error
// class pure counting cannot see. Number-safety: cd sums stay well
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index 865999b..da6bcaa 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -440,6 +440,8 @@ export class StagingManager {
*/
async duplicateStats(boundaryMs: number, sampleLimit = 20): Promise<{
rows: number; duplicates: number;
+ /** EXACT count of ids with ≥2 pre-boundary copies — never derived from the display sample. */
+ migrationDuplicateGroups: number;
sample: Array<{ _id: string; copies: number; migratedCopies: number; min_cd_ms: number; max_cd_ms: number }>;
}> {
const parts = await this.ch().query({
@@ -450,14 +452,15 @@ export class StagingManager {
format: 'JSONEachRow',
});
const partitions = await parts.json<{ partition: string; r: string }>();
- let rows = 0, duplicates = 0;
+ let rows = 0, duplicates = 0, migrationDuplicateGroups = 0;
const sample: Array<{ _id: string; copies: number; migratedCopies: number; min_cd_ms: number; max_cd_ms: number }> = [];
for (const p of partitions) {
rows += Number(p.r);
const res = await this.ch().query({
query: `SELECT _id, count() AS c, countIf(cd < fromUnixTimestamp64Milli({b:Int64})) AS mc,
toUnixTimestamp64Milli(min(cd)) AS lo, toUnixTimestamp64Milli(max(cd)) AS hi,
- sum(c - 1) OVER () AS excess
+ sum(c - 1) OVER () AS excess,
+ sum(mc >= 2) OVER () AS mg
FROM (SELECT _id, cd FROM ${this.fq(this.config.table)} WHERE _partition_id = {p:String})
GROUP BY _id HAVING c > 1
ORDER BY mc DESC, c DESC LIMIT {lim:UInt32}`,
@@ -465,14 +468,14 @@ export class StagingManager {
format: 'JSONEachRow',
clickhouse_settings: { max_bytes_before_external_group_by: '4000000000' },
});
- const groups = await res.json<{ _id: string; c: string; mc: string; lo: string; hi: string; excess: string }>();
- if (groups.length > 0) duplicates += Number(groups[0].excess);
+ const groups = await res.json<{ _id: string; c: string; mc: string; lo: string; hi: string; excess: string; mg: string }>();
+ if (groups.length > 0) { duplicates += Number(groups[0].excess); migrationDuplicateGroups += Number(groups[0].mg); }
for (const g of groups) {
if (sample.length >= sampleLimit) break;
sample.push({ _id: g._id, copies: Number(g.c), migratedCopies: Number(g.mc), min_cd_ms: Number(g.lo), max_cd_ms: Number(g.hi) });
}
}
- return { rows, duplicates, sample };
+ return { rows, duplicates, migrationDuplicateGroups, sample };
}
@@ -536,7 +539,7 @@ export class StagingManager {
*/
async countAndSumLiveCdRange(lowerCdMs: number, upperCdMs: number, scope?: { a: string; e: string; n?: string } | null): Promise<{ n: number; sumCd: number }> {
const res = await this.ch().query({
- query: `SELECT count() AS c, sum(toUnixTimestamp64Milli(cd) % 4294967296) AS s FROM ${this.fq(this.config.table)}
+ query: `SELECT count() AS c, toUInt64(sum(toUnixTimestamp64Milli(cd) % 4294967296)) % 4294967296 AS s FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64})
AND cd < fromUnixTimestamp64Milli({hi:Int64})
${this.scopeSql(scope)}`,
diff --git a/tests/integration/dedupe-overlap.test.ts b/tests/integration/dedupe-overlap.test.ts
index eb305fa..141c181 100644
--- a/tests/integration/dedupe-overlap.test.ts
+++ b/tests/integration/dedupe-overlap.test.ts
@@ -16,6 +16,7 @@ import { MongoClient } from 'mongodb';
import { createClient, type ClickHouseClient } from '@clickhouse/client';
import { runDedupeOverlap, newDedupeOverlapState } from '../../src/runtime/dedupe-overlap.ts';
+import { StagingManager } from '../../src/target/staging-manager.ts';
import { HashResolver } from '../../src/transform/hash-resolver.ts';
import { loadConfig } from '../../src/config/loader.ts';
import type { Config } from '../../src/config/schema.ts';
@@ -188,4 +189,26 @@ describe('tee-overlap dedupe', () => {
expect(await chCount("_id = 'mirror_10' AND a = 'sibling_app'")).toBe(1);
expect(await chCount("_id = 'mirror_10'")).toBe(1);
});
+
+ it('duplicateStats counts migration-duplicate groups exactly, beyond the display-sample cap', async () => {
+ // 25 duplicated ids below the boundary — more than the 20-group sample
+ const rows: Record[] = [];
+ for (let i = 0; i < 25; i++) {
+ const cd = FLIP - 7_200_000 + i * 1_000;
+ rows.push(chRow(`dupg_${i}`, cd), chRow(`dupg_${i}`, cd + 1));
+ }
+ await ch.insert({ table: `${DB}.drill_events`, values: rows, format: 'JSONEachRow' });
+ const staging = new StagingManager(
+ { url: CH_URL, database: DB, table: 'drill_events', username: 'default', password: CH_PASSWORD, queryTimeoutMs: 30_000 },
+ logger,
+ );
+ await staging.connect();
+ try {
+ const stats = await staging.duplicateStats(Date.now());
+ expect(stats.migrationDuplicateGroups).toBe(25);
+ expect(stats.sample.length).toBeLessThanOrEqual(20);
+ } finally {
+ await staging.close();
+ }
+ });
});
diff --git a/tests/integration/final-check.test.ts b/tests/integration/final-check.test.ts
index 062043f..c839a1e 100644
--- a/tests/integration/final-check.test.ts
+++ b/tests/integration/final-check.test.ts
@@ -275,6 +275,24 @@ describe('final check: the interpreted sign-off', () => {
await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'n_0'` });
});
+ it('a same-cd identity swap (count AND checksum survive) is caught by id coverage', async () => {
+ // restore the docs the drift test removed so the window is clean again
+ await mc.db(DB).collection(COLL).insertMany(
+ [50, 51, 52, 53, 54, 55, 56, 57].map((i) => ({
+ _id: `m_${i}`, uid: 'u', did: 'd', ts: START + i * 12_000, cd: new Date(START + i * 12_000), sg: {}, c: 1,
+ })) as never[],
+ );
+ const cd = START + 80 * 12_000;
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'm_80'` });
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('ghost_80', cd)] });
+ const out = await check({ cutoverMs: CUTOVER });
+ expect(out.verdict).toBe('FAIL');
+ expect(out.problems.join(' ')).toContain('swapped');
+ expect(out.audit?.checksumMismatchWindows).toEqual([]); // the swap is invisible to the checksum by design
+ await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id = 'ghost_80'` });
+ await ch.insert({ table: `${DB}.drill_events`, format: 'JSONEachRow', values: [chRow('m_80', cd)] });
+ });
+
it('a WHOLE window missing from the target → FAIL (the audit calls it pending, the check must not)', async () => {
// stale ledger says done, but every row of the window is gone from CH
await ch.command({ query: `DELETE FROM ${DB}.drill_events WHERE _id LIKE 'm\\_%'` });
From 26cf33e7defd130d51710c25f94b9be504bd50e1 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 14:30:28 +0300
Subject: [PATCH 14/64] fix(ledger): epoch-ms validation on bound application,
NaN-proof sample counts
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
- applyBoundNow (both boundary endpoints) rejects epoch-seconds values and
future timestamps before pruning — a 1970-era bound would have deleted
every pending chunk and persisted a nonsense cutover.
- samples inputs on final-check and audit-content endpoints fall back to
the default on non-numeric JSON instead of propagating NaN (which made
contentAudit sample zero documents while quick mode still passed);
contentAudit itself also floors invalid values.
Co-Authored-By: Claude Fable 5
---
src/runtime/chunk-orchestrator.ts | 1 +
src/runtime/ledger-engine.ts | 25 ++++++++++++++-----------
2 files changed, 15 insertions(+), 11 deletions(-)
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index 807b708..63ac624 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -1641,6 +1641,7 @@ export class ChunkOrchestrator {
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}> {
const { config, staging } = this.d;
+ if (!Number.isFinite(samplesPerCollection) || samplesPerCollection <= 0) samplesPerCollection = 500;
const p = this.contentAuditProgress;
p.running = true; p.sampled = 0; p.matched = 0; p.mismatches = [];
try {
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 1576fa2..5b8a0fb 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -372,7 +372,7 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise 0) return { started: false, reason: `other pods are actively migrating (${busyCnt.map((row) => row.pod).join(', ')}) — a mid-run audit reports false mismatches; audit after completion` };
- const samples = Math.min(10_000, Math.max(50, req.body?.samples ?? 500));
+ const samples = Math.min(10_000, Math.max(50, typeof req.body?.samples === 'number' && Number.isFinite(req.body.samples) ? req.body.samples : 500));
auditContentState.status = 'running'; auditContentState.result = null; auditContentState.error = null;
void orchestrator.contentAudit(samples)
.then((r) => { auditContentState.result = r as unknown as Record; auditContentState.status = 'completed'; })
@@ -383,14 +383,6 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise {
- if (typeof v !== 'number' || !Number.isFinite(v)) return `${name} (epoch ms) required`;
- if (v < 1_000_000_000_000) return `${name}=${v} looks like epoch SECONDS — pass milliseconds (×1000)`;
- if (v > Date.now() + 60_000) return `${name} is in the future`;
- return null;
- };
const finalCheckState: FinalCheckResult = newFinalCheckResult();
app.post<{ Body: { cutoverMs?: number; samples?: number; deep?: boolean } }>('/control/final-check', async (req) => {
if (finalCheckState.status === 'running') return { started: false, reason: 'final check already running' };
@@ -410,7 +402,7 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise {
+ if (typeof v !== 'number' || !Number.isFinite(v)) return `${name} (epoch ms) required`;
+ if (v < 1_000_000_000_000) return `${name}=${v} looks like epoch SECONDS — pass milliseconds (×1000)`;
+ if (v > Date.now() + 60_000) return `${name} is in the future`;
+ return null;
+ };
+
let boundaryApplied: Record | null = null;
const applyBoundNow = async (boundMs: number, source: string): Promise> => {
- if (!Number.isFinite(boundMs) || boundMs <= 0) return { applied: false, reason: 'boundMs (epoch ms) required' };
+ const msErr = epochMsError(boundMs, 'boundMs');
+ if (msErr) return { applied: false, reason: msErr };
if (config.ledger.dryRun) return { applied: false, reason: 'dry run — apply on the real run' };
if (envBoundAtBoot !== null) {
return { applied: false, reason: `bound already pinned via LEDGER_CD_UPPER_BOUND=${envBoundAtBoot} — change it in the deployment config, not here` };
From 6c6168557c385886d1ad0147657253993cc601d8 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 15:05:38 +0300
Subject: [PATCH 15/64] =?UTF-8?q?fix(ledger):=20page=20id=20query-params?=
=?UTF-8?q?=20at=202,000=20=E2=80=94=20ClickHouse=20form-field=20limit?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Field report: the audit's id spot-check died with 'Poco::Exception …
HTML Form Exception: Field value too long'. ClickHouse receives query
params as HTTP form fields capped by http_max_field_value_size (128 KiB
default), and a 5,000-id Array(String) parameter is ~130 KB. All
id-parameter queries (fetchLiveCdByIds, count/distinct/delete matching
ids) now page at 2,000 ids (~52 KB) — safe under default limits.
Co-Authored-By: Claude Fable 5
---
src/target/staging-manager.ts | 25 +++++++++++++++++--------
1 file changed, 17 insertions(+), 8 deletions(-)
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index da6bcaa..efb5785 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -327,6 +327,15 @@ export class StagingManager {
* Used when retrying a chunk that was already (partially) promoted — redo
* must start from a clean window or verify-then-attach would skip it.
*/
+ /**
+ * Ids per query when passed as a {ids:Array(String)} parameter. ClickHouse
+ * receives query params as HTTP form fields capped by
+ * http_max_field_value_size (128 KiB default): ~5,000 ObjectId strings
+ * already trip "HTML Form Exception: Field value too long" (field report).
+ * 2,000 ids ≈ 52 KB — safely under default limits everywhere.
+ */
+ private static readonly ID_PARAM_PAGE = 2_000;
+
private scopeSql(scope?: { a: string; e: string; n?: string } | null): string {
if (!scope) return '';
return 'AND a = {sa:String} AND e = {se:String}' + (scope.n !== undefined ? ' AND n = {sn:String}' : '');
@@ -562,8 +571,8 @@ export class StagingManager {
? 'AND cd >= fromUnixTimestamp64Milli({blo:Int64}) AND cd <= fromUnixTimestamp64Milli({bhi:Int64})'
: '';
const out = new Map();
- for (let i = 0; i < ids.length; i += 10_000) {
- const page = ids.slice(i, i + 10_000);
+ for (let i = 0; i < ids.length; i += StagingManager.ID_PARAM_PAGE) {
+ const page = ids.slice(i, i + StagingManager.ID_PARAM_PAGE);
const res = await this.ch().query({
query: `SELECT _id, toUnixTimestamp64Milli(cd) AS cd_ms FROM ${this.fq(this.config.table)}
WHERE _id IN {ids:Array(String)} ${bound}`,
@@ -590,8 +599,8 @@ export class StagingManager {
/** DISTINCT given ids present live in [fromMs, toMs) — duplicate rows of one id never vouch for another id's absence. Scope keeps a same-_id row in a SIBLING collection from vouching either. */
async countDistinctMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
let total = 0;
- for (let i = 0; i < ids.length; i += 50_000) {
- const page = ids.slice(i, i + 50_000);
+ for (let i = 0; i < ids.length; i += StagingManager.ID_PARAM_PAGE) {
+ const page = ids.slice(i, i + StagingManager.ID_PARAM_PAGE);
const res = await this.ch().query({
// alias must not be 'n' — the scope filter references the real column n
query: `SELECT uniqExact(_id) AS cnt FROM ${this.fq(this.config.table)}
@@ -609,8 +618,8 @@ export class StagingManager {
/** Live rows in [fromMs, toMs) whose _id is one of the given ids, scoped to a collection's (a,e,n) when known. */
async countMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
let total = 0;
- for (let i = 0; i < ids.length; i += 50_000) {
- const page = ids.slice(i, i + 50_000);
+ for (let i = 0; i < ids.length; i += StagingManager.ID_PARAM_PAGE) {
+ const page = ids.slice(i, i + StagingManager.ID_PARAM_PAGE);
const res = await this.ch().query({
query: `SELECT count() AS cnt FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
@@ -631,8 +640,8 @@ export class StagingManager {
* keeps each DELETE partition-prunable on multi-billion-row tables.
*/
async deleteMatchingIdsInWindow(ids: string[], fromMs: number, toMs: number, scope?: { a: string; e: string; n?: string } | null): Promise {
- for (let i = 0; i < ids.length; i += 50_000) {
- const page = ids.slice(i, i + 50_000);
+ for (let i = 0; i < ids.length; i += StagingManager.ID_PARAM_PAGE) {
+ const page = ids.slice(i, i + StagingManager.ID_PARAM_PAGE);
await this.ch().command({
query: `DELETE FROM ${this.fq(this.config.table)}
WHERE cd >= fromUnixTimestamp64Milli({lo:Int64}) AND cd < fromUnixTimestamp64Milli({hi:Int64})
From b489a8d733b00af9347cacbf91f07c5f5866eef5 Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 15:09:06 +0300
Subject: [PATCH 16/64] =?UTF-8?q?docs(runbook):=20what=20quick=20vs=20deep?=
=?UTF-8?q?=20can=20and=20cannot=20see=20=E2=80=94=20deep=20is=20the=20dec?=
=?UTF-8?q?ommissioning=20gate,=20not=20routine=20distrust?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Co-Authored-By: Claude Fable 5
---
docs/RUNBOOK.md | 22 ++++++++++++++++++----
1 file changed, 18 insertions(+), 4 deletions(-)
diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md
index f74f5eb..e6f2772 100644
--- a/docs/RUNBOOK.md
+++ b/docs/RUNBOOK.md
@@ -105,10 +105,24 @@ Two tiers:
source. Catches everything that can happen AFTER reading. Capped at
PASS WITH NOTES — the note names what it did not re-prove.
- **Deep** (opt-in — hours on large runs): additionally recounts EVERY
- window against the source with cd-checksum fingerprints. This is the one
- check that would catch a self-consistently under-reading reader, so run
- it once before the source is deleted; while the source still exists,
- quick is enough for routine confidence.
+ window against the source with cd-checksum fingerprints and sampled
+ identity coverage. It is not distrust of the ledger — chunk reads are
+ already recounted against the source at migration time — it is the only
+ check that derives everything from the two databases alone, with zero
+ reliance on the tool's own records. Run it once, as the gate before the
+ source is deleted; while the source exists, quick is enough.
+
+What each layer can and cannot see:
+
+| Failure class | Caught by |
+|---|---|
+| Under-read at read time (source count ≠ read tally) | the migration itself, per chunk (source-count guard) |
+| Rows lost or duplicated in ClickHouse after attach | quick — ledger-vs-target verify |
+| Skipped documents | DLQ accounting (both tiers; unresolved = FAIL) |
+| Wrong content in migrated rows | quick — random content samples vs source |
+| Docs written into already-done windows later (imports, restores, backdated cds) | deep only — the ledger is blind to them by design |
+| Count-preserving identity swaps (same count, same cd-sum, different docs) | deep only — checksum + sampled id coverage |
+| Routine retention deleting source docs (drift) | deep classifies it exactly (spot-checked, never assumed benign) |
- Dashboard: the **Final check** card → *Run final check* (tick *deep source
recount* for the pre-teardown gate). On tee/mirror runs without a stored
From 1e304c9944970913460eaa012873a075ec55e48b Mon Sep 17 00:00:00 2001
From: Arturs Sosins
Date: Mon, 21 Sep 2026 16:09:02 +0300
Subject: [PATCH 17/64] fix(ledger): verify in both tiers with cutover
awareness, sweep-row verification, bound rollback, persistent boundary
question
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Sixth review round:
- verifyMigration runs in BOTH final-check tiers (deep alone misread a
migration duplicate as retention drift), takes a cutover and SKIPS
post-cutover chunks (their windows mix native rows into the same scope —
and after a tee-overlap dedupe their migrated rows were deleted on
purpose; the prescribed post-cleanup check no longer false-fails).
- null-cd sweep rows are verified directly against the source's cd:null
ids (scoped fetchLiveCdByIds): losing swept rows after attach was
invisible to window counts, which tolerate the sweep by design.
- apply-bound rolls back: after storing, a second prune + claims re-check
detect a raced claim and CLEAR the stored bound instead of leaving a
half-applied bound behind a racing worker; a guard-held engine resumes
automatically when a bound applies or no-mirror is declared.
- the boundary guard no longer treats mapped state or a passing startup
probe as a permanent license: only a bound or the explicit no-mirror ack
settles the question, restarts re-ask it when the target is live, and a
5-minute probe re-raises it if the target BEGINS ingesting mid-run.
- unscopable-in-multi collections get strided unscoped id coverage in the
deep audit (their only per-window evidence), feeding the same
idCoverageMissing FAIL.
Co-Authored-By: Claude Fable 5
---
src/runtime/chunk-orchestrator.ts | 68 +++++++++++++++++++++++++--
src/runtime/final-check.ts | 20 ++++----
src/runtime/ledger-engine.ts | 26 +++++++---
src/runtime/ledger-rebuild.ts | 22 ++++++++-
src/state/ledger-store.ts | 5 ++
src/target/staging-manager.ts | 8 ++--
tests/integration/final-check.test.ts | 1 +
7 files changed, 127 insertions(+), 23 deletions(-)
diff --git a/src/runtime/chunk-orchestrator.ts b/src/runtime/chunk-orchestrator.ts
index 63ac624..0c0f013 100644
--- a/src/runtime/chunk-orchestrator.ts
+++ b/src/runtime/chunk-orchestrator.ts
@@ -144,6 +144,7 @@ export class ChunkOrchestrator {
private probeOkStreak = 0;
private autoResuming = false;
private resumeProbeTimer: NodeJS.Timeout | null = null;
+ private guardProbeTimer: NodeJS.Timeout | null = null;
private lastReclaimAt = 0;
private monitorTimer: ReturnType | null = null;
@@ -222,8 +223,9 @@ export class ChunkOrchestrator {
try {
if ((await this.d.ledger.getStoredBound(this.runId)) !== null) return 'proceed';
if (await this.d.ledger.getUnboundedAck(this.runId)) return 'proceed';
- const counts = await this.d.ledger.statusCounts(this.runId);
- if (Object.values(counts).reduce((a, b) => a + b, 0) > 0) return 'proceed'; // resumed run: decided already
+ // no shortcut for runs with mapped state: only a bound or an explicit
+ // no-mirror answer settles the question — restarts re-ask it when
+ // the target is live (one click; the ack persists cluster-wide)
const live = await this.d.staging.hasLiveCdSince(Date.now() - GUARD_LIVE_LOOKBACK_MS);
return live ? 'hold' : 'proceed';
} catch (err) {
@@ -308,6 +310,27 @@ export class ChunkOrchestrator {
}, 15_000);
this.resumeProbeTimer.unref?.();
+ // The boundary question does not expire at startup: a mirror that comes
+ // online MID-RUN (target liveness appearing later) re-raises it — the
+ // startup probe passing once is not a permanent license to run unbounded.
+ if (!this.dryRun) {
+ this.guardProbeTimer = setInterval(() => {
+ void (async () => {
+ try {
+ if (this.status !== 'running' || this.paused) return;
+ if (this.d.config.ledger.cdUpperBoundMs != null || this.d.config.ledger.unboundedOk) return;
+ if ((await this.d.ledger.getStoredBound(this.runId)) !== null) return;
+ if (await this.d.ledger.getUnboundedAck(this.runId)) return;
+ if (await this.d.staging.hasLiveCdSince(Date.now() - GUARD_LIVE_LOOKBACK_MS)) {
+ this.pause('boundary-unset');
+ this.logger.warn('GUARD: the target began receiving live data mid-run with no bound set — answer the mirror question (set-boundary or allow-unbounded) to continue');
+ }
+ } catch { /* transient — next tick re-checks */ }
+ })();
+ }, 300_000);
+ this.guardProbeTimer.unref?.();
+ }
+
if (this.dryRun) {
await this.d.staging.createDryRunTable();
this.logger.warn(
@@ -463,6 +486,7 @@ export class ChunkOrchestrator {
if (this.monitorTimer) clearInterval(this.monitorTimer);
if (this.resumeProbeTimer) clearInterval(this.resumeProbeTimer);
+ if (this.guardProbeTimer) clearInterval(this.guardProbeTimer);
this.status = this.stopping ? 'stopped' : 'completed';
this.finishedAt = Date.now();
if (this.status === 'completed' && !this.dryRun) {
@@ -2122,7 +2146,7 @@ export class ChunkOrchestrator {
* against its verified expectation, plus table totals. Exact, minutes at
* most — run before sign-off or any time trust is in question.
*/
- async verifyMigration(): Promise> {
+ async verifyMigration(upToMs: number | null = null): Promise> {
const { ledger, staging } = this.d;
const all = await ledger.listAll(this.runId);
const byCollection = new Map();
@@ -2132,6 +2156,7 @@ export class ChunkOrchestrator {
let checked = 0;
let unscopedSkipped = 0;
+ let pastCutoverSkipped = 0;
const collectionCount = new Set(all.map((c) => c.collection)).size;
const mismatches: Array<{ chunk: string; expected: number; live: number }> = [];
const targets = all.filter((chunk) => chunk.status === 'done' && !this.isNullCdChunk(chunk as ChunkDoc));
@@ -2152,6 +2177,12 @@ export class ChunkOrchestrator {
const chunk = targets[i];
const scope = this.scopeOf(chunk as ChunkDoc);
if (!scope && collectionCount > 1) { unscopedSkipped++; continue; }
+ // Chunks past the cutover cannot be count-compared at all: their
+ // windows mix natively-ingested rows into the same (a,e,n) scope,
+ // and after a tee-overlap dedupe their migrated rows were deleted
+ // on purpose. Skip and report them — the cutover-scoped region is
+ // what this verification vouches for.
+ if (upToMs !== null && chunk.upper_cd > upToMs) { pastCutoverSkipped++; continue; }
const live = await staging.countLiveInCdRange(chunk.lower_cd, chunk.upper_cd, scope);
const relaxed = byCollection.get(chunk.collection) === true;
const bad = relaxed ? live < chunk.rows_expected : live !== chunk.rows_expected;
@@ -2161,6 +2192,36 @@ export class ChunkOrchestrator {
}
}));
+ // Null-cd sweep rows live INSIDE regular windows at ts-derived cds, so
+ // the window comparison above deliberately tolerates them — which also
+ // means their loss would be invisible. Verify them directly: the
+ // source's cd:null ids must still exist live (scoped when possible).
+ this.verifyProgress.phase = 'verifying null-cd sweep rows';
+ const db = this.d.mongoReader.getDatabase();
+ for (const chunk of all) {
+ if (!this.isNullCdChunk(chunk as ChunkDoc) || chunk.status !== 'done' || chunk.rows_expected <= 0) continue;
+ const idDocs = await db.collection(chunk.collection)
+ .find({ cd: null }, { projection: { _id: 1, ts: 1 } }).limit(1_000_000).toArray();
+ if (idDocs.length === 0) continue;
+ let lo = Infinity, hi = -Infinity;
+ const sweepIds: string[] = [];
+ for (const d of idDocs) {
+ sweepIds.push(String(d._id));
+ const tsMs = toEpochMillis(d.ts);
+ if (tsMs !== null && tsMs > 0) {
+ const c = clampDateTime64(tsMs);
+ if (c < lo) lo = c;
+ if (c > hi) hi = c;
+ }
+ }
+ const scope = this.scopeOf(chunk as ChunkDoc);
+ const liveSweep = await this.d.staging.fetchLiveCdByIds(sweepIds, lo <= hi ? { loMs: lo, hiMs: hi } : undefined, scope);
+ checked++;
+ if (liveSweep.size < chunk.rows_expected) {
+ mismatches.push({ chunk: chunk._id, expected: chunk.rows_expected, live: liveSweep.size });
+ }
+ }
+
this.verifyProgress.phase = 'scanning for duplicates (per partition)';
// Duplicate detection + attribution, partition by partition — exact for
@@ -2196,6 +2257,7 @@ export class ChunkOrchestrator {
ok: mismatches.length === 0 && migrationDuplicates === 0,
checkedChunks: checked,
unscopedSkipped,
+ pastCutoverSkipped,
mismatches,
table: { rows: dup.rows, distinctIds: dup.rows - dup.duplicates, duplicates: dup.duplicates },
duplicateSample,
diff --git a/src/runtime/final-check.ts b/src/runtime/final-check.ts
index 359f12b..388ab14 100644
--- a/src/runtime/final-check.ts
+++ b/src/runtime/final-check.ts
@@ -64,7 +64,7 @@ interface ContentAuditRunner {
sampled: number; matched: number; missing: number; different: number;
mismatches: Array<{ _id: string; collection: string; kind: string; fields?: string[] }>;
}>;
- verifyMigration(): Promise>;
+ verifyMigration(upToMs?: number | null): Promise>;
}
export async function runFinalCheck(
@@ -135,16 +135,14 @@ export async function runFinalCheck(
}
if (dlqPending === 0 && dlqWaived === 0) out.passes.push('Dead-letter queue is empty — no document was skipped.');
- // ── 3a. QUICK tier: target vs the run's own ledger (minutes) ──────────
+ // ── 3a. Target vs the run's own ledger — BOTH tiers ────────────────────
// Catches everything that happened AFTER reading: lost partitions, rows
- // deleted from the live table, duplicate attribution. What it cannot see
- // is a self-consistently under-reading reader — the ledger agreeing with
- // itself while the source held more. That class is covered
- // probabilistically by the content samples below, and exactly by the
- // deep recount — which is why quick mode never returns a plain PASS.
- if (!deep) {
+ // deleted from the live table, and exact duplicate attribution (which
+ // the deep recount alone would misread as retention drift). Cutover-
+ // aware: post-cutover windows mix native rows and are skipped here.
+ {
out.phase = 'verifying the target against the run ledger';
- const verify = await deps.orchestrator.verifyMigration();
+ const verify = await deps.orchestrator.verifyMigration(cutoverMs);
const vMism = (verify.mismatches as Array> | undefined) ?? [];
const vDup = Number((verify as Record).migrationDuplicates ?? 0);
if (verify.ok !== true) {
@@ -160,7 +158,9 @@ export async function runFinalCheck(
} else {
out.passes.push('Target verified against the run ledger: every migrated chunk window holds exactly the recorded row count, with no migration-side duplicates.');
}
- out.notes.push(`Quick mode: the ledger itself was not re-proven against the source. Per-chunk verification at attach time plus the random content samples below cover that class probabilistically — run the DEEP check ({"deep": true}, or the checkbox in the dashboard) before deleting the source if you want the full recount + checksum fingerprints.`);
+ if (!deep) {
+ out.notes.push(`Quick mode: the ledger itself was not re-proven against the source. Per-chunk verification at attach time plus the random content samples below cover that class probabilistically — run the DEEP check ({"deep": true}, or the checkbox in the dashboard) before deleting the source if you want the full recount + checksum fingerprints.`);
+ }
}
// ── 3b. DEEP tier: full source recount + cd-checksum fingerprint ──────
diff --git a/src/runtime/ledger-engine.ts b/src/runtime/ledger-engine.ts
index 5b8a0fb..bde8571 100644
--- a/src/runtime/ledger-engine.ts
+++ b/src/runtime/ledger-engine.ts
@@ -509,12 +509,25 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise 0) {
+ throw new Error(`pods claimed chunks during apply (${claimsAfter.map((c) => `${c.pod}×${c.count}`).join(', ')})`);
+ }
+ const total = { deleted: (pruned.deleted + pruned2.deleted), clamped: (pruned.clamped + pruned2.clamped) };
+ logger.warn({ boundMs, iso: new Date(boundMs).toISOString(), source, ...total }, 'Run bound applied — pods adopt it on their next map pass');
+ // a guard-held engine has its answer now
+ if (orchestrator.getStats().pauseReason === 'boundary-unset') orchestrator.resume();
+ return { applied: true, boundMs, iso: new Date(boundMs).toISOString(), ...total };
+ } catch (raceErr) {
+ await ledger.clearStoredBound(config.ledger.runId).catch(() => {});
+ return { applied: false, reason: `apply raced concurrent claiming and was ROLLED BACK (${(raceErr as Error).message}) — pause all pods, let in-flight chunks finish, then apply again` };
+ }
} catch (err) {
return { applied: false, reason: (err as Error).message };
}
@@ -552,6 +565,7 @@ export async function runLedgerEngine(config: Config, logger: Logger): Promise {
await ledger.setUnboundedAck(config.ledger.runId, config.worker.podId);
+ if (orchestrator.getStats().pauseReason === 'boundary-unset') orchestrator.resume();
logger.warn('Operator declared no-mirror: unbounded run allowed — held pods release within seconds');
return { allowed: true, note: 'held pods release within ~3s; the decision is stored cluster-wide in mig_run_config' };
});
diff --git a/src/runtime/ledger-rebuild.ts b/src/runtime/ledger-rebuild.ts
index 20af3d7..73f794e 100644
--- a/src/runtime/ledger-rebuild.ts
+++ b/src/runtime/ledger-rebuild.ts
@@ -180,7 +180,7 @@ export async function rebuildLedger(opts: {
}
summary.nullCdDocs = nullCdIds.length;
const liveNullCd = nullCdIds.length > 0
- ? await staging.fetchLiveCdByIds(nullCdIds, derivedLo <= derivedHi ? { loMs: derivedLo, hiMs: derivedHi } : undefined)
+ ? await staging.fetchLiveCdByIds(nullCdIds, derivedLo <= derivedHi ? { loMs: derivedLo, hiMs: derivedHi } : undefined, scope)
: new Map();
summary.nullCdSwept = liveNullCd.size;
const sweptCds = [...liveNullCd.values()].sort((a, b) => a - b);
@@ -281,6 +281,26 @@ export async function rebuildLedger(opts: {
}
}
}
+ // Unscopable collection among others: no per-window recount exists,
+ // so sampled id coverage is the ONLY per-window evidence — run it on
+ // the same stride (unscoped lookup; _ids are effectively unique).
+ if (checkOnly && unscopableInMulti && mongoCount > 0 && idChecks < 300 && idx % 25 === 0) {
+ idChecks++;
+ const uSample = (await coll
+ .find({ cd: { $gte: new Date(b.lowerCd), $lt: new Date(b.upperCd) } }, { projection: { _id: 1 } })
+ .limit(5_000).toArray()).map((d) => String(d._id));
+ const uPresent = await staging.countDistinctMatchingIdsInWindow(uSample, b.lowerCd, b.upperCd, null);
+ const uUnresolved = unresolved > 0
+ ? await dlq.countUnresolvedMatchingIds(runId, collection, uSample, b.lowerCd, b.upperCd)
+ : 0;
+ const uMissing = uSample.length - uPresent - uUnresolved;
+ if (uMissing > 0 && (progress.idCoverageMissing ?? []).length < 200) {
+ (progress.idCoverageMissing ?? (progress.idCoverageMissing = [])).push({
+ collection, lowerCd: new Date(b.lowerCd).toISOString(), upperCd: new Date(b.upperCd).toISOString(),
+ sampled: uSample.length, missing: uMissing,
+ });
+ }
+ }
// Identity coverage: a doc swapped for ANOTHER doc with the same cd
// keeps the count AND the cd-sum — sampled distinct-id presence is
// the axis that sees WHICH documents exist. Strided (every 25th
diff --git a/src/state/ledger-store.ts b/src/state/ledger-store.ts
index 81a8091..8ea2006 100644
--- a/src/state/ledger-store.ts
+++ b/src/state/ledger-store.ts
@@ -598,6 +598,11 @@ export class LedgerStore {
);
}
+ /** Roll back a bound whose post-store verification failed — apply must never leave a half-applied bound behind. */
+ async clearStoredBound(runId: string): Promise {
+ await this.rc().updateOne({ _id: runId }, { $unset: { cd_upper_bound_ms: '', set_at: '', set_by: '' } });
+ }
+
async setStoredBound(runId: string, boundMs: number, setBy: string): Promise {
await this.rc().updateOne(
{ _id: runId },
diff --git a/src/target/staging-manager.ts b/src/target/staging-manager.ts
index efb5785..709a2d2 100644
--- a/src/target/staging-manager.ts
+++ b/src/target/staging-manager.ts
@@ -564,9 +564,11 @@ export class StagingManager {
* queries; used by ledger rebuild to attribute null-cd sweep rows (their
* cd is ts-derived and lands inside regular chunks' windows).
*/
- async fetchLiveCdByIds(ids: string[], cdBounds?: { loMs: number; hiMs: number }): Promise