From bcf0fd8515e6f4ea2741acd3d9d4fcad53c6459c Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 21:38:27 +0000 Subject: [PATCH 01/26] test: freeze the v1.4.2 counterexample matrix, RED on current main Thirteen cases, every expected status, exit code, candidate-file count and comparable count written down BEFORE any repair exists, and then run against 77e3677 to record what it actually does. Seven of them are defects the current legacy optimizer really has, on these exact fixtures: - a history built entirely from bounded sweeps promotes a trend candidate, and --accept-partial changes nothing because history quality is never read at all - the newest INVALID record picks the comparison anchor and then removes itself, stranding six eligible records (comparable 0) - HOST_BEHAVIOR_SHIFT reports exit 30 while writing the candidate file - six records claiming COMPLETE with sessions=turns=carry_bytes=0 promote - a transcript that lost a record to torn JSON still reports the sweep COMPLETE, and the run reports NO_ACTION - --max-files 1 gives the carry sweep and the skill-listing scan two different single-source samples - a schema-1 record, from a generation that had no evidence_quality field at all, is read as COMPLETE Four cases are positive controls (a clean COMPLETE population must still promote; an invalid record in another scope must not poison this one; schema-4 evidence must stay refused by name; a flat population must stay NO_ACTION) and two pin the strict-exit contract. No source file is touched by this commit: the suite is expected to fail here, and the matrix in docs/V142_COUNTEREXAMPLES.md records both the frozen expectations and what main did. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 69 ++++++++ tests/test_evidence_integrity.py | 270 +++++++++++++++++++++++++++++++ 2 files changed, 339 insertions(+) create mode 100644 docs/V142_COUNTEREXAMPLES.md create mode 100644 tests/test_evidence_integrity.py diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md new file mode 100644 index 0000000..0bee373 --- /dev/null +++ b/docs/V142_COUNTEREXAMPLES.md @@ -0,0 +1,69 @@ +# v1.4.2 — frozen counterexample matrix + +Written **before** the repair, against `main` at `77e3677`. Every expectation below is what the +optimizer *must* do; the second block records what it actually did when the matrix was frozen. An +expectation may only change afterwards if the original expectation is independently proven wrong, +and the change must be recorded in §3. + +Scope: the **legacy** optimizer path (`tools/optimize.py`, history schemas 0/1/2, live sweeps from +`tools/carry.py`). Schema-4 typed evidence stays unsupported by this optimizer and the v1.4 typed +promotion stays shadow-only; neither is touched here. + +## 1. The matrix + +| case | fixture | status | strict exit | candidate files | comparable | evidence quality | +|---|---|---|---|---|---|---| +| **R142_01** | 6 schema-2 records, `evidence_quality=PARTIAL`, `skipped_by_limit=5`, no live sweep, no flag | `PARTIAL_EVIDENCE` | 40 | 0 | 6 | `PARTIAL` | +| **R142_01P** | same fixture **with** `--accept-partial` | `CANDIDATE` | 10 | 1 | 6 | `PARTIAL` | +| **R142_02** | 6 eligible `COMPLETE` records + 1 newer `INVALID` record (`sessions=0`, `scanned>0`), incomparable corpus size | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **R142_03** | 6 `COMPLETE` records whose runtime set changes mid-history, share moving ≥ 5 pp | `HOST_BEHAVIOR_SHIFT` | 30 | 0 | 6 | `COMPLETE` | +| **R142_04** | 6 records claiming `COMPLETE` with `sessions=turns=carry_bytes=0` | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | `INVALID` | +| **R142_05** | live sweep over a transcript holding one torn JSON record, with **and** without `--accept-partial` | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | `DEGRADED` (`carry` reports `PARTIAL`, never `COMPLETE`) | +| **R142_06** | `--max-files 1` where discovery order is the reverse of mtime order | carry sweep and skill-listing scan select the **same single source** (the newest) | — | — | — | — | +| **U1** | 6 schema-1 records (a generation with no `evidence_quality` field), with and without `--accept-partial` | `PARTIAL_EVIDENCE` | 40 | 0 | 6 | `UNKNOWN` | +| **P1** | 6 `COMPLETE` records, acquisition counters all zero | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **P3** | P1's population + one `INVALID` record in **another scope** | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **P4** | one current-generation (schema-4 envelope) record | `INSUFFICIENT_DATA` | 20 | 0 | 0 | unsupported, refused by name | +| **S_NOACTION** | 6 `COMPLETE` records, share flat (no finding over threshold) | `NO_ACTION` | 0 | 0 | 6 | `COMPLETE` | +| **S_LOCK** | P1's fixture with the emit lock already held | `ALREADY_RUNNING` | 41 | 0 | 6 | `COMPLETE` | + +Quality vocabulary for the legacy law (worst wins, never a majority): + +```text +COMPLETE < PARTIAL < DEGRADED < EMPTY < UNKNOWN < INVALID +``` + +* `PARTIAL` = a bound the caller chose (`--max-files`). Only this one is rescued by + `--accept-partial`. +* `DEGRADED` = evidence that was selected and then lost (unreadable, oversize, malformed, + identity changed under the read, conflicting records). A loss is never a chosen bound, so + `--accept-partial` does not accept it. +* `UNKNOWN` = the record cannot attest its own quality: a schema older than the field, or a + schema-2 record without the acquisition counters its writer always wrote. +* `INVALID` / `EMPTY` = the producer's own terms for a sweep that found nothing usable, plus any + record whose own numbers contradict its claim. + +## 2. Observed on `main` 77e3677 when the matrix was frozen + +Run against these exact fixtures, before a line of the repair existed: + +```text +R142_01 CANDIDATE exit 10 files 1 comparable 6 bounded history promotes +R142_01P CANDIDATE exit 10 files 1 comparable 6 the flag changed nothing: it was never read +R142_02 INSUFFICIENT_DATA exit 20 files 0 comparable 0 the INVALID record anchored, six eligible stranded +R142_03 HOST_BEHAVIOR_SHIFT exit 30 files 1 comparable 6 the refusing status still wrote the file +R142_04 CANDIDATE exit 10 files 1 comparable 6 sessions=turns=carry_bytes=0 promoted +R142_05 carry quality=COMPLETE, malformed=1 a lost record is not a loss +R142_05 NO_ACTION exit 0 files 0 and the run reports nothing wrong +R142_06 listing sample read the OLDEST transcript two definitions of one bound +U1 CANDIDATE exit 10 files 1 comparable 6 a field that never existed read as COMPLETE +P1 CANDIDATE exit 10 files 1 comparable 6 (already correct) +P3 INSUFFICIENT_DATA exit 20 files 0 comparable 0 an invalid record in ANOTHER scope chose the scope +P4 INSUFFICIENT_DATA exit 20 files 0 comparable 0 (already correct) +S_NOACTION NO_ACTION exit 0 files 0 comparable 6 (already correct) +S_LOCK ALREADY_RUNNING exit 41 files 0 comparable 6 (already correct) +``` + +## 3. Deviations from the frozen expectations + +None. diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py new file mode 100644 index 0000000..885ebd4 --- /dev/null +++ b/tests/test_evidence_integrity.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 +"""v1.4.2 — one case per row of docs/V142_COUNTEREXAMPLES.md. + +The legacy optimizer promoted findings from evidence it had never checked: a history built from +bounded sweeps, a record claiming a completeness its own numbers contradict, a population anchored +on the one record that was thrown away, a sweep that lost records to torn JSON, and a status that +refused promotion while still writing the candidate file to disk. + +Every case below is a counterexample first and a regression second: it was RED on 77e3677, and the +matrix that says what it must do was frozen before the repair existed. + +Standalone: run this file.""" +import json +import os +import subprocess +import sys +import tempfile + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +OPT = os.path.join(ROOT, "tools", "optimize.py") +sys.path.insert(0, os.path.join(ROOT, "tools")) +import carry # noqa: E402 +import optimize # noqa: E402 + +P = F = 0 +TS0 = 1_750_000_000 + + +def check(label, got, want): + global P, F + if got == want: + P += 1 + print(f" PASS {label}") + else: + F += 1 + print(f" FAIL {label}: got {got!r}, want {want!r}") + + +def rec(i, share, quality="COMPLETE", schema=2, counters=True, sessions=40, turns=1000, + carry_bytes=10 ** 7, scanned=100, scope="default", workload="code", runtime="2.1.270", + unreadable=0, oversize=0, skipped=0): + """One legacy history record, shaped like the one carry.history() wrote in 1.2/1.3. + + `counters=False` is the record a hand-edited file or a back-filled migration produces: it + claims a quality it never acquired the facts for.""" + r = {"schema_version": schema, "record_type": "carry_run", "run_id": f"r{scope}{i}", + "scope_id": scope, "workload_class": workload, "ts": TS0 + i * 604800, + "sessions": sessions, "turns": turns, "carry_bytes": carry_bytes, "scanned": scanned, + "runtimes": {runtime: 40}, "models": {"m1": 40}, + "shares": {"Bash": round(share, 4), "Read": round(100.0 - share, 4)}, + "bpt": {"Bash": 1.0, "Read": 1.0}} + if schema >= 2: + r["evidence_quality"] = quality + if counters: + r.update(unreadable=unreadable, oversize=oversize, skipped_by_limit=skipped) + return r + + +def write(path, rows): + with open(path, "w", encoding="utf-8") as fh: + for r in rows: + fh.write((r if isinstance(r, str) else json.dumps(r)) + "\n") + return path + + +def run(rows, extra=(), scan=(), lock=False): + """Run the optimizer as the scheduler runs it -> (exit code, parsed --json, files on disk).""" + d = tempfile.mkdtemp(prefix="sw-142-") + hist = write(os.path.join(d, "history.jsonl"), rows) + out = os.path.join(d, "cand") + if lock: + os.makedirs(out, exist_ok=True) + open(os.path.join(out, ".optimize.lock"), "w").write("1") + argv = [sys.executable, OPT, "--history", hist, "--ledger", os.path.join(d, "none.jsonl"), + "--emit-candidate", out, "--json", "--strict-exit", "--scan"] + list(scan) + list(extra) + p = subprocess.run(argv, capture_output=True, text=True, timeout=300) + try: + j = json.loads(p.stdout) + except Exception: + raise SystemExit(f"optimizer produced no JSON (rc={p.returncode}):\n{p.stdout}\n{p.stderr}") + files = [f for _b, _d, fs in os.walk(out) for f in fs if f != ".optimize.lock"] + return p.returncode, j, len(files) + + +def case(label, rows, status, code, files, comparable=None, extra=(), scan=(), lock=False, + history_quality=None): + rc, j, n = run(rows, extra=extra, scan=scan, lock=lock) + check(f"{label}: status", j["status"], status) + check(f"{label}: strict exit", rc, code) + check(f"{label}: candidate files on disk", n, files) + if comparable is not None: + check(f"{label}: comparable", j["history"]["comparable"], comparable) + if history_quality is not None: + check(f"{label}: history quality", j["history"].get("quality"), history_quality) + return j + + +def transcript(path, turns=80, listing=False, torn=False): + """A minimal Claude-Code-shaped transcript: enough turns to be a session, optionally one + skill listing attachment, optionally one torn JSON record.""" + rows = [] + if listing: + rows.append(json.dumps({"type": "user", "attachment": { + "type": "skill_listing", + "content": "- alpha: does a thing that is described at some length\n" + "- beta: does another thing, also described at some length\n"}})) + for t in range(turns): + rows.append(json.dumps({"type": "assistant", "message": { + "id": f"msg_{t}", "usage": {"input_tokens": 10, "output_tokens": 7}, + "content": [{"type": "tool_use", "name": "Bash", "input": {"command": "ls -la"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 400}]}})) + if torn: + rows[len(rows) // 2] = '{"type": "assistant", "message": {"usage": {"out' + with open(path, "w", encoding="utf-8") as fh: + fh.write("\n".join(rows) + "\n") + return path + + +def main(): + d = tempfile.mkdtemp(prefix="sw-142-fx-") + + # ---------------------------------------------------------------- the law itself + print("\nlegacy evidence-quality law") + check("worst wins, not the majority", + optimize.worst_quality(["COMPLETE", "COMPLETE", "PARTIAL", "COMPLETE"]), "PARTIAL") + check("no evidence is not bad evidence", optimize.worst_quality([]), "COMPLETE") + check("a schema older than the field cannot claim the field", + optimize.record_quality(rec(0, 40.0, schema=1, counters=False)), "UNKNOWN") + check("schema 2 without the counters its writer always wrote", + optimize.record_quality(rec(0, 40.0, counters=False)), "UNKNOWN") + check("counters present and zero: COMPLETE is attested", + optimize.record_quality(rec(0, 40.0)), "COMPLETE") + check("a chosen bound is PARTIAL", + optimize.record_quality(rec(0, 40.0, skipped=5)), "PARTIAL") + check("a loss is DEGRADED, not a chosen bound", + optimize.record_quality(rec(0, 40.0, unreadable=3)), "DEGRADED") + check("the claim cannot be better than the counters", + optimize.record_quality(rec(0, 40.0, quality="COMPLETE", oversize=1)), "DEGRADED") + check("the counters cannot be better than the claim", + optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), "PARTIAL") + check("looked and found nothing usable: INVALID", + optimize.record_quality(rec(0, 40.0, sessions=0, scanned=100)), "INVALID") + check("had nothing to look at: EMPTY", + optimize.record_quality(rec(0, 40.0, sessions=0, scanned=0)), "EMPTY") + check("sessions without turns is not a sweep that happened", + optimize.record_quality(rec(0, 40.0, turns=0)), "INVALID") + check("sessions without carry is not a sweep that happened", + optimize.record_quality(rec(0, 40.0, carry_bytes=0)), "INVALID") + check("a count that is not a count", + optimize.record_quality(rec(0, 40.0, unreadable=-1)), "INVALID") + check("a boolean is not a count", + optimize.record_quality(rec(0, 40.0, oversize=True)), "INVALID") + check("--accept-partial accepts the bound it is named for", + optimize.promotable("PARTIAL", True), True) + check("--accept-partial does not accept a loss", + optimize.promotable("DEGRADED", True), False) + check("--accept-partial does not accept what cannot be verified", + optimize.promotable("UNKNOWN", True), False) + check("--accept-partial does not accept an invalid record", + optimize.promotable("INVALID", True), False) + check("COMPLETE needs no flag", optimize.promotable("COMPLETE", False), True) + check("PARTIAL without the flag stays refused", optimize.promotable("PARTIAL", False), False) + + # ---------------------------------------------------------------- R142_01 / R142_01P + print("\nR142_01 - a history built from bounded sweeps is not a population") + partial = [rec(i, 30.0 + 3.0 * i, quality="PARTIAL", skipped=5) for i in range(6)] + case("R142_01", partial, "PARTIAL_EVIDENCE", 40, 0, comparable=6, history_quality="PARTIAL") + case("R142_01P (--accept-partial)", partial, "CANDIDATE", 10, 1, comparable=6, + extra=["--accept-partial"], history_quality="PARTIAL") + + # ---------------------------------------------------------------- R142_02 + print("\nR142_02 - the anchor comes from evidence that survives its own filter") + stranding = [rec(i, 30.0 + 3.0 * i) for i in range(6)] + stranding.append(rec(9, 55.0, quality="INVALID", sessions=0, turns=100000, carry_bytes=0)) + case("R142_02", stranding, "CANDIDATE", 10, 1, comparable=6, history_quality="COMPLETE") + + # ---------------------------------------------------------------- R142_03 + print("\nR142_03 - a status that refuses promotion writes nothing") + shifted = ([rec(i, 30.0 + 4.0 * i, runtime="2.1.270") for i in range(3)] + + [rec(i, 30.0 + 4.0 * i, runtime="2.1.290") for i in range(3, 6)]) + case("R142_03", shifted, "HOST_BEHAVIOR_SHIFT", 30, 0, comparable=6) + + # ---------------------------------------------------------------- R142_04 + print("\nR142_04 - a record cannot claim a completeness its own numbers contradict") + impossible = [rec(i, 30.0 + 3.0 * i, sessions=0, turns=0, carry_bytes=0) for i in range(6)] + case("R142_04", impossible, "PARTIAL_EVIDENCE", 40, 0, comparable=0) + + # ---------------------------------------------------------------- R142_05 + print("\nR142_05 - a sweep that lost records to torn JSON is not COMPLETE") + clean = transcript(os.path.join(d, "clean.jsonl")) + torn = transcript(os.path.join(d, "torn.jsonl"), torn=True) + a_clean, a_torn = carry.accumulate([clean], min_turns=1), carry.accumulate([torn], min_turns=1) + check("control: a clean sweep is still COMPLETE", a_clean["quality"], "COMPLETE") + check("the torn record is counted", a_torn["malformed"], 1) + check("and the sweep is no longer COMPLETE", a_torn["quality"] != "COMPLETE", True) + check("the optimizer reads it as a loss, not a bound", + optimize.sweep_quality(a_torn), "DEGRADED") + check("control: the clean sweep reads COMPLETE", optimize.sweep_quality(a_clean), "COMPLETE") + check("a loss is not rescued by --accept-partial", + optimize.promotable(optimize.sweep_quality(a_torn), True), False) + case("R142_05 (live, torn)", [], "PARTIAL_EVIDENCE", 40, 0, scan=[torn, "--min-turns", "1"]) + case("R142_05 (live, torn, --accept-partial)", [], "PARTIAL_EVIDENCE", 40, 0, + scan=[torn, "--min-turns", "1"], extra=["--accept-partial"]) + + # ---------------------------------------------------------------- R142_06 + print("\nR142_06 - one definition of a bounded sample") + paths = [] + for n, name in enumerate(("a.jsonl", "b.jsonl", "c.jsonl")): + p = transcript(os.path.join(d, name), turns=60, listing=(name == "a.jsonl")) + os.utime(p, (TS0 + n * 1000, TS0 + n * 1000)) # a oldest, c newest + paths.append(p) + check("the bound is the newest N", + [os.path.basename(p) for p in carry.bounded_paths(paths, 1)], ["c.jsonl"]) + check("and it does not depend on discovery order", + carry.bounded_paths(list(reversed(paths)), 1), carry.bounded_paths(paths, 1)) + check("an unbounded sweep keeps every source", carry.bounded_paths(paths, 0), paths) + _rc, j, _n = run([], scan=paths + ["--max-files", "1", "--min-turns", "1"]) + check("the listing scan reads the sweep's sample, not a discovery-order slice", + j["listing"], None) + _rc, j2, _n = run([], scan=paths + ["--min-turns", "1"]) + check("control: unbounded, the listing in the oldest transcript IS read", + bool(j2["listing"]), True) + + # ---------------------------------------------------------------- U1 + print("\nU1 - a generation that never had the field cannot have defaulted to COMPLETE") + old = [rec(i, 30.0 + 3.0 * i, schema=1, counters=False) for i in range(6)] + case("U1", old, "PARTIAL_EVIDENCE", 40, 0, comparable=6, history_quality="UNKNOWN") + case("U1 (--accept-partial does not rescue it)", old, "PARTIAL_EVIDENCE", 40, 0, + extra=["--accept-partial"]) + + # ---------------------------------------------------------------- positive controls + print("\npositive controls - a gate that refuses everything is not a gate") + good = [rec(i, 30.0 + 3.0 * i) for i in range(6)] + case("P1 clean COMPLETE population", good, "CANDIDATE", 10, 1, comparable=6, + history_quality="COMPLETE") + case("P3 an invalid record in ANOTHER scope does not poison this one", + good + [rec(9, 55.0, quality="INVALID", sessions=0, scope="other")], + "CANDIDATE", 10, 1, comparable=6) + envelope = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}) + rc, j, n = run([envelope]) + check("P4 current-generation evidence stays unsupported", j["status"], "INSUFFICIENT_DATA") + check("P4 strict exit", rc, 20) + check("P4 refused by name, not consumed", + any("current-generation" in k for k in j["history"]["rejected"]), True) + check("P4 nothing written", n, 0) + check("P4 the supported schema set is unchanged", j["history_schema_supported"], [0, 1, 2]) + flat = [rec(i, 50.0) for i in range(6)] + case("S_NOACTION nothing moved", flat, "NO_ACTION", 0, 0, comparable=6) + case("S_LOCK another run holds the lock", good, "ALREADY_RUNNING", 41, 0, lock=True) + + # ---------------------------------------------------------------- the emission gate itself + print("\nthe gate is in the emitter, not only in its caller") + f = optimize.finding("x", "CANDIDATE", "head", "ev", scope="s", bucket=1) + for status in ("PARTIAL_EVIDENCE", "HOST_BEHAVIOR_SHIFT", "INSUFFICIENT_DATA", + "INTERNAL_ERROR", "ALREADY_RUNNING", "NO_ACTION"): + out = os.path.join(d, "gate_" + status) + w, e, fail = optimize.emit_candidates([f], out, status) + check(f"{status} writes nothing", (w, e, fail, os.path.exists(out)), ([], [], [], False)) + out = os.path.join(d, "gate_CANDIDATE") + w, _e, _f = optimize.emit_candidates([f], out, "CANDIDATE") + check("CANDIDATE still writes", (len(w), os.path.exists(out)), (1, True)) + + print(f"\n{P} PASS, {F} FAIL") + return 1 if F else 0 + + +if __name__ == "__main__": + sys.exit(main()) From 1c9720811ba66d32ee30e0da04b887d6dcc4e339 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 21:50:03 +0000 Subject: [PATCH 02/26] fix: one legacy evidence-quality law, an eligibility-first anchor, one emission gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The legacy optimizer decided five different things about evidence in five different places, and one of them decided nothing at all. This makes the question one function, asks it of the evidence each finding actually rests on, and puts the candidate-writing gate where every caller has to pass through it. What changed, in the order the counterexamples found it: - `record_quality()` / `sweep_quality()` are the law. The worst of what a record claims and what its own numbers can show, over the counters the producer has written since schema 2. A schema-1 record has no evidence_quality field at all, so a schema-1 record carrying COMPLETE is a claim its writer could not have made: UNKNOWN. A schema-2 record without the acquisition counters cannot attest completeness either. - PARTIAL and DEGRADED are now different things. A bound the caller asked for (`--max-files`) is PARTIAL and is exactly what `--accept-partial` adopts; evidence that was selected and then lost is DEGRADED and no flag accepts it. carry.LOSS_FIELDS / BOUND_FIELDS is the single vocabulary both sides read. - A sweep that lost records to torn JSON is no longer COMPLETE. `malformed` joins the loss counters, so carry.sweep_label() degrades the sweep and the optimizer reads it as DEGRADED. - `eligible_anchor()` picks the comparison anchor from records that survive their own filter. The newest INVALID record used to choose the scope, the workload class and the corpus size for everyone else and then remove itself, stranding a whole eligible population. - The history trend is gated. It was `if True:` — a trend over bounded sweeps was indistinguishable from one swept in full. Now it is a CANDIDATE on eligible history and an OBSERVED finding, with the reason in its evidence line, on anything else. - Each finding is gated on ITS OWN evidence: live findings on the sweep, the trend on the eligible history, the guard on its own ledger sample. A partial history no longer blocks a live finding it never supported. - `emit_candidates()` takes the status and writes nothing unless it is CANDIDATE. HOST_BEHAVIOR_SHIFT reported exit 30 while leaving a specification on disk; the gate now lives in the emitter, so the CLI, the long-run simulator and any future caller route through it. - `carry.bounded_paths()` is the one definition of "the newest N", used by both the carry sweep and the skill-listing scan. One `--max-files 1` run was analysing two different single-source populations. Schema-4 evidence stays unsupported by this optimizer and the typed promotion stays shadow-only. Proved rather than asserted: the encoded v1.4 record for a clean, a torn and a bounded sweep has a byte-identical sha256 before and after this commit, and the typed reader's loss_observed verdict is unchanged. Only the legacy `quality` label moved, for the torn sweep, which is the defect being repaired. Four fixtures claimed COMPLETE without the acquisition counters a real sweep always writes (multi-agent, mutation, long-run, readiness). Each now states them, and each edit is locked by an added assertion that the same record WITHOUT them refuses to promote — so the fixture change cannot hide the gate it was making room for. skills/ and hooks/ are untouched: CANONICAL_BODY_SHA256 is unchanged. Co-Authored-By: Claude Opus 5 --- experiments/aivos/bash_residue.py | 3 +- experiments/aivos/longrun.py | 8 +- experiments/aivos/readiness.py | 25 ++- tests/test_evidence_integrity.py | 24 ++- tests/test_multiagent.py | 26 ++- tools/carry.py | 94 ++++++---- tools/optimize.py | 280 ++++++++++++++++++++++++++---- 7 files changed, 379 insertions(+), 81 deletions(-) diff --git a/experiments/aivos/bash_residue.py b/experiments/aivos/bash_residue.py index 21ecc39..45badc6 100644 --- a/experiments/aivos/bash_residue.py +++ b/experiments/aivos/bash_residue.py @@ -44,8 +44,7 @@ def main(): a = ap.parse_args() paths, _ = profiles.resolve(a.scan or []) - if a.max_files and len(paths) > a.max_files: - paths = sorted(paths, key=os.path.getmtime, reverse=True)[:a.max_files] + paths = carry.bounded_paths(paths, a.max_files) # one definition of "the newest N" sizes = collections.defaultdict(list) total = collections.Counter() diff --git a/experiments/aivos/longrun.py b/experiments/aivos/longrun.py index 2d0313d..c790b19 100644 --- a/experiments/aivos/longrun.py +++ b/experiments/aivos/longrun.py @@ -30,8 +30,12 @@ def rec(ts, shares, scope, turns=1000, workload="", runtimes=None): + # The acquisition counters a real 1.2/1.3 sweep always wrote. Since 1.4.2 a record claiming + # COMPLETE without them cannot carry a promotion, so a simulation that omitted them would be + # simulating a population no producer ever writes. return {"schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(), - "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7, + "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7, "scanned": 100, + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "scope_id": scope, "workload_class": workload, "evidence_quality": "COMPLETE", "runtimes": runtimes or {"2.1.270": 40}, "models": {"m1": 40}, "shares": shares, "bpt": {k: 1.0 for k in shares}} @@ -164,7 +168,7 @@ def longrun(days, cycles, out, plateau=20): f = optimize.analyse(lv, h, None, None, scope=role) st = optimize.overall_status(f, h, lv, optimize.population(keep), False) statuses[st] += 1 - w, e, _fail = optimize.emit_candidates(f, cand) + w, e, _fail = optimize.emit_candidates(f, cand, st) written += len(w) existing += len(e) day_new += len(w) diff --git a/experiments/aivos/readiness.py b/experiments/aivos/readiness.py index fbbf244..5db515a 100644 --- a/experiments/aivos/readiness.py +++ b/experiments/aivos/readiness.py @@ -130,7 +130,11 @@ def main(): "schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(), "ts": 1_750_000_000 + i * 604800, "sessions": 60, "turns": 3000, "carry_bytes": 10 ** 8, "scope_id": "governed", "workload_class": "audit", - "evidence_quality": "COMPLETE", "runtimes": {"2.1.271": 60}, "models": {"m": 60}, + "evidence_quality": "COMPLETE", "scanned": 120, + # the acquisition counters the producer writes; since 1.4.2 a COMPLETE claim + # without them cannot promote (tests/test_evidence_integrity.py) + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, + "runtimes": {"2.1.271": 60}, "models": {"m": 60}, "shares": {"Bash": 40.0 + i * 9, "Read": 60.0 - i * 9}, "bpt": {"Bash": 1.0, "Read": 1.0}}) + "\n") @@ -153,6 +157,25 @@ def main(): check("JSON leaks no secret", CANARY in rj.stdout, False) check("JSON leaks no path", ("/home/" in rj.stdout) or (ws in rj.stdout), False) + # The same population with the counters stripped claims a completeness it cannot show: it must + # refuse, and it must write nothing. Without this, the fixture edit above could hide the gate. + bare_hist = os.path.join(state, "bare_history.jsonl") + bare_cand = os.path.join(state, "bare_candidates") + with open(hist, encoding="utf-8") as fh, open(bare_hist, "w", encoding="utf-8") as out_fh: + for line in fh: + o = json.loads(line) + for k in ("unreadable", "oversize", "skipped_by_limit"): + o.pop(k, None) + out_fh.write(json.dumps(o) + "\n") + rb = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "optimize.py"), + "--history", bare_hist, "--ledger", os.path.join(state, "none.jsonl"), + "--scan", "--scope-id", "governed", "--json", "--strict-exit", + "--emit-candidate", bare_cand], + capture_output=True, text=True, timeout=900) + check("a record that cannot attest its own sweep does not promote", rb.returncode, 40) + check("and nothing is written for it", + [f for r_, _d, fs in os.walk(bare_cand) for f in fs], []) + specs = [os.path.join(r_, f) for r_, _d, fs in os.walk(cand) for f in fs if f.endswith(".md")] check("a specification was written, outside the governed repository", bool(specs), True) check("specifications live outside the workspace", diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 885ebd4..75b1726 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -152,15 +152,15 @@ def main(): check("a boolean is not a count", optimize.record_quality(rec(0, 40.0, oversize=True)), "INVALID") check("--accept-partial accepts the bound it is named for", - optimize.promotable("PARTIAL", True), True) + optimize.may_promote("PARTIAL", True), True) check("--accept-partial does not accept a loss", - optimize.promotable("DEGRADED", True), False) + optimize.may_promote("DEGRADED", True), False) check("--accept-partial does not accept what cannot be verified", - optimize.promotable("UNKNOWN", True), False) + optimize.may_promote("UNKNOWN", True), False) check("--accept-partial does not accept an invalid record", - optimize.promotable("INVALID", True), False) - check("COMPLETE needs no flag", optimize.promotable("COMPLETE", False), True) - check("PARTIAL without the flag stays refused", optimize.promotable("PARTIAL", False), False) + optimize.may_promote("INVALID", True), False) + check("COMPLETE needs no flag", optimize.may_promote("COMPLETE", False), True) + check("PARTIAL without the flag stays refused", optimize.may_promote("PARTIAL", False), False) # ---------------------------------------------------------------- R142_01 / R142_01P print("\nR142_01 - a history built from bounded sweeps is not a population") @@ -198,7 +198,7 @@ def main(): optimize.sweep_quality(a_torn), "DEGRADED") check("control: the clean sweep reads COMPLETE", optimize.sweep_quality(a_clean), "COMPLETE") check("a loss is not rescued by --accept-partial", - optimize.promotable(optimize.sweep_quality(a_torn), True), False) + optimize.may_promote(optimize.sweep_quality(a_torn), True), False) case("R142_05 (live, torn)", [], "PARTIAL_EVIDENCE", 40, 0, scan=[torn, "--min-turns", "1"]) case("R142_05 (live, torn, --accept-partial)", [], "PARTIAL_EVIDENCE", 40, 0, scan=[torn, "--min-turns", "1"], extra=["--accept-partial"]) @@ -222,6 +222,16 @@ def main(): check("control: unbounded, the listing in the oldest transcript IS read", bool(j2["listing"]), True) + # ---------------------------------------------------------------- the bound itself + print("\nthe bound a caller asked for is not a loss (and is still not COMPLETE)") + bounded = carry.accumulate(paths, min_turns=1, max_files=1) + check("a bounded sweep records what it skipped", bounded["skipped_by_limit"], 2) + check("the producer labels it PARTIAL", bounded["quality"], "PARTIAL") + check("the optimizer reads a bound, not a loss", optimize.sweep_quality(bounded), "PARTIAL") + check("refused without the flag", + optimize.may_promote(optimize.sweep_quality(bounded), False), False) + check("adopted with it", optimize.may_promote(optimize.sweep_quality(bounded), True), True) + # ---------------------------------------------------------------- U1 print("\nU1 - a generation that never had the field cannot have defaulted to COMPLETE") old = [rec(i, 30.0 + 3.0 * i, schema=1, counters=False) for i in range(6)] diff --git a/tests/test_multiagent.py b/tests/test_multiagent.py index 851b49e..4507f7d 100644 --- a/tests/test_multiagent.py +++ b/tests/test_multiagent.py @@ -32,9 +32,14 @@ def check(label, got, want): def rec(ts, shares, scope="default", turns=1000, sessions=40, **kw): + # The acquisition counters are part of the record, not decoration: carry.history() has written + # them since schema 2, and since 1.4.2 a record that claims COMPLETE without them cannot carry + # a promotion (see tests/test_evidence_integrity.py). A fixture that omitted them was asserting + # a completeness no real sweep asserts that way. r = {"schema_version": 2, "record_type": "carry_run", "ts": ts, "sessions": sessions, "turns": turns, "carry_bytes": 10 ** 7, "scanned": 100, "scope_id": scope, "workload_class": "", "evidence_quality": "COMPLETE", + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "run_id": kw.pop("run_id", None) or carry.new_run_id(), "shares": shares, "bpt": {k: 1.0 for k in shares}} r.update(kw) @@ -295,12 +300,27 @@ def main(): far["candidate_id"] != f1["candidate_id"], True) check("candidate_id membawa scope-nya", f1["candidate_id"].split("-")[-2], "x") + # Lock the rule the fixture above now satisfies: the SAME record without its counters claims a + # completeness it cannot show, and a population of those cannot promote anything. + bare = [{k: v for k, v in r.items() if k not in ("unreadable", "oversize", "skipped_by_limit")} + for r in [rec(100 + i * 86400 * 7, {"Bash": 30.0 + i * 8, "Read": 70.0 - i * 8}, + scope="bare") for i in range(6)]] + keep_bare, _dropped_bare = optimize.comparable(bare) + check("record tanpa counters: tetap dibandingkan, tapi kualitasnya UNKNOWN", + (len(keep_bare), optimize.history_quality(keep_bare)), (6, "UNKNOWN")) + h_bare = {"comparable": keep_bare, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"} + check("dan UNKNOWN tak melahirkan kandidat, dgn atau tanpa --accept-partial", + {x["state"] for x in optimize.analyse(None, h_bare, None, None, scope="bare")} + | {x["state"] for x in optimize.analyse(None, h_bare, None, None, scope="bare", + accept_partial=True)}, {"OBSERVED"}) + out2 = os.path.join(d, "cand_dedup") - w, e, _fail = optimize.emit_candidates([f1], out2) + w, e, _fail = optimize.emit_candidates([f1], out2, "CANDIDATE") check("emisi pertama menulis", (len(w), len(e)), (1, 0)) - w, e, _fail = optimize.emit_candidates([f2], out2) + w, e, _fail = optimize.emit_candidates([f2], out2, "CANDIDATE") check("emisi kedua dgn bukti setara: EXISTING, nol penulisan ulang", (len(w), len(e)), (0, 1)) - w, e, _fail = optimize.emit_candidates([far], out2) + w, e, _fail = optimize.emit_candidates([far], out2, "CANDIDATE") check("bukti yang benar-benar bergerak: kandidat baru ditulis", (len(w), len(e)), (1, 0)) # ------------------------------------------------------------ 8b. identitas tren stabil diff --git a/tools/carry.py b/tools/carry.py index a67ca08..1a6a2f8 100755 --- a/tools/carry.py +++ b/tools/carry.py @@ -35,6 +35,14 @@ # ASCII, and the constant was calibrated on the same len(). +# Evidence a sweep SELECTED and then could not turn into a measurement. A chosen bound is NOT one +# of them, which is the whole distinction `--accept-partial` rests on: a caller may adopt the bound +# it asked for, and may never adopt a file it could not read or a record it could not parse. +# One vocabulary, read by the producer below and by the optimizer's legacy law, so the two cannot +# drift into disagreeing about what "partial" meant. +LOSS_FIELDS = ("unreadable", "oversize", "malformed", "identity_changed", "conflicted_sources") +BOUND_FIELDS = ("skipped_by_limit",) + HISTORY_SCHEMA = 2 # the generation 1.3 wrote. Kept as the name older callers import. # 0 = pre-1.2 records with no schema field · 1 = + population identity # 2 = + run_id / scope_id / evidence quality (multi-agent safety) @@ -231,6 +239,42 @@ def _identity(path): "mtime_ns": getattr(st, "st_mtime_ns", int(st.st_mtime * 1e9))} +def bounded_paths(paths, max_files): + """The ONE definition of a bounded sample: the newest `max_files` sources by mtime. + + A bound has to mean "the most RECENT N", not "the first N the filesystem listed". An + alphabetical prefix of a long-lived archive is a sample of whatever was created first, which + for a 24x7 population is the least informative slice there is. + + It lives here, alone, because a bounded run had two of these: the carry sweep took the newest N + and the skill-listing scan took a discovery-order slice of the same list, so one `--max-files 1` + run analysed two different single-source populations and reported them as one. + """ + paths = list(paths) + if not max_files or len(paths) <= max_files: + return paths + try: + return sorted(paths, key=lambda q: os.path.getmtime(q), reverse=True)[:max_files] + except OSError: + # A path that lost its mtime cannot order the sample; the bound still has to hold. + return paths[:max_files] + + +def sweep_label(facts): + """The 1.3-generation word for how a sweep went, from the counters the sweep itself kept. + + Four values, and PARTIAL has to carry two different meanings: a bound the caller ASKED for, and + evidence that was selected and then LOST. The label keeps the vocabulary a 1.3 reader knows; + a consumer that must tell the two apart reads LOSS_FIELDS and BOUND_FIELDS, which is exactly + what the optimizer's legacy law does — `--accept-partial` may adopt a bound, never a loss. + """ + if not facts.get("sessions"): + return "INVALID" if facts.get("scanned") else "EMPTY" + if any(facts.get(k) for k in LOSS_FIELDS + BOUND_FIELDS): + return "PARTIAL" + return "COMPLETE" + + def accumulate(paths, min_turns=50, max_files=0): carry, size, usage = collections.Counter(), collections.Counter(), collections.Counter() runtimes, models = collections.Counter(), collections.Counter() @@ -252,14 +296,7 @@ def accumulate(paths, min_turns=50, max_files=0): usage_conflicted_transcripts = identity_conflicted_transcripts = 0 usage_exact_measurement_excluded = identity_exact_measurement_excluded = 0 conflicted_sources = records_rejected = 0 - # A bound has to mean "the most RECENT N", not "the first N the filesystem listed". An - # alphabetical prefix of a long-lived archive is a sample of whatever was created first, which - # for a 24x7 population is the least informative slice there is. - if max_files and len(paths) > max_files: - try: - paths = sorted(paths, key=lambda q: os.path.getmtime(q), reverse=True)[:max_files] - except OSError: - paths = list(paths)[:max_files] + paths = bounded_paths(paths, max_files) for p in paths: scanned += 1 try: @@ -372,29 +409,24 @@ def accumulate(paths, min_turns=50, max_files=0): "not_attempted": max(0, selected - accounted), "malformed": malformed + oversize, "records_rejected": records_rejected, "dirs_unreadable": vanished} - # Provenance, not decoration: an analyser that cannot tell a complete sweep from a sweep that - # hit unreadable files or a file cap will happily call a bounded corpus the population. - quality = "COMPLETE" - if unreadable or oversize or skipped_by_limit or conflicted_sources: - quality = "PARTIAL" - if sessions == 0: - quality = "INVALID" if scanned else "EMPTY" - return dict(sessions=sessions, turns=turns, lengths=sorted(lengths), - carry=carry, size=size, usage=usage, runtimes=runtimes, models=models, - unreadable=unreadable, short=short, scanned=scanned, oversize=oversize, - skipped_by_limit=skipped_by_limit, quality=quality, - # v1.4 acquisition facts. `quality` above stays for the 1.3 reader; it is NOT - # what the new record carries, and no record asserts it. - sources=sources, parsed=parsed_files, counters=counters, - sample_bound=max_files or 0, malformed=malformed, - identity_changed=identity_changed, empty_source=empty_source, - usage_conflicts=usage_conflicts, out_of_order=out_of_order, - identity_conflicts=identity_conflicts, - usage_conflicted_transcripts=usage_conflicted_transcripts, - identity_conflicted_transcripts=identity_conflicted_transcripts, - conflicted_sources=conflicted_sources, - usage_exact_measurement_excluded=usage_exact_measurement_excluded, - identity_exact_measurement_excluded=identity_exact_measurement_excluded) + facts = dict(sessions=sessions, turns=turns, lengths=sorted(lengths), + carry=carry, size=size, usage=usage, runtimes=runtimes, models=models, + unreadable=unreadable, short=short, scanned=scanned, oversize=oversize, + skipped_by_limit=skipped_by_limit, + # v1.4 acquisition facts. `quality` below stays for the 1.3 reader; it is NOT + # what the new record carries, and no record asserts it. + sources=sources, parsed=parsed_files, counters=counters, + sample_bound=max_files or 0, malformed=malformed, + identity_changed=identity_changed, empty_source=empty_source, + usage_conflicts=usage_conflicts, out_of_order=out_of_order, + identity_conflicts=identity_conflicts, + usage_conflicted_transcripts=usage_conflicted_transcripts, + identity_conflicted_transcripts=identity_conflicted_transcripts, + conflicted_sources=conflicted_sources, + usage_exact_measurement_excluded=usage_exact_measurement_excluded, + identity_exact_measurement_excluded=identity_exact_measurement_excluded) + facts["quality"] = sweep_label(facts) + return facts # Relative price of one token in each bucket, base input = 1.0. A bucket's share of the diff --git a/tools/optimize.py b/tools/optimize.py index b27fe0e..a46e87c 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -36,7 +36,9 @@ import skills as skills_tool # noqa: E402 (listing usage — reused) OPTIMIZER_VERSION = "1.0" -OUTPUT_SCHEMA_VERSION = 1 # shape of --json; bump when a field's meaning changes +OUTPUT_SCHEMA_VERSION = 2 # shape of --json; bump when a field's meaning changes + # 2 = `evidence_quality` is DERIVED (may read DEGRADED/UNKNOWN), + # and `history.quality` reports the eligible history's worst THRESHOLD_SCHEMA_VERSION = 1 # bump when any threshold below changes, with a reason and a test # ---------------------------------------------------------------- frozen thresholds @@ -62,6 +64,130 @@ SCHEMA_SUPPORTED = (0, 1, 2) # 0 = pre-1.2, 1 = + population identity, 2 = + run/scope/quality STATES = ("OBSERVED", "HYPOTHESIS", "CANDIDATE", "EXPERIMENTAL", "PROVEN", "REJECTED") +# ---------------------------------------------------------------- the legacy evidence-quality law +# ONE place decides what a piece of legacy evidence is worth, because the alternative was five: +# load_history checked a vocabulary, comparable() checked two values, analyse() read the live +# sweep's word for it, the trend read nothing at all, and emit_candidates() never asked. A finding +# could then be promoted from a corpus nobody had swept completely. +# +# Worst wins, never a majority: one bounded observation among nine complete ones still means part +# of the evidence was never swept, and nine neighbours cannot launder it. +QUALITY_RANK = {"COMPLETE": 0, "PARTIAL": 1, "DEGRADED": 2, "EMPTY": 3, "UNKNOWN": 4, "INVALID": 5} +SCHEMA_WITH_QUALITY = 2 # the first history schema that records how a sweep was taken +# The counters every schema-2 record's writer wrote, split by what they mean. Nothing here is +# invented for an older schema: a record that never carried these cannot be asked to show them, +# and is UNKNOWN rather than trusted. +RECORD_LOSS_COUNTERS = ("unreadable", "oversize") +RECORD_BOUND_COUNTERS = ("skipped_by_limit",) +RECORD_COUNTERS = RECORD_LOSS_COUNTERS + RECORD_BOUND_COUNTERS + + +def worst_quality(qualities): + """The worst quality in the set. An empty set is COMPLETE: a finding that draws on no sampled + evidence is not degraded by sampling that had nothing to do with it.""" + worst = "COMPLETE" + for q in qualities: + if QUALITY_RANK.get(q, QUALITY_RANK["UNKNOWN"]) > QUALITY_RANK[worst]: + worst = q if q in QUALITY_RANK else "UNKNOWN" + return worst + + +def may_promote(quality, accept_partial): + """May a finding resting on evidence of this quality become a CANDIDATE? + + `--accept-partial` means one thing: the caller declares the bound they asked for to be the + intended corpus. It is not a switch for evidence that was lost (DEGRADED), for a record that + cannot attest itself (UNKNOWN), or for one the producer's own rules call INVALID/EMPTY. + """ + return quality == "COMPLETE" or (quality == "PARTIAL" and bool(accept_partial)) + + +def _derived_record_quality(rec, schema): + """What a record's OWN numbers prove, ignoring what it claims.""" + nums = {} + for k in RECORD_COUNTERS + ("sessions", "turns", "carry_bytes", "scanned"): + if k not in rec: + continue + v = rec[k] + if isinstance(v, bool) or not isinstance(v, int) or v < 0: + return "INVALID" # a count that is not a count: the record is not trustworthy + nums[k] = v + if nums.get("sessions", 1) == 0: + # the producer's own terms: a sweep that looked and found nothing usable is INVALID; one + # that had nothing to look at is EMPTY + return "INVALID" if nums.get("scanned", 0) > 0 else "EMPTY" + if "sessions" in nums and (nums.get("turns", 1) == 0 or nums.get("carry_bytes", 1) == 0): + # Sessions were counted, so turns were counted and carry was measured. A record reporting + # sessions with neither is not a quiet sweep, it is an impossible one. + return "INVALID" + if any(nums.get(k, 0) for k in RECORD_LOSS_COUNTERS): + return "DEGRADED" + if any(nums.get(k, 0) for k in RECORD_BOUND_COUNTERS): + return "PARTIAL" + if schema >= SCHEMA_WITH_QUALITY and not all(k in nums for k in RECORD_COUNTERS): + return "UNKNOWN" # claims a completeness it cannot show + return "COMPLETE" + + +def record_quality(rec): + """Evidence quality of ONE legacy history record: the worst of what it claims and what it can + show. + + A record may only ever describe itself as no better than its own numbers. Two ways it fails to + attest at all: the value is missing or unrecognised, or the record declares a schema older than + the field itself — schema 0/1 predates `evidence_quality`, so a schema-1 record carrying + COMPLETE is asserting something its own writer could not have known. That is a claim from a + hand-edited file or a back-filled migration, not provenance. The record stays readable and + stays in the report; it simply cannot carry a promotion. + """ + if not isinstance(rec, dict): + return "INVALID" + try: + schema = int(rec.get("schema_version") or 0) + except (TypeError, ValueError): + schema = 0 + claimed = rec.get("evidence_quality") + if not isinstance(claimed, str) or claimed not in QUALITY_RANK: + claimed = "UNKNOWN" + if schema < SCHEMA_WITH_QUALITY: + claimed = "UNKNOWN" + return worst_quality([claimed, _derived_record_quality(rec, schema)]) + + +def sweep_quality(live): + """Evidence quality of the LIVE sweep, from the counters carry.accumulate() kept. + + The producer reports one word for two different things (see carry.sweep_label): a bound the + caller asked for and evidence that was lost both read PARTIAL. Here they separate, because only + the first is something `--accept-partial` may adopt. The counters are the evidence and the + label is a summary of them, so the worst of the two governs. + """ + if not live: + return "COMPLETE" # no live sweep is not bad live evidence; it is none + claimed = live.get("quality") + if not isinstance(claimed, str) or claimed not in QUALITY_RANK: + claimed = "COMPLETE" + if not live.get("sessions"): + return "INVALID" if live.get("scanned") else "EMPTY" + derived = "COMPLETE" + if any(live.get(k) for k in carry.LOSS_FIELDS): + derived = "DEGRADED" + elif any(live.get(k) for k in carry.BOUND_FIELDS): + derived = "PARTIAL" + return worst_quality([claimed, derived]) + + +def history_quality(records): + """The worst quality among the records ELIGIBLE to support a finding. + + Eligibility is not decided here: comparable() already decided it — same scope, same workload + class, compatible corpus size, quality that can carry a comparison at all. A partial record + belonging to another agent's scope is not evidence for this run and must not block it. That + half matters as much as the fail-closed half: a gate that blocks on evidence a finding never + used is not correct, it is merely stuck. + """ + return worst_quality([record_quality(r) for r in (records or [])]) + # machine-readable outcome. The CLI exits 0 for every VALID run by default (a scheduler must not # treat "nothing to do" as breakage); --strict-exit maps the status to the exit code instead. STATUS = {"NO_ACTION": 0, "CANDIDATE": 10, "INSUFFICIENT_DATA": 20, "HOST_BEHAVIOR_SHIFT": 30, @@ -203,17 +329,41 @@ def by_scope(recs): return dict(out) +def eligible_anchor(recs): + """The newest record that may speak for a population -> record or None. + + Eligibility comes FIRST. The anchor used to be the newest record of any kind, and the very next + line threw it away for being INVALID — after it had already decided which scope, which workload + class and which corpus size every other record was measured against. One broken sweep at the + top of the file could strand an entire eligible population. + """ + usable = [r for r in recs if record_quality(r) not in ("INVALID", "EMPTY")] + return max(usable, key=lambda r: r.get("ts") or 0) if usable else None + + def comparable(recs): - """Records that may be compared with the newest one: same scope, same workload class, - compatible corpus size, usable evidence quality.""" + """Records that may be compared with the newest ELIGIBLE one: same scope, same workload class, + compatible corpus size, quality that can carry a comparison at all. + + INVALID and EMPTY records are dropped here; PARTIAL, DEGRADED and UNKNOWN ones stay, because + they ARE part of the population and the report should show them. What they cannot do is carry + a promotion — that is the promotion gate's job, and it reads history_quality() over exactly the + records this function kept. + """ if not recs: return [], [] - newest = max(recs, key=lambda r: r.get("ts") or 0) + newest = eligible_anchor(recs) + if newest is None: + return [], [(r, "evidence quality " + record_quality(r)) for r in recs] n_turn = newest.get("turns") or 0 n_scope = scope_of(newest) n_work = str(newest.get("workload_class") or "") keep, dropped = [], [] for r in recs: + q = record_quality(r) + if q in ("INVALID", "EMPTY"): + dropped.append((r, "evidence quality " + q)) + continue if scope_of(r) != n_scope: dropped.append((r, f"scope {scope_of(r)!r} vs {n_scope!r}")) continue @@ -221,9 +371,6 @@ def comparable(recs): dropped.append((r, f"workload class {str(r.get('workload_class') or '')!r} vs {n_work!r} " "— WORKLOAD_SHIFT, not a trend")) continue - if r.get("evidence_quality") in ("INVALID", "EMPTY"): - dropped.append((r, "evidence quality " + str(r.get("evidence_quality")))) - continue t = r.get("turns") or 0 if n_turn and t and max(n_turn, t) / min(n_turn, t) > CORPUS_TURN_RATIO_MAX: dropped.append((r, f"corpus {t:,} turns vs {n_turn:,}")) @@ -336,8 +483,14 @@ def finding(cid, state, headline, evidence, scope="default", bucket=0, hypothesi def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): out = [] - quality = (live or {}).get("quality", "COMPLETE") if live else "COMPLETE" - usable = accept_partial or quality == "COMPLETE" + # Every candidate-producing path below states which evidence it rests on, and asks the same + # question about exactly that evidence. A finding drawing on the live sweep is not blocked by a + # partial history it never read, and a finding drawing on history is not waved through because + # today's sweep happened to be clean. + quality = sweep_quality(live) if live else "COMPLETE" + hist_q = history_quality(hist.get("comparable", [])) + usable = may_promote(quality, accept_partial) # findings that rest on the live sweep + hist_usable = may_promote(hist_q, accept_partial) # findings that rest on the history run_ids = [r.get("run_id") for r in hist.get("comparable", []) if r.get("run_id")] # 1. concentration — only sayable over a corpus, never over one session, and never from a @@ -395,19 +548,27 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): f"which is the same movement, not a second one" continue reported.append((k, per_month)) - if True: - cid = "trend-" + re.sub(r"[^a-z0-9]+", "-", k.lower()).strip("-") - spike = (" — but the last value sits z=%+.1f from its own history, so this slope " - "may be one spike rather than a shift" % t["z"]) if abs(t["z"]) >= TREND_MIN_Z else "" - out.append(finding(cid, "CANDIDATE", f"{k} share is moving", - f"{per_month:+.1f} pp/month over {t['n']} records in scope {scope!r}, " - f"last value z={t['z']:+.1f}{spike}", scope=scope, - # DIRECTION, not magnitude. A fitted slope decays as its window - # grows even when the world stopped moving, so bucketing the - # magnitude mints a fresh proposal every time the estimator - # settles. One sustained movement is one proposal; a REVERSAL - # is a new one, which is exactly when a human should look again. - bucket=1 if per_month > 0 else -1, + cid = "trend-" + re.sub(r"[^a-z0-9]+", "-", k.lower()).strip("-") + spike = (" — but the last value sits z=%+.1f from its own history, so this slope " + "may be one spike rather than a shift" % t["z"]) if abs(t["z"]) >= TREND_MIN_Z else "" + # DIRECTION, not magnitude. A fitted slope decays as its window grows even when the + # world stopped moving, so bucketing the magnitude mints a fresh proposal every time + # the estimator settles. One sustained movement is one proposal; a REVERSAL is a new + # one, which is exactly when a human should look again. + direction = 1 if per_month > 0 else -1 + ev = (f"{per_month:+.1f} pp/month over {t['n']} records in scope {scope!r}, " + f"last value z={t['z']:+.1f}{spike} (evidence {hist_q})") + if not hist_usable: + # The movement is real arithmetic over records that cannot say how they were + # acquired, so it is reported and never promoted. A trend over bounded sweeps is + # a trend in the sample, not in the population. + out.append(finding(cid, "OBSERVED", f"{k} share is moving", + ev + f" — history evidence is {hist_q}, not a population", + scope=scope, bucket=direction, + metric=f"slope of {k} share", invariants=(SCOPE_INVARIANT,))) + else: + out.append(finding(cid, "CANDIDATE", f"{k} share is moving", ev, scope=scope, + bucket=direction, hypothesis=f"the change in {k} is a shift, not a spike, and has a cause worth naming", metric=f"slope of {k} share", effect="unknown until measured", risk="a trend can come from the workload, not from SameWrite", @@ -415,6 +576,10 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): invariants=(SCOPE_INVARIANT,), benchmark="identify the cause before proposing a rule")) # 3. the guard: does it still pay for itself? (rule retirement is a first-class outcome) + # Its evidence is the hook's OWN ledger — writes it saw, no-ops it prevented — not a sample + # of transcripts. No sweep or history quality can make it better or worse, so the eligibility + # decision here is explicit and separate: the ledger's own sample size is its gate. + ledger_usable = bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES if ledger and ledger["writes"]: if ledger["writes"] >= LEDGER_MIN_WRITES: if ledger["rate"] >= LEDGER_MIN_NOOP_RATE: @@ -423,7 +588,9 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): f"({ledger['rate']:.1f}% >= {LEDGER_MIN_NOOP_RATE}%)", scope=scope, bucket=bucket_of(ledger["rate"]), metric="ledger deny rate")) else: - out.append(finding("noop-guard-retire", "CANDIDATE", "the no-op guard may have stopped paying", + out.append(finding("noop-guard-retire", + "CANDIDATE" if ledger_usable else "OBSERVED", + "the no-op guard may have stopped paying", f"{ledger['prevented']} of {ledger['writes']} writes prevented " f"({ledger['rate']:.1f}% < {LEDGER_MIN_NOOP_RATE}%)", scope=scope, bucket=bucket_of(ledger["rate"]), @@ -467,14 +634,32 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): def overall_status(findings, hist, live, pop, accept_partial): + """The run's one word — and the thing that decides whether anything is written. + + CANDIDATE now precedes PARTIAL_EVIDENCE, which it did not before. That is not a loosening: a + finding only reaches CANDIDATE after its OWN evidence passed the promotion gate, so a run that + has one is a run whose promoted findings rest on evidence that was eligible. The old order + blocked a valid history-only finding because today's live sweep happened to be bounded — a + refusal by evidence the finding never used. + """ if pop.get("host_shift"): return "HOST_BEHAVIOR_SHIFT" - q = (live or {}).get("quality", "COMPLETE") if live else "COMPLETE" - if q in ("PARTIAL", "INVALID") and not accept_partial: - return "PARTIAL_EVIDENCE" if any(f["state"] == "CANDIDATE" for f in findings): return "CANDIDATE" - if not live and len(hist.get("comparable", [])) < MIN_HISTORY_FOR_TREND: + # Evidence that EXISTS and was refused is a different answer from evidence that is missing: + # PARTIAL_EVIDENCE tells the caller a flag or a full sweep would change the outcome, while + # INSUFFICIENT_DATA tells them to keep collecting. + refused = [] + if live: + refused.append(sweep_quality(live)) + comp = hist.get("comparable", []) + if comp: + refused.append(history_quality(comp)) + elif any(record_quality(r) in ("INVALID", "EMPTY") for r, _why in hist.get("dropped", [])): + refused.append("INVALID") + if any(not may_promote(q, accept_partial) for q in refused): + return "PARTIAL_EVIDENCE" + if not live and len(comp) < MIN_HISTORY_FOR_TREND: return "INSUFFICIENT_DATA" return "NO_ACTION" @@ -554,12 +739,22 @@ def spec_text(f): """ -def emit_candidates(findings, outdir): - """Atomic, deduplicated, and never inside a governed tree by default. +def emit_candidates(findings, outdir, status): + """Atomic, deduplicated, never inside a governed tree by default — and only under a status that + permits promotion at all. -> (written, existing, failed). A candidate whose id already exists is NOT rewritten: the - evidence bucket is part of the id, so a file reappears only when the evidence actually moved.""" + evidence bucket is part of the id, so a file reappears only when the evidence actually moved. + + The status gate lives HERE rather than at the one call site that used to need it. A status that + refuses promotion while a file lands on disk is the worst outcome available: the operator reads + exit 30 or 40 and the next reader of the directory finds a specification that looks approved. + Putting the gate in the emitter means every caller — the CLI, the long-run simulator, a future + scheduler — routes through it, and no new caller can forget. + """ written, existing, failed = [], [], [] + if status != "CANDIDATE": + return written, existing, failed for f in findings: if f["state"] != "CANDIDATE": continue @@ -631,7 +826,12 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop L.append(f" live scan : {live['sessions']} sessions / {live['turns']:,} turns " f"({live['scanned']} transcripts, {live['short']} below the turn floor, " f"{live.get('unreadable', 0)} unreadable, {live.get('oversize', 0)} oversized lines)") - L.append(f" evidence : {live.get('quality', 'COMPLETE')}") + L.append(f" evidence : {sweep_quality(live)}" + + (f" (the sweep reports {live.get('quality')})" + if sweep_quality(live) != live.get("quality") else "")) + if hist["comparable"]: + L.append(f" history : {history_quality(hist['comparable'])} " + f"(worst of {len(hist['comparable'])} eligible records)") L.append("") if live and live["sessions"]: C = sum(live["carry"].values()) or 1 @@ -711,7 +911,12 @@ def main(argv=None): if a.scope_id: scope, scoped = a.scope_id, scopes.get(a.scope_id, []) elif recs: - scope = scope_of(max(recs, key=lambda r: r.get("ts") or 0)) + # Which scope gets analysed is an anchor too: taking the newest record of ANY quality let a + # single INVALID sweep in another agent's scope send the whole run to a population that was + # never going to be analysable. The newest record that can speak chooses; if none can, the + # newest record still names a scope, and the status below says why nothing came of it. + anchor = eligible_anchor(recs) or max(recs, key=lambda r: r.get("ts") or 0) + scope = scope_of(anchor) scoped = scopes[scope] else: scope, scoped = "default", [] @@ -731,7 +936,9 @@ def main(argv=None): if paths: live = carry.accumulate(paths, min_turns=a.min_turns, max_files=a.max_files) try: - listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths) + # The SAME bounded sample the sweep used. Two selections of "the newest N" is two + # populations reported as one. + listing, uses, sess = skills_tool.scan(carry.bounded_paths(paths, a.max_files)) if listing: ent = skills_tool.parse_listing(listing) rows = skills_tool.tally(ent, uses, sess) @@ -754,7 +961,7 @@ def main(argv=None): if not got: status = "ALREADY_RUNNING" else: - written, existing, failed = emit_candidates(findings, a.emit_candidate) + written, existing, failed = emit_candidates(findings, a.emit_candidate, status) if a.json: print(json.dumps({ @@ -767,8 +974,11 @@ def main(argv=None): "status": status, "status_code": STATUS.get(status, STATUS["INTERNAL_ERROR"]), "scope": {"analysed": scope, "known": scopes_seen, "records_in_scope": len(scoped), "comparable": len(keep)}, - "evidence_quality": (live or {}).get("quality", "NO_SCAN") if live else "NO_SCAN", + # DERIVED, not the producer's summary word: a sweep that lost records reads DEGRADED + # here even though the 1.3 label for it is PARTIAL (output_schema_version 2). + "evidence_quality": sweep_quality(live) if live else "NO_SCAN", "history": {"records": len(recs), "comparable": len(keep), + "quality": history_quality(keep), "rejected": dict(rejected), "time_order": hist["time_order"]}, "ledger": ledger, "live": ({"sessions": live["sessions"], "turns": live["turns"], "scanned": live["scanned"], From 33867801212f9bbc9692266d830330546801dfc2 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 21:50:08 +0000 Subject: [PATCH 03/26] test: a mutant per repaired invariant, and CI runs the new suite MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Six new mutants, each restoring exactly one of the defects the frozen matrix found, each RED on the mutant and GREEN on the real source: M_PARTIAL_PROMOTES may_promote() returns True M_INVALID_ANCHOR the anchor is chosen before the filter again M_HOST_SHIFT_WRITES the emitter's status gate is deleted M_MALFORMED_COMPLETE `malformed` leaves the loss vocabulary M_BOUND_SAMPLE_DIVERGES the listing scan takes its own slice again M_TREND_QUALITY_BYPASS the history trend stops asking None of them is a string mutant: every one changes behaviour that the oracle observes through a real run — a status, a comparable count, a file on disk, a sweep's quality, a JSON listing field. Two existing mutation patterns moved with the code and were re-pointed (the PARTIAL gate and the trend's direction), and the shared record fixture states its acquisition counters for the same reason the other fixtures do. README: 1311 assertions in seventeen suites (CI enforces this number). Co-Authored-By: Claude Opus 5 --- .github/workflows/test.yml | 4 ++ README.md | 2 +- tests/test_mutation.py | 142 +++++++++++++++++++++++++++++++++++-- 3 files changed, 142 insertions(+), 6 deletions(-) diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index b7a4a5f..cdbf18f 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -49,6 +49,10 @@ jobs: run: python3 tests/test_evidence_phase2.py - name: offline optimizer (privacy canary, corruption, zero policy mutation) run: python3 tests/test_optimize.py + - name: legacy evidence integrity - the frozen v1.4.2 counterexample matrix + # Ambiguous, incomplete, invalid, non-comparable or host-shifted evidence must not produce + # a candidate specification unless the bounded-evidence exception was explicitly asked for. + run: python3 tests/test_evidence_integrity.py - name: the 1.4.0 release ships shadow evaluation and nothing that writes # Sebuah rilis paling mudah "menyalakan" sesuatu tanpa sengaja saat versinya dinaikkan. # Suite ini yang membuat kalimat "promotion/persistence tidak aktif" bisa diperiksa. diff --git a/README.md b/README.md index 93cccf8..6baa590 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1201 assertions in sixteen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1311 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 7d89c17..4c1bd83 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -41,9 +41,12 @@ def acc(sessions=40, turns=1000, quality="COMPLETE", carry_map=None): "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "quality": quality, "carry": collections.Counter(carry_map or {{"Bash": 90, "Read": 10}})}} def rec(ts, shares, scope="default", turns=1000, run_id=None): + # acquisition counters included: since 1.4.2 a record claiming COMPLETE without the + # counters its writer always wrote is UNKNOWN, and UNKNOWN cannot promote return {{"schema_version": 2, "record_type": "carry_run", "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10**7, "scope_id": scope, "workload_class": "", "evidence_quality": "COMPLETE", "run_id": run_id or carry.new_run_id(), + "scanned": 100, "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "shares": shares, "bpt": {{k: 1.0 for k in shares}}}} def w(path, rows): with open(path, "w", encoding="utf-8") as fh: @@ -122,7 +125,7 @@ def mutant(pairs): """), ("fail-closed: bukti PARTIAL tak boleh melahirkan kandidat", - [("optimize.py", 'usable = accept_partial or quality == "COMPLETE"', "usable = True")], + [("optimize.py", "usable = may_promote(quality, accept_partial)", "usable = True")], """ f = optimize.analyse(acc(quality="PARTIAL"), EMPTY_HIST, None, None, scope="s") assert [x["state"] for x in f] == ["OBSERVED"], "kandidat lahir dari sapuan setengah jadi" @@ -148,8 +151,8 @@ def mutant(pairs): """ f = optimize.analyse(acc(), EMPTY_HIST, None, None, scope="x")[0] out = os.path.join(D, "c") - optimize.emit_candidates([f], out) - w2, e2, _ = optimize.emit_candidates([f], out) + optimize.emit_candidates([f], out, "CANDIDATE") + w2, e2, _ = optimize.emit_candidates([f], out, "CANDIDATE") assert (len(w2), len(e2)) == (0, 1), "penjadwal menulis ulang usulan yang sama tiap siklus" """), @@ -375,8 +378,8 @@ def mutant(pairs): """), ("identitas tren stabil: jendela membesar bukan usulan baru", - [("optimize.py", " bucket=1 if per_month > 0 else -1,", - " bucket=bucket_of(abs(per_month)),")], + [("optimize.py", " direction = 1 if per_month > 0 else -1", + " direction = bucket_of(abs(per_month))")], """ # Deret harus NAIK lalu MENDATAR. Deret linier sempurna punya kemiringan yang sama di # jendela mana pun, jadi ia tak bisa membedakan identitas-dari-arah dari @@ -538,6 +541,135 @@ def hist_of(n): c = evidence_history.read_container(p) assert c.lines_rejected == 1, c.lines_rejected """), + + # ------------------------------------ v1.4.2: the legacy optimizer's evidence-integrity gate + ("M_PARTIAL_PROMOTES: riwayat PARTIAL tak boleh dipromosikan tanpa izin eksplisit", + [("optimize.py", + 'return quality == "COMPLETE" or (quality == "PARTIAL" and bool(accept_partial))', + "return True")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + for r in rows: + r["evidence_quality"] = "PARTIAL" + r["skipped_by_limit"] = 5 + keep, dropped = optimize.comparable(rows) + h = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": dropped, + "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="p") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + st = optimize.overall_status(f, h, None, optimize.population(keep), False) + assert st == "PARTIAL_EVIDENCE", st + """), + + ("M_INVALID_ANCHOR: jangkar komparabilitas hanya dari rekaman yang lolos filternya sendiri", + [("optimize.py", + 'usable = [r for r in recs if record_quality(r) not in ("INVALID", "EMPTY")]', + "usable = list(recs)")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + bad = rec(100 + 9 * 604800, {"Bash": 55.0, "Read": 45.0}, scope="a", turns=100000) + bad["evidence_quality"] = "INVALID" + bad["sessions"] = 0 + keep, _d = optimize.comparable(rows + [bad]) + assert len(keep) == 6, len(keep) + """), + + ("M_HOST_SHIFT_WRITES: status yang menolak promosi tidak menulis berkas", + [("optimize.py", ' if status != "CANDIDATE":\n return written, existing, failed\n', + "")], + """ + f = optimize.finding("x", "CANDIDATE", "h", "e", scope="s", bucket=1) + out = os.path.join(D, "gate") + for st in ("HOST_BEHAVIOR_SHIFT", "PARTIAL_EVIDENCE", "INSUFFICIENT_DATA", "ALREADY_RUNNING"): + w, e, fail = optimize.emit_candidates([f], out, st) + assert (w, e, fail) == ([], [], []), (st, w, e, fail) + assert not os.path.exists(out), st + w, e, fail = optimize.emit_candidates([f], out, "CANDIDATE") + assert len(w) == 1, (w, e, fail) + """), + + ("M_MALFORMED_COMPLETE: transcript dengan baris JSON robek bukan sapuan COMPLETE", + [("carry.py", + 'LOSS_FIELDS = ("unreadable", "oversize", "malformed", "identity_changed", "conflicted_sources")', + 'LOSS_FIELDS = ("unreadable", "oversize", "identity_changed", "conflicted_sources")')], + """ + d = tempfile.mkdtemp() + p = os.path.join(d, "t.jsonl") + rows = [] + for t in range(30): + rows.append(json.dumps({"type": "assistant", "message": {"id": "m%d" % t, + "usage": {"output_tokens": 5}, + "content": [{"type": "tool_use", "name": "Bash", + "input": {"command": "ls"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 200}]}})) + rows[11] = chr(123) + '"type": "assistant", "message": {"usage": {"out' + open(p, "w").write(chr(10).join(rows) + chr(10)) + a = carry.accumulate([p], min_turns=1) + assert a["malformed"] == 1, a["malformed"] + assert a["quality"] != "COMPLETE", a["quality"] + assert optimize.sweep_quality(a) == "DEGRADED", optimize.sweep_quality(a) + """), + + ("M_BOUND_SAMPLE_DIVERGES: sapuan dan pindaian listing memakai sampel terbatas yang SAMA", + [("optimize.py", + "listing, uses, sess = skills_tool.scan(carry.bounded_paths(paths, a.max_files))", + "listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths)")], + """ + import subprocess + d = tempfile.mkdtemp() + paths = [] + for n, name in enumerate(("a.jsonl", "b.jsonl", "c.jsonl")): + q = os.path.join(d, name) + rows = [] + if name == "a.jsonl": + rows.append(json.dumps({"type": "user", "attachment": {"type": "skill_listing", + "content": chr(10).join(["- alpha: does a thing described at length", + "- beta: does another thing, also at length", + ""])}})) + for t in range(40): + rows.append(json.dumps({"type": "assistant", "message": {"id": "m%d" % t, + "usage": {"output_tokens": 5}, + "content": [{"type": "tool_use", "name": "Bash", + "input": {"command": "ls"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 200}]}})) + open(q, "w").write(chr(10).join(rows) + chr(10)) + os.utime(q, (1750000000 + n * 1000, 1750000000 + n * 1000)) + paths.append(q) + empty = os.path.join(d, "none.jsonl") + open(empty, "w").write("") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", empty, "--ledger", empty, "--json", + "--scan"] + paths + ["--max-files", "1", "--min-turns", "1"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["listing"] is None, j["listing"] + r2 = subprocess.run([sys.executable, opt, "--history", empty, "--ledger", empty, "--json", + "--scan"] + paths + ["--min-turns", "1"], + capture_output=True, text=True, timeout=300) + assert json.loads(r2.stdout)["listing"], "kontrol: sapuan penuh HARUS membaca listing itu" + """), + + ("M_TREND_QUALITY_BYPASS: tren dari riwayat tak layak dilaporkan, bukan dipromosikan", + [("optimize.py", " if not hist_usable:", " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="q") + for i in range(6)] + for r in rows: + r["evidence_quality"] = "PARTIAL" + r["skipped_by_limit"] = 5 + keep, dropped = optimize.comparable(rows) + h = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": dropped, + "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="q") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + g = optimize.analyse(None, h, None, None, scope="q", accept_partial=True) + assert [x["state"] for x in g] == ["CANDIDATE"], [x["state"] for x in g] + """), + ] From f2a43944cb52bdf49ac5e9a32dd0a9762e2da1a6 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 21:57:33 +0000 Subject: [PATCH 04/26] fix: four more ways evidence could claim more than it showed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three from an adversarial pass over the new law, one from a cross-family review of the same diff. Each has a case in the suite; the first also has a mutant. - A retry that reuses a run_id could launder its own sweep. The duplicate is dropped as an OBSERVATION, but the quality it reported is part of what that id actually saw, so the worse of the two now travels with the record that survives (QUALITY_FLOOR, in memory only — nothing is written back to the history file). - `schema_version` of "2" (a string) or 2.0 (a float) was accepted as schema 2 by int(). A version this reader cannot name is not a newer generation to trust; it is an older one to doubt. - Zero counters say "nothing went wrong"; they do not say a sweep happened. A schema-2 record carrying zeroed counters and no `sessions`, `turns` or `carry_bytes` at all read COMPLETE — a share vector with no population behind it. It now reads UNKNOWN. (cross-family review) - One source whose mtime could not be read sent the ENTIRE bounded selection back to a discovery-order slice — the exact sample bounded_paths() exists to avoid. The fallback is now per source: a file that cannot be dated cannot claim to be the newest, and the rest still order by mtime. (cross-family review) README: 1330 assertions. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 48 ++++++++++++++++++++++++++++++++ tests/test_mutation.py | 22 +++++++++++++++ tools/carry.py | 17 +++++++---- tools/optimize.py | 35 +++++++++++++++++++---- 5 files changed, 112 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index 6baa590..7f43837 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1311 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1330 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 75b1726..df000d2 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -239,6 +239,54 @@ def main(): case("U1 (--accept-partial does not rescue it)", old, "PARTIAL_EVIDENCE", 40, 0, extra=["--accept-partial"]) + # ---------------------------------------------------------------- a retry cannot launder + print("\nR142_07 - a retry cannot upgrade what its own run_id saw") + dup = [rec(i, 30.0 + 3.0 * i) for i in range(5)] + dup.append(rec(5, 45.0)) + dup[-1]["run_id"] = "dup" + worse = rec(5, 45.0, quality="PARTIAL", skipped=7) + worse["run_id"] = "dup" + dup.append(worse) + hist_path = write(os.path.join(d, "dup.jsonl"), dup) + recs, rejected, _lines = optimize.load_history(hist_path) + check("the retry is dropped as an observation", len(recs), 6) + check("and counted", rejected["duplicate run_id (retry)"], 1) + keep, _dropped = optimize.comparable(recs) + check("but its quality travels with the record that survived", + optimize.history_quality(keep), "PARTIAL") + case("R142_07", dup, "PARTIAL_EVIDENCE", 40, 0, history_quality="PARTIAL") + case("R142_07 (--accept-partial adopts the bound)", dup, "CANDIDATE", 10, 1, + extra=["--accept-partial"]) + check("a schema_version this reader cannot name is not a newer one", + optimize.record_quality(dict(rec(0, 40.0), schema_version="2")), "UNKNOWN") + check("nor is a float one", + optimize.record_quality(dict(rec(0, 40.0), schema_version=2.0)), "UNKNOWN") + + # ------------------------------------------------- round-1 cross-family review findings + print("\nR142_08 - what a record must SHOW before its zeroes mean anything") + no_corpus = {"schema_version": 2, "record_type": "carry_run", "ts": TS0, + "evidence_quality": "COMPLETE", "unreadable": 0, "oversize": 0, + "skipped_by_limit": 0, "shares": {"Bash": 60.0, "Read": 40.0}, + "bpt": {"Bash": 1.0, "Read": 1.0}} + check("zero counters do not say a sweep happened", + optimize.record_quality(no_corpus), "UNKNOWN") + for missing in ("sessions", "turns", "carry_bytes"): + partial_rec = {k: v for k, v in rec(0, 40.0).items() if k != missing} + check(f"a record without {missing} cannot attest completeness", + optimize.record_quality(partial_rec), "UNKNOWN") + + print("\nR142_09 - one undateable source does not collapse the bound") + undated = [] + for n, name in enumerate(("x.jsonl", "y.jsonl", "z.jsonl")): + q = transcript(os.path.join(d, name), turns=5) + os.utime(q, (TS0 + n * 1000, TS0 + n * 1000)) # z newest + undated.append(q) + os.remove(undated[2]) # ...and now undateable + picked = [os.path.basename(q) for q in carry.bounded_paths(undated, 2)] + check("the datable sources still order by mtime", picked, ["y.jsonl", "x.jsonl"]) + check("and discovery order still does not matter", + carry.bounded_paths(list(reversed(undated)), 2), carry.bounded_paths(undated, 2)) + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") good = [rec(i, 30.0 + 3.0 * i) for i in range(6)] diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 4c1bd83..d5a10ee 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -670,6 +670,28 @@ def hist_of(n): assert [x["state"] for x in g] == ["CANDIDATE"], [x["state"] for x in g] """), + + ("M_DEDUP_LAUNDERS: retry dgn run_id sama tak boleh menaikkan kualitas yang bertahan", + [("optimize.py", """ kept = seen[rid] + worse = worst_quality([record_quality(kept), record_quality(r)]) + if worse != record_quality(kept): + kept[QUALITY_FLOOR] = worse + continue""", + " continue")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="d") + for i in range(5)] + good = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="d", run_id="dup") + bad = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="d", run_id="dup") + bad["evidence_quality"] = "PARTIAL" + bad["skipped_by_limit"] = 7 + p = w(os.path.join(D, "dup.jsonl"), rows + [good, bad]) + recs, rej, _ = optimize.load_history(p) + keep, _d = optimize.comparable(recs) + q = optimize.history_quality(keep) + assert q == "PARTIAL", q + """), + ] diff --git a/tools/carry.py b/tools/carry.py index 1a6a2f8..22088f1 100755 --- a/tools/carry.py +++ b/tools/carry.py @@ -253,11 +253,18 @@ def bounded_paths(paths, max_files): paths = list(paths) if not max_files or len(paths) <= max_files: return paths - try: - return sorted(paths, key=lambda q: os.path.getmtime(q), reverse=True)[:max_files] - except OSError: - # A path that lost its mtime cannot order the sample; the bound still has to hold. - return paths[:max_files] + + def age(q): + # Per SOURCE, not per sweep: one file whose mtime cannot be read used to send the whole + # selection back to a discovery-order slice, which is the very sample this function + # exists to avoid. A source that cannot be dated simply cannot claim to be the newest. + # (cross-family review, round 1) + try: + return os.path.getmtime(q) + except OSError: + return float("-inf") + + return sorted(paths, key=age, reverse=True)[:max_files] def sweep_label(facts): diff --git a/tools/optimize.py b/tools/optimize.py index a46e87c..48ada24 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -77,6 +77,9 @@ # The counters every schema-2 record's writer wrote, split by what they mean. Nothing here is # invented for an older schema: a record that never carried these cannot be asked to show them, # and is UNKNOWN rather than trusted. +# Set by load_history() when a duplicate run_id shows a worse sweep than the record that survives. +# It is a private, in-memory annotation; nothing writes it back to a file. +QUALITY_FLOOR = "_evidence_quality_floor" RECORD_LOSS_COUNTERS = ("unreadable", "oversize") RECORD_BOUND_COUNTERS = ("skipped_by_limit",) RECORD_COUNTERS = RECORD_LOSS_COUNTERS + RECORD_BOUND_COUNTERS @@ -124,6 +127,12 @@ def _derived_record_quality(rec, schema): return "DEGRADED" if any(nums.get(k, 0) for k in RECORD_BOUND_COUNTERS): return "PARTIAL" + if not all(k in nums for k in ("sessions", "turns", "carry_bytes")): + # Zero counters say "nothing went wrong"; they do not say a sweep happened. A record whose + # corpus fields are absent entirely has a share vector and no population behind it, and + # reading its zeroed counters as completeness would trust a measurement nobody took. + # (cross-family review, round 1) + return "UNKNOWN" if schema >= SCHEMA_WITH_QUALITY and not all(k in nums for k in RECORD_COUNTERS): return "UNKNOWN" # claims a completeness it cannot show return "COMPLETE" @@ -142,16 +151,21 @@ def record_quality(rec): """ if not isinstance(rec, dict): return "INVALID" - try: - schema = int(rec.get("schema_version") or 0) - except (TypeError, ValueError): + schema = rec.get("schema_version", 0) + if isinstance(schema, bool) or not isinstance(schema, int): + # "2" is a string, 2.0 is a float, True is neither. A version this reader cannot name is + # not a newer generation to be trusted; it is an older one to be doubted. schema = 0 claimed = rec.get("evidence_quality") if not isinstance(claimed, str) or claimed not in QUALITY_RANK: claimed = "UNKNOWN" if schema < SCHEMA_WITH_QUALITY: claimed = "UNKNOWN" - return worst_quality([claimed, _derived_record_quality(rec, schema)]) + qualities = [claimed, _derived_record_quality(rec, schema)] + floor = rec.get(QUALITY_FLOOR) + if floor: # absent is not UNKNOWN: most records carry no floor at all + qualities.append(floor) + return worst_quality(qualities) def sweep_quality(live): @@ -310,14 +324,23 @@ def load_history(path): continue ok, why = valid_record(o) (recs.append(o) if ok else rejected.__setitem__(why, rejected[why] + 1)) - seen, uniq = set(), [] + seen, uniq = {}, [] for r in recs: rid = r.get("run_id") if isinstance(rid, str) and rid: if rid in seen: rejected["duplicate run_id (retry)"] += 1 + # The retry is dropped as an OBSERVATION, never as provenance. One run_id that + # says COMPLETE once and PARTIAL once cannot be resolved in favour of the better + # claim: the sweep that reported less is part of what this id actually saw, and + # keeping the better one would let a retry launder a bounded sweep into a + # complete one. The floor travels with the record the law already reads. + kept = seen[rid] + worse = worst_quality([record_quality(kept), record_quality(r)]) + if worse != record_quality(kept): + kept[QUALITY_FLOOR] = worse continue - seen.add(rid) + seen[rid] = r uniq.append(r) return uniq, rejected, lines From 13d2cded2e14573a13fa4c082905105e8b737170 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 21:57:43 +0000 Subject: [PATCH 05/26] docs: record the cases added after the matrix was frozen Four rows, each from an attack on the repair rather than on the original defect, kept in their own section so the frozen matrix stays readable as what it was when it was frozen. No expectation in section 1 changed. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index 0bee373..a9ffd2a 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -67,3 +67,15 @@ S_LOCK ALREADY_RUNNING exit 41 files 0 comparable 6 (already correc ## 3. Deviations from the frozen expectations None. + +## 4. Cases added AFTER the freeze + +Not changed expectations — new rows, each from an attack on the repair itself rather than on the +original defect. They are listed separately so the frozen matrix stays readable as what it was. + +| case | fixture | expectation | +|---|---|---| +| **R142_07** | two records sharing one `run_id`, the first `COMPLETE`, the retry `PARTIAL` with `skipped_by_limit=7` | the retry is dropped as an observation and counted, and the quality it reported travels with the record that survives: `PARTIAL_EVIDENCE` / exit 40 / 0 files, `CANDIDATE` with `--accept-partial` | +| **R142_07b** | `schema_version` of `"2"` (string) or `2.0` (float) | `UNKNOWN` — a version this reader cannot name is not a newer generation to trust | +| **R142_08** | schema-2 record with zeroed counters and no `sessions` / `turns` / `carry_bytes` at all | `UNKNOWN` — zero counters say nothing went wrong, not that a sweep happened (cross-family review, round 1) | +| **R142_09** | `--max-files 2` where one of three sources cannot be dated | the datable sources still order by mtime; one unreadable mtime no longer sends the whole selection back to a discovery-order slice (cross-family review, round 1) | From 909d1054a9bf575664005b38df258a4d0e923131 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 22:04:56 +0000 Subject: [PATCH 06/26] fix: four HIGHs a second reviewer family found, all of them "evidence nobody checked" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A cross-family adversarial pass over the repair itself. Every one of these produced a CANDIDATE and a file on disk before this commit. - Counters that cannot describe one sweep. `sessions=40` out of `scanned=1` — or out of `scanned=0` — read COMPLETE, because the law checked each counter's type and sign but never their relation. A session is a transcript that was scanned and cleared the turn floor, so the producer can never report more sessions than it scanned. That combination is now INVALID, and `scanned` joins the fields a schema-2 record must carry before its zeroes mean anything. - Damage to the history CONTAINER never reached the evidence decision. A torn line in the history file was counted in `history.rejected` and then ignored: the trend built from the surviving records was promoted as if nothing had been lost. `container_quality()` reads it as a loss (DEGRADED, so no flag adopts it) and both `analyse()` and `overall_status()` take the worst of it and the records. A refusal by design — current-generation evidence this optimizer does not read — and a deduplicated retry are NOT damage; both have controls. - A ledger that lost a line still retired the guard. 100 readable writes plus one torn line promoted `noop-guard-retire`. `ledger_usable` now requires `rejected == 0`, and a ledger that exists but cannot be opened is no longer indistinguishable from no ledger at all. - A run could report CANDIDATE and exit 10 while every candidate file failed to write. The status has to say what happened: nothing landed, so the run is INTERNAL_ERROR (exit 50), with the failure named in the JSON. Four more mutants, each RED on the mutant and GREEN on the real source (41/41 now): M_IMPOSSIBLE_COUNTERS, M_CONTAINER_DAMAGE_IGNORED, M_LEDGER_TORN_PROMOTES, M_EMIT_FAILURE_SILENT. README: 1360 assertions. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 66 ++++++++++++++++++++++++++++- tests/test_mutation.py | 73 ++++++++++++++++++++++++++++++++ tools/optimize.py | 54 ++++++++++++++++++++--- 4 files changed, 187 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 7f43837..8ad6fb2 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1330 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1360 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index df000d2..3cdb8e0 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -117,8 +117,13 @@ def transcript(path, turns=80, listing=False, torn=False): return path +EMPTY_HIST = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + + def main(): d = tempfile.mkdtemp(prefix="sw-142-fx-") + good_rows = [rec(i, 30.0 + 3.0 * i) for i in range(6)] # ---------------------------------------------------------------- the law itself print("\nlegacy evidence-quality law") @@ -287,9 +292,68 @@ def main(): check("and discovery order still does not matter", carry.bounded_paths(list(reversed(undated)), 2), carry.bounded_paths(undated, 2)) + print("\nR142_10 - counters that cannot describe one sweep") + check("more sessions than transcripts scanned", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=1)), "INVALID") + check("sessions out of a sweep that scanned nothing", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=0)), "INVALID") + check("...and the bound flag does not rescue that either", + optimize.may_promote(optimize.record_quality( + rec(0, 40.0, quality="PARTIAL", sessions=40, scanned=0, skipped=5)), True), False) + check("control: sessions within what was scanned", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=40)), "COMPLETE") + case("R142_10", [rec(i, 30.0 + 3.0 * i, sessions=40, scanned=1) for i in range(6)], + "PARTIAL_EVIDENCE", 40, 0, comparable=0) + + print("\nR142_11 - a torn line in the history file is lost evidence, not a footnote") + torn_hist = [json.dumps(r) for r in good_rows] + ['{"torn":'] + j = case("R142_11", torn_hist, "PARTIAL_EVIDENCE", 40, 0, history_quality="DEGRADED") + check("the torn line is still counted", j["history"]["rejected"].get("unparseable line"), 1) + case("R142_11 (--accept-partial does not adopt a loss)", torn_hist, + "PARTIAL_EVIDENCE", 40, 0, extra=["--accept-partial"]) + envelope_line = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}) + case("R142_11 control: a refusal by design is not damage", + [json.dumps(r) for r in good_rows] + [envelope_line], "CANDIDATE", 10, 1, + history_quality="COMPLETE") + + print("\nR142_12 - a ledger that lost a line is not a field sample") + ledger_dir = tempfile.mkdtemp(dir=d) + clean_ledger = os.path.join(ledger_dir, "clean.jsonl") + with open(clean_ledger, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n") + torn_ledger = os.path.join(ledger_dir, "torn.jsonl") + with open(torn_ledger, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n" + '{"event":' + "\n") + clean = optimize.load_ledger(clean_ledger) + lost = optimize.load_ledger(torn_ledger) + check("control: a clean sample retires the guard", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), clean, None)], ["CANDIDATE"]) + check("a sample that lost a line only observes", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), lost, None)], ["OBSERVED"]) + check("a ledger that exists and cannot be read is not 'no ledger'", + (optimize.load_ledger(ledger_dir) or {}).get("rejected"), 1) + + print("\nR142_13 - a status that promised a candidate and could not write one") + fail_dir = tempfile.mkdtemp(dir=d) + findings = optimize.analyse(None, {"comparable": good_rows, "total": 6, "in_scope": 6, + "rejected": {}, "dropped": [], "time_order": "ok"}, + None, None) + blocked = [f["candidate_id"] for f in findings if f["state"] == "CANDIDATE"][0] + open(os.path.join(fail_dir, blocked), "w").write("a file where a directory must go") + hist_file = write(os.path.join(fail_dir, "h.jsonl"), good_rows) + argv = [sys.executable, OPT, "--history", hist_file, "--ledger", + os.path.join(fail_dir, "none.jsonl"), "--scan", "--emit-candidate", fail_dir, + "--json", "--strict-exit"] + proc = subprocess.run(argv, capture_output=True, text=True, timeout=300) + jf = json.loads(proc.stdout) + check("nothing landed, so the status does not claim it did", jf["status"], "INTERNAL_ERROR") + check("and the exit code follows the status", proc.returncode, 50) + check("the failure is named", len(jf["candidates_failed"]), 1) + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") - good = [rec(i, 30.0 + 3.0 * i) for i in range(6)] + good = good_rows case("P1 clean COMPLETE population", good, "CANDIDATE", 10, 1, comparable=6, history_quality="COMPLETE") case("P3 an invalid record in ANOTHER scope does not poison this one", diff --git a/tests/test_mutation.py b/tests/test_mutation.py index d5a10ee..7286674 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -692,6 +692,79 @@ def hist_of(n): assert q == "PARTIAL", q """), + + ("M_IMPOSSIBLE_COUNTERS: sesi lebih banyak daripada transcript yang dipindai = mustahil", + [("optimize.py", ' if "scanned" in nums and nums["scanned"] < nums.get("sessions", 0):', + " if False:")], + """ + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r["scanned"] = 1 + q = optimize.record_quality(r) + assert q == "INVALID", q + """), + + ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history menurunkan kualitas evidence", + [("optimize.py", ' if count and any(str(reason).startswith(d) for d in CONTAINER_DAMAGE):', + " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="c") + for i in range(6)] + p = w(os.path.join(D, "torn_hist.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + recs, rej, _ = optimize.load_history(p) + keep, dropped = optimize.comparable(recs) + h = {"comparable": keep, "total": len(recs), "in_scope": len(recs), "rejected": rej, + "dropped": dropped, "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="c") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + st = optimize.overall_status(f, h, None, optimize.population(keep), True) + assert st == "PARTIAL_EVIDENCE", st + """), + + ("M_LEDGER_TORN_PROMOTES: ledger yang kehilangan baris bukan sampel lapangan", + [("optimize.py", """ ledger_usable = (bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES + and not ledger.get("rejected"))""", + " ledger_usable = bool(ledger) and ledger.get(\"writes\", 0) >= LEDGER_MIN_WRITES")], + """ + p = os.path.join(D, "led.jsonl") + with open(p, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + fh.write('{"event":' + chr(10)) + led = optimize.load_ledger(p) + assert led["writes"] == 100 and led["rejected"] == 1, led + EMPTY = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + f = optimize.analyse(None, EMPTY, led, None) + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + """), + + ("M_EMIT_FAILURE_SILENT: kandidat yang gagal ditulis tak boleh dilaporkan CANDIDATE", + [("optimize.py", """ if failed and not written:""", " if False:")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="e") + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), rows) + empty = os.path.join(d, "none.jsonl") + open(empty, "w").write("") + out = os.path.join(d, "cand") + os.makedirs(out) + keep, _dr = optimize.comparable(rows) + hh = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"} + cid = [x["candidate_id"] for x in optimize.analyse(None, hh, None, None, scope="e") + if x["state"] == "CANDIDATE"][0] + open(os.path.join(out, cid), "w").write("a file where a directory must go") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", + "--emit-candidate", out, "--json", "--strict-exit"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["status"] == "INTERNAL_ERROR", (j["status"], j["candidates_failed"]) + assert r.returncode == 50, r.returncode + """), + ] diff --git a/tools/optimize.py b/tools/optimize.py index 48ada24..5cda933 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -123,11 +123,17 @@ def _derived_record_quality(rec, schema): # Sessions were counted, so turns were counted and carry was measured. A record reporting # sessions with neither is not a quiet sweep, it is an impossible one. return "INVALID" + if "scanned" in nums and nums["scanned"] < nums.get("sessions", 0): + # A session is a transcript that was scanned AND cleared the turn floor, so the producer + # can never report more sessions than it scanned. Forty sessions out of one scanned file + # is not a sweep that went well; it is a record describing a sweep that cannot have + # happened. (cross-family review, round 1) + return "INVALID" if any(nums.get(k, 0) for k in RECORD_LOSS_COUNTERS): return "DEGRADED" if any(nums.get(k, 0) for k in RECORD_BOUND_COUNTERS): return "PARTIAL" - if not all(k in nums for k in ("sessions", "turns", "carry_bytes")): + if not all(k in nums for k in ("sessions", "turns", "carry_bytes", "scanned")): # Zero counters say "nothing went wrong"; they do not say a sweep happened. A record whose # corpus fields are absent entirely has a share vector and no population behind it, and # reading its zeroed counters as completeness would trust a measurement nobody took. @@ -191,6 +197,28 @@ def sweep_quality(live): return worst_quality([claimed, derived]) +# Reasons load_history() counts that mean a record EXISTED and could not be read as one. A +# refusal by design (current-generation evidence this optimizer does not read) and a deduplicated +# retry are not damage: the first is a contract, the second is bookkeeping. +CONTAINER_DAMAGE = ("unparseable line", "record above the size cap", "history unreadable", + "not an object", "no shares", "non-numeric share", "share out of range", + "implausible", "unknown evidence_quality", "unknown record_type") + + +def container_quality(rejected): + """What the history FILE itself can attest, separately from the records inside it. + + A torn line in the history is a record that was written and cannot be read — a loss, not a + bound, so no flag adopts it. It used to appear only as a line in `history.rejected` while the + trend built from the surviving records was promoted as if nothing had been lost. + (cross-family review, round 1) + """ + for reason, count in (rejected or {}).items(): + if count and any(str(reason).startswith(d) for d in CONTAINER_DAMAGE): + return "DEGRADED" + return "COMPLETE" + + def history_quality(records): """The worst quality among the records ELIGIBLE to support a finding. @@ -449,7 +477,9 @@ def load_ledger(path): try: fh = open(path, encoding="utf-8", errors="replace") except OSError: - return None + # A ledger that EXISTS and cannot be read is not the same as no ledger at all: the guard's + # evidence is missing rather than absent by design, and `rejected` says so. + return {"writes": 0, "prevented": 0, "rejected": 1, "rate": 0.0, "unreadable": True} with fh: for line in fh: line = line.strip() @@ -511,7 +541,8 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): # partial history it never read, and a finding drawing on history is not waved through because # today's sweep happened to be clean. quality = sweep_quality(live) if live else "COMPLETE" - hist_q = history_quality(hist.get("comparable", [])) + hist_q = worst_quality([history_quality(hist.get("comparable", [])), + container_quality(hist.get("rejected"))]) usable = may_promote(quality, accept_partial) # findings that rest on the live sweep hist_usable = may_promote(hist_q, accept_partial) # findings that rest on the history run_ids = [r.get("run_id") for r in hist.get("comparable", []) if r.get("run_id")] @@ -602,7 +633,8 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): # Its evidence is the hook's OWN ledger — writes it saw, no-ops it prevented — not a sample # of transcripts. No sweep or history quality can make it better or worse, so the eligibility # decision here is explicit and separate: the ledger's own sample size is its gate. - ledger_usable = bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES + ledger_usable = (bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES + and not ledger.get("rejected")) if ledger and ledger["writes"]: if ledger["writes"] >= LEDGER_MIN_WRITES: if ledger["rate"] >= LEDGER_MIN_NOOP_RATE: @@ -676,10 +708,13 @@ def overall_status(findings, hist, live, pop, accept_partial): if live: refused.append(sweep_quality(live)) comp = hist.get("comparable", []) + damage = container_quality(hist.get("rejected")) if comp: - refused.append(history_quality(comp)) + refused.append(worst_quality([history_quality(comp), damage])) elif any(record_quality(r) in ("INVALID", "EMPTY") for r, _why in hist.get("dropped", [])): refused.append("INVALID") + elif damage != "COMPLETE": + refused.append(damage) if any(not may_promote(q, accept_partial) for q in refused): return "PARTIAL_EVIDENCE" if not live and len(comp) < MIN_HISTORY_FOR_TREND: @@ -985,6 +1020,12 @@ def main(argv=None): status = "ALREADY_RUNNING" else: written, existing, failed = emit_candidates(findings, a.emit_candidate, status) + if failed and not written: + # The run promised a candidate and nothing landed — an unwritable directory, a + # path that is a regular file, a full disk. Reporting CANDIDATE and exit 10 + # tells a scheduler a specification exists; the status has to say otherwise. + # (cross-family review, round 1) + status = "INTERNAL_ERROR" if a.json: print(json.dumps({ @@ -1001,7 +1042,8 @@ def main(argv=None): # here even though the 1.3 label for it is PARTIAL (output_schema_version 2). "evidence_quality": sweep_quality(live) if live else "NO_SCAN", "history": {"records": len(recs), "comparable": len(keep), - "quality": history_quality(keep), + "quality": worst_quality([history_quality(keep), + container_quality(rejected)]), "rejected": dict(rejected), "time_order": hist["time_order"]}, "ledger": ledger, "live": ({"sessions": live["sessions"], "turns": live["turns"], "scanned": live["scanned"], From 588de91bd6e70bd63e5dad0639046a589138001d Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 22:08:16 +0000 Subject: [PATCH 07/26] fix: narrow container damage to lines that were lost, not lines that were foreign MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first cut degraded the history for any rejected line, including a well-formed JSON line that simply is not a carry record. In a shared file one stray append would then block a real population forever — a fail-closed patch that refuses evidence it should still read is also a defect. Damage is now what it says: a torn line, a line past the size cap, a file that could not be opened. A foreign line and a refusal by design are counted and reported, and do not degrade; both have controls in the suite. The limit is written down rather than implied: a corrupted carry record that still parses as JSON is reported and does not degrade. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 4 ++++ tools/optimize.py | 17 +++++++++++------ 3 files changed, 16 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 8ad6fb2..c8f97cd 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1360 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1364 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 3cdb8e0..191a8b1 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -313,6 +313,10 @@ def main(): "PARTIAL_EVIDENCE", 40, 0, extra=["--accept-partial"]) envelope_line = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, "payload": {}, "certificate": {}}) + foreign = json.dumps({"note": "another tool's line in a shared file"}) + case("R142_11 control: a foreign line that parses is counted, not called damage", + [json.dumps(r) for r in good_rows] + [foreign], "CANDIDATE", 10, 1, + history_quality="COMPLETE") case("R142_11 control: a refusal by design is not damage", [json.dumps(r) for r in good_rows] + [envelope_line], "CANDIDATE", 10, 1, history_quality="COMPLETE") diff --git a/tools/optimize.py b/tools/optimize.py index 5cda933..bd2bea0 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -197,12 +197,17 @@ def sweep_quality(live): return worst_quality([claimed, derived]) -# Reasons load_history() counts that mean a record EXISTED and could not be read as one. A -# refusal by design (current-generation evidence this optimizer does not read) and a deduplicated -# retry are not damage: the first is a contract, the second is bookkeeping. -CONTAINER_DAMAGE = ("unparseable line", "record above the size cap", "history unreadable", - "not an object", "no shares", "non-numeric share", "share out of range", - "implausible", "unknown evidence_quality", "unknown record_type") +# Reasons load_history() counts that mean a record EXISTED and could not be read as one: a torn +# line, a line past the size cap, a file that could not be opened. Deliberately NOT here: +# * a refusal by design — current-generation evidence this optimizer does not read — which is a +# contract, not a loss, and would otherwise block every history in the middle of a migration; +# * a deduplicated retry, which is bookkeeping; +# * a well-formed line that is not a carry record at all ("no shares", "not an object"). That +# line may never have been one — another tool's entry in a shared file — and treating a +# foreign line as lost evidence would let one stray append block a real population forever. +# The limit is deliberate: a CORRUPTED carry record that still parses as JSON is counted and +# reported, and does not degrade. Say so rather than claim a coverage this does not have. +CONTAINER_DAMAGE = ("unparseable line", "record above the size cap", "history unreadable") def container_quality(rejected): From 486004e3fd00dda07d8d03d08dcd099e17ffd98c Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 22:09:37 +0000 Subject: [PATCH 08/26] test: pin what the scope fallback can and cannot do MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A confirmation-round reviewer read the `eligible_anchor(recs) or newest` fallback in main() as a way back in for an ineligible record. Their exact input reproduces the observation — the run names the scope of the one INVALID record it found — and it stops there: status PARTIAL_EVIDENCE, exit 40, nothing on disk, because the fallback is only reached when NOTHING in the file is eligible, and then there is no population to strand and nothing to promote. Both halves are now tests rather than an argument: the reviewer's input, and the same record added to a real population in another scope, where the eligible records still choose the scope and still promote. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 12 ++++++++++++ tools/optimize.py | 7 +++++-- 3 files changed, 18 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index c8f97cd..042445c 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1364 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1374 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 191a8b1..6953f8c 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -355,6 +355,18 @@ def main(): check("and the exit code follows the status", proc.returncode, 50) check("the failure is named", len(jf["candidates_failed"]), 1) + print("\nR142_14 - the scope fallback is a label on an empty run, not a way back in") + only_bad = [{"schema_version": 2, "record_type": "carry_run", "ts": TS0, "sessions": 0, + "scanned": 1, "scope_id": "attack", "evidence_quality": "INVALID", + "shares": {"Bash": 100.0}}] + j = case("R142_14 (nothing eligible anywhere)", only_bad, "PARTIAL_EVIDENCE", 40, 0, + comparable=0) + check("the scope it names is the one real record's scope", j["scope"]["analysed"], "attack") + prod = [dict(r, scope_id="prod", run_id="prod%d" % i) for i, r in enumerate(good_rows)] + j2 = case("R142_14 (an eligible population is never stranded by it)", prod + only_bad, + "CANDIDATE", 10, 1, comparable=6) + check("and the eligible population chooses the scope", j2["scope"]["analysed"], "prod") + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") good = good_rows diff --git a/tools/optimize.py b/tools/optimize.py index bd2bea0..d309095 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -976,8 +976,11 @@ def main(argv=None): elif recs: # Which scope gets analysed is an anchor too: taking the newest record of ANY quality let a # single INVALID sweep in another agent's scope send the whole run to a population that was - # never going to be analysable. The newest record that can speak chooses; if none can, the - # newest record still names a scope, and the status below says why nothing came of it. + # never going to be analysable. The newest record that CAN speak chooses. + # The fallback names a scope only when NOTHING in the file is eligible — and then there is + # no population to strand and nothing that can be promoted: the status is PARTIAL_EVIDENCE + # and the emitter is closed. Both halves are pinned by tests (a reviewer read the fallback + # as a way back in; it is a label on an empty run, verified by execution). anchor = eligible_anchor(recs) or max(recs, key=lambda r: r.get("ts") or 0) scope = scope_of(anchor) scoped = scopes[scope] From 42a98f0d26a1ccfc010baecb7144bc4a7af9d0f5 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 22:11:00 +0000 Subject: [PATCH 09/26] style: keep the counter vocabulary and the dedup floor in their own blocks Co-Authored-By: Claude Opus 5 --- tools/optimize.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tools/optimize.py b/tools/optimize.py index d309095..28231ad 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -77,12 +77,12 @@ # The counters every schema-2 record's writer wrote, split by what they mean. Nothing here is # invented for an older schema: a record that never carried these cannot be asked to show them, # and is UNKNOWN rather than trusted. -# Set by load_history() when a duplicate run_id shows a worse sweep than the record that survives. -# It is a private, in-memory annotation; nothing writes it back to a file. -QUALITY_FLOOR = "_evidence_quality_floor" RECORD_LOSS_COUNTERS = ("unreadable", "oversize") RECORD_BOUND_COUNTERS = ("skipped_by_limit",) RECORD_COUNTERS = RECORD_LOSS_COUNTERS + RECORD_BOUND_COUNTERS +# Set by load_history() when a duplicate run_id shows a worse sweep than the record that survives. +# A private, in-memory annotation; nothing writes it back to a file. +QUALITY_FLOOR = "_evidence_quality_floor" def worst_quality(qualities): From ef0c749c2c58a285afbc568bdec4c51727e2aa81 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Mon, 21 Sep 2026 22:15:49 +0000 Subject: [PATCH 10/26] fix: three HIGHs and a MEDIUM from the confirmation round MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The confirmation lane attacked the repair rather than the original defect and found four more places where a word was taken for evidence. - A record could say PARTIAL while every counter it carried said nothing happened, and --accept-partial adopted it. A bound shows up in skipped_by_limit and a loss shows up in a loss counter; a PARTIAL claim that no counter can explain cannot say WHY it was partial, so it is UNKNOWN and no flag adopts it. - The record-level loss vocabulary was two counters wide (unreadable, oversize) while the contract promises more. A record carrying malformed, malformed_lines, identity_changed, conflicted_sources or records_rejected read COMPLETE — and a retry carrying one could launder itself through dedup, because the floor was computed with the same short list. REQUIRED (what the schema-2 writer always wrote, so its absence means UNKNOWN) and LOSS (what must be honoured when present) are now separate lists. - Container damage was an allowlist of three reasons, so a rejection valid_record() learns to make later would default to "not damage". It is now the other way round: a rejected line is damage unless it is one of three named exceptions — a refusal by design, a deduplicated retry, or a well-formed line that was never a carry record. An unsupported schema-3 record and a record whose shares do not sum to a population now close the emitter instead of being footnotes. - A ledger line that is neither `checked` nor `denied` was counted as nothing at all, so a ledger full of unknown events looked like a clean sample. It counts as rejected, which the guard's own gate reads. Four more mutants (45/45). The schema-4 contract is still untouched: the encoded v1.4 record for a clean, a torn and a bounded sweep has the same sha256 as on main. README: 1399 assertions. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 35 +++++++++++++++++- tests/test_mutation.py | 63 +++++++++++++++++++++++++++++++- tools/optimize.py | 55 +++++++++++++++++++--------- 4 files changed, 135 insertions(+), 20 deletions(-) diff --git a/README.md b/README.md index 042445c..0ba588e 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1374 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1399 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 6953f8c..adad575 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -143,7 +143,16 @@ def main(): check("the claim cannot be better than the counters", optimize.record_quality(rec(0, 40.0, quality="COMPLETE", oversize=1)), "DEGRADED") check("the counters cannot be better than the claim", - optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), "PARTIAL") + optimize.record_quality(rec(0, 40.0, quality="PARTIAL", skipped=5)), "PARTIAL") + check("...but a PARTIAL claim no counter can explain is not a bound to adopt", + optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), "UNKNOWN") + check("and the flag does not adopt it either", + optimize.may_promote(optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), True), + False) + for counter in ("malformed", "malformed_lines", "identity_changed", "conflicted_sources", + "records_rejected"): + check(f"a record carrying {counter} is not COMPLETE", + optimize.record_quality(dict(rec(0, 40.0), **{counter: 1})), "DEGRADED") check("looked and found nothing usable: INVALID", optimize.record_quality(rec(0, 40.0, sessions=0, scanned=100)), "INVALID") check("had nothing to look at: EMPTY", @@ -367,6 +376,30 @@ def main(): "CANDIDATE", 10, 1, comparable=6) check("and the eligible population chooses the scope", j2["scope"]["analysed"], "prod") + print("\nR142_15 - a rejected line is damage unless it is one of the named exceptions") + for line, label, status, quality in ( + ('{"schema_version":3,"record_type":"carry_run","shares":{"Bash":60.0,"Read":40.0}}', + "a record from a schema this reader does not know", "PARTIAL_EVIDENCE", "DEGRADED"), + ('{"schema_version":2,"record_type":"carry_run","shares":{"Bash":10.0}}', + "a carry record whose shares do not sum to a population", "PARTIAL_EVIDENCE", + "DEGRADED"), + ('{"note": "another tool\'s line in a shared file"}', + "a line that was never a carry record", "CANDIDATE", "COMPLETE")): + case(f"R142_15: {label}", [json.dumps(r) for r in good_rows] + [line], + status, 40 if status == "PARTIAL_EVIDENCE" else 10, + 0 if status == "PARTIAL_EVIDENCE" else 1, history_quality=quality) + + print("\nR142_16 - a ledger line that is neither a write nor a denial is a line lost") + led_dir = tempfile.mkdtemp(dir=d) + unknown_event = os.path.join(led_dir, "unknown.jsonl") + with open(unknown_event, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n") + fh.write('{"event": "garbage"}\n') + led = optimize.load_ledger(unknown_event) + check("the unaccountable line is counted", (led["writes"], led["rejected"]), (100, 1)) + check("and the guard only observes", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), led, None)], ["OBSERVED"]) + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") good = good_rows diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 7286674..67df227 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -704,7 +704,8 @@ def hist_of(n): """), ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history menurunkan kualitas evidence", - [("optimize.py", ' if count and any(str(reason).startswith(d) for d in CONTAINER_DAMAGE):', + [("optimize.py", + ' if count and not any(x in str(reason) for x in NOT_CONTAINER_DAMAGE):', " if False:")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="c") @@ -765,6 +766,66 @@ def hist_of(n): assert r.returncode == 50, r.returncode """), + + ("M_BARE_PARTIAL_CLAIM: klaim PARTIAL tanpa counter yang menjelaskannya bukan bound", + [("optimize.py", """ if claimed == "PARTIAL" and derived == "COMPLETE":""", " if False:")], + """ + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r["evidence_quality"] = "PARTIAL" + q = optimize.record_quality(r) + assert q == "UNKNOWN", q + assert optimize.may_promote(q, True) is False, q + """), + + ("M_RECORD_LOSS_IGNORED: counter kehilangan pada record ikut menentukan kualitas", + [("optimize.py", + 'RECORD_LOSS_COUNTERS = ("unreadable", "oversize", "malformed", "malformed_lines",\n "identity_changed", "conflicted_sources", "records_rejected")', + 'RECORD_LOSS_COUNTERS = ("unreadable", "oversize")')], + """ + for counter in ("malformed", "identity_changed", "conflicted_sources"): + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r[counter] = 1 + q = optimize.record_quality(r) + assert q == "DEGRADED", (counter, q) + """), + + ("M_STRUCTURAL_REJECT_NOT_DAMAGE: penolakan struktural menutup emitter, bukan sekadar dicatat", + [("optimize.py", + 'NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares")', + 'NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares", "unsupported schema_version", "shares sum to")')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="g") + for i in range(6)] + bad = chr(123) + '"schema_version":3,"record_type":"carry_run","shares":' \ + + chr(123) + '"Bash":60.0,"Read":40.0' + chr(125) + chr(125) + p = w(os.path.join(D, "struct.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _ = optimize.load_history(p) + keep, dropped = optimize.comparable(recs) + h = {"comparable": keep, "total": len(recs), "in_scope": len(recs), "rejected": rej, + "dropped": dropped, "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="g") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + """), + + ("M_LEDGER_UNKNOWN_EVENT: baris ledger yang tak terhitung adalah baris yang hilang", + [("optimize.py", """ else: + # Neither a write nor a prevented write: a line this reader cannot account for. + # Counting it as nothing at all let a ledger full of unknown events look like a + # clean sample. (cross-family review, confirmation round) + rejected += 1""", " else:\n pass")], + """ + p = os.path.join(D, "unknown_led.jsonl") + with open(p, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + fh.write('{"event": "garbage"}' + chr(10)) + led = optimize.load_ledger(p) + assert led["rejected"] == 1, led + EMPTY = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + f = optimize.analyse(None, EMPTY, led, None) + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + """), + ] diff --git a/tools/optimize.py b/tools/optimize.py index 28231ad..5a88a5d 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -74,12 +74,19 @@ # of the evidence was never swept, and nine neighbours cannot launder it. QUALITY_RANK = {"COMPLETE": 0, "PARTIAL": 1, "DEGRADED": 2, "EMPTY": 3, "UNKNOWN": 4, "INVALID": 5} SCHEMA_WITH_QUALITY = 2 # the first history schema that records how a sweep was taken -# The counters every schema-2 record's writer wrote, split by what they mean. Nothing here is -# invented for an older schema: a record that never carried these cannot be asked to show them, -# and is UNKNOWN rather than trusted. -RECORD_LOSS_COUNTERS = ("unreadable", "oversize") +# Two different questions, and conflating them cost a HIGH in review: +# REQUIRED — what the schema-2 writer ALWAYS wrote, so a record that lacks them cannot attest +# completeness. Nothing is invented for an older schema; a record that never carried +# these is UNKNOWN rather than trusted. +# LOSS — every counter that, WHEN PRESENT, means evidence was selected and then lost. A +# reader must honour a loss it can see even if that counter came from a later writer +# (`malformed_lines` is the 1.3.1-era name, `malformed` the current one). +RECORD_REQUIRED_COUNTERS = ("unreadable", "oversize", "skipped_by_limit") +RECORD_LOSS_COUNTERS = ("unreadable", "oversize", "malformed", "malformed_lines", + "identity_changed", "conflicted_sources", "records_rejected") RECORD_BOUND_COUNTERS = ("skipped_by_limit",) -RECORD_COUNTERS = RECORD_LOSS_COUNTERS + RECORD_BOUND_COUNTERS +RECORD_COUNTERS = tuple(dict.fromkeys(RECORD_REQUIRED_COUNTERS + RECORD_LOSS_COUNTERS + + RECORD_BOUND_COUNTERS)) # Set by load_history() when a duplicate run_id shows a worse sweep than the record that survives. # A private, in-memory annotation; nothing writes it back to a file. QUALITY_FLOOR = "_evidence_quality_floor" @@ -139,7 +146,7 @@ def _derived_record_quality(rec, schema): # reading its zeroed counters as completeness would trust a measurement nobody took. # (cross-family review, round 1) return "UNKNOWN" - if schema >= SCHEMA_WITH_QUALITY and not all(k in nums for k in RECORD_COUNTERS): + if schema >= SCHEMA_WITH_QUALITY and not all(k in nums for k in RECORD_REQUIRED_COUNTERS): return "UNKNOWN" # claims a completeness it cannot show return "COMPLETE" @@ -167,7 +174,14 @@ def record_quality(rec): claimed = "UNKNOWN" if schema < SCHEMA_WITH_QUALITY: claimed = "UNKNOWN" - qualities = [claimed, _derived_record_quality(rec, schema)] + derived = _derived_record_quality(rec, schema) + if claimed == "PARTIAL" and derived == "COMPLETE": + # A bound the caller asked for shows up in `skipped_by_limit`; a loss shows up in a loss + # counter. A record that says PARTIAL while every counter it carries says nothing happened + # cannot say WHY it was partial — and `--accept-partial` adopts a bound, not a word. + # (cross-family review, confirmation round) + claimed = "UNKNOWN" + qualities = [claimed, derived] floor = rec.get(QUALITY_FLOOR) if floor: # absent is not UNKNOWN: most records carry no floor at all qualities.append(floor) @@ -197,17 +211,19 @@ def sweep_quality(live): return worst_quality([claimed, derived]) -# Reasons load_history() counts that mean a record EXISTED and could not be read as one: a torn -# line, a line past the size cap, a file that could not be opened. Deliberately NOT here: +# A rejected line is DAMAGE by default: the history held something that could not be read as a +# record, and a trend built from the survivors is built over a gap. The exceptions are named, and +# they are the only two things a rejection can be that are not a loss, plus the one shape that was +# never our record to begin with: # * a refusal by design — current-generation evidence this optimizer does not read — which is a # contract, not a loss, and would otherwise block every history in the middle of a migration; -# * a deduplicated retry, which is bookkeeping; -# * a well-formed line that is not a carry record at all ("no shares", "not an object"). That -# line may never have been one — another tool's entry in a shared file — and treating a -# foreign line as lost evidence would let one stray append block a real population forever. -# The limit is deliberate: a CORRUPTED carry record that still parses as JSON is counted and -# reported, and does not degrade. Say so rather than claim a coverage this does not have. -CONTAINER_DAMAGE = ("unparseable line", "record above the size cap", "history unreadable") +# * a deduplicated retry, which is bookkeeping (its quality already travels via QUALITY_FLOOR); +# * a well-formed JSON line that is not a carry record at all. In a shared file that is another +# tool's entry, and treating it as lost evidence would let one stray append block a real +# population forever. +# Listed this way round on purpose: a reason added to valid_record() later defaults to DAMAGE +# rather than slipping through an allowlist nobody updated. (cross-family review, confirmation round) +NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares") def container_quality(rejected): @@ -219,7 +235,7 @@ def container_quality(rejected): (cross-family review, round 1) """ for reason, count in (rejected or {}).items(): - if count and any(str(reason).startswith(d) for d in CONTAINER_DAMAGE): + if count and not any(x in str(reason) for x in NOT_CONTAINER_DAMAGE): return "DEGRADED" return "COMPLETE" @@ -503,6 +519,11 @@ def load_ledger(path): checked += 1 elif ev == "denied": denied += 1 + else: + # Neither a write nor a prevented write: a line this reader cannot account for. + # Counting it as nothing at all let a ledger full of unknown events look like a + # clean sample. (cross-family review, confirmation round) + rejected += 1 total = checked + denied return {"writes": total, "prevented": denied, "rejected": rejected, "rate": (100.0 * denied / total) if total else 0.0} From 5cfcf7ee0d7f27ba0769f59f8bc6c56c888db8b9 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 06:02:35 +0000 Subject: [PATCH 11/26] docs: the confirmation round's cases, in the same post-freeze section Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index a9ffd2a..a08fb30 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -79,3 +79,12 @@ original defect. They are listed separately so the frozen matrix stays readable | **R142_07b** | `schema_version` of `"2"` (string) or `2.0` (float) | `UNKNOWN` — a version this reader cannot name is not a newer generation to trust | | **R142_08** | schema-2 record with zeroed counters and no `sessions` / `turns` / `carry_bytes` at all | `UNKNOWN` — zero counters say nothing went wrong, not that a sweep happened (cross-family review, round 1) | | **R142_09** | `--max-files 2` where one of three sources cannot be dated | the datable sources still order by mtime; one unreadable mtime no longer sends the whole selection back to a discovery-order slice (cross-family review, round 1) | + +Confirmation round, attacking the repair again: + +| case | fixture | expectation | +|---|---|---| +| **R142_15** | a `PARTIAL` claim with every counter at zero | `UNKNOWN` — a bound shows up in `skipped_by_limit` and a loss in a loss counter; a claim no counter can explain is not a bound `--accept-partial` may adopt | +| **R142_15b** | a record carrying `malformed`, `malformed_lines`, `identity_changed`, `conflicted_sources` or `records_rejected` | `DEGRADED` — a loss must be honoured wherever the reader can see it, not only in the two counters the first cut looked at | +| **R142_16** | history + one record from an unsupported schema 3, and one whose shares sum to 10 | `PARTIAL_EVIDENCE` / exit 40 / 0 files — a rejected line is damage unless it is a refusal by design, a deduplicated retry, or a line that was never a carry record (control: a foreign line still promotes) | +| **R142_17** | ledger of 100 `checked` lines plus one `{"event": "garbage"}` | the unaccountable line counts as rejected and the guard only observes | From fe78cbc4ddac11eb60be2c0e6a394f380ca0710d Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 06:06:37 +0000 Subject: [PATCH 12/26] =?UTF-8?q?fix:=20round-4=20findings=20=E2=80=94=20o?= =?UTF-8?q?ne=20selection=20per=20run,=20one=20impossibility=20rule,=20no?= =?UTF-8?q?=20half=20files?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A fourth adversarial pass returned no HIGH that survived checking (the one HIGH claimed a missing isinstance guard in load_ledger that is present, and a list-valued ledger line is counted as rejected, verified by execution). Four MEDIUMs were real: - `sweep_quality()` did not apply the impossibility rule its record-level twin applies: a live sweep reporting sessions with zero turns read COMPLETE. It is INVALID, like the record. - `bounded_paths()` was called twice per run — once by the sweep, once for the listing — so an mtime that became unreadable between the two calls produced two different samples again. The selection is made ONCE in main() and handed to both; `accumulate(selected=...)` takes it, and still reports the bound against the whole discovered population. - A line that CLAIMS to be one of our records and carries no shares is a corrupted record, not another tool's entry. It now reads as container damage; a line that claims nothing still does not. - A candidate write that failed mid-way left its `.tmp-` file behind. It is removed on the failure path. Two limits stay, stated rather than papered over: container damage is a property of the FILE, so a torn line degrades every scope in it (an unparseable line has no scope to attribute it to), and HOST_BEHAVIOR_SHIFT still closes the emitter for findings that do not rest on the shifted population — that ordering is the release contract's, not this patch's. 45/45 mutants. README: 1408 assertions. The schema-4 encoded record for a clean, a torn and a bounded sweep still has the same sha256 as main. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 38 ++++++++++++++++++++++++++++ tests/test_mutation.py | 5 ++-- tools/carry.py | 7 ++++-- tools/optimize.py | 43 ++++++++++++++++++++++++-------- 5 files changed, 79 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 0ba588e..dcf3676 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1399 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1408 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index adad575..dd52868 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -400,6 +400,44 @@ def main(): check("and the guard only observes", [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), led, None)], ["OBSERVED"]) + print("\nR142_18 - round-4 findings: the live sweep, the shares reason, the temp file") + live_impossible = {"sessions": 40, "turns": 0, "scanned": 40, "unreadable": 0, "oversize": 0, + "skipped_by_limit": 0, "malformed": 0, "identity_changed": 0, + "conflicted_sources": 0, "quality": "COMPLETE"} + check("a live sweep with sessions and no turns is impossible too", + optimize.sweep_quality(live_impossible), "INVALID") + check("control: the same sweep with turns is COMPLETE", + optimize.sweep_quality(dict(live_impossible, turns=2000)), "COMPLETE") + claims_ours = json.dumps({"schema_version": 2, "record_type": "carry_run", "ts": TS0, + "sessions": 40, "turns": 1000, "carry_bytes": 10 ** 7}) + case("R142_18: a carry record with no shares is damage", + [json.dumps(r) for r in good_rows] + [claims_ours], + "PARTIAL_EVIDENCE", 40, 0, history_quality="DEGRADED") + tmp_dir = tempfile.mkdtemp(dir=d) + blocked_id = [f["candidate_id"] for f in optimize.analyse( + None, {"comparable": good_rows, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"}, None, None) if f["state"] == "CANDIDATE"][0] + open(os.path.join(tmp_dir, blocked_id), "w").write("a file where a directory must go") + hist2 = write(os.path.join(tmp_dir, "h.jsonl"), good_rows) + subprocess.run([sys.executable, OPT, "--history", hist2, "--ledger", + os.path.join(tmp_dir, "none.jsonl"), "--scan", "--emit-candidate", tmp_dir, + "--json", "--strict-exit"], capture_output=True, text=True, timeout=300) + check("a failed write leaves no half-written file behind", + [f for _b, _dd, fs in os.walk(tmp_dir) for f in fs if ".tmp-" in f], []) + bound_dir = tempfile.mkdtemp(dir=d) + trio = [] + for n, name in enumerate(("p.jsonl", "q.jsonl", "r.jsonl")): + q = transcript(os.path.join(bound_dir, name), turns=30) + os.utime(q, (TS0 + n * 1000, TS0 + n * 1000)) + trio.append(q) + once = carry.bounded_paths(trio, 2) + swept = carry.accumulate(trio, min_turns=1, max_files=2, selected=once) + check("a caller can hand the sweep the selection it already made", + sorted(os.path.basename(x) for x in swept["parsed"]), + sorted(os.path.basename(x) for x in once)) + check("and the bound is still reported against the whole population", + swept["skipped_by_limit"], 1) + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") good = good_rows diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 67df227..1c68c00 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -614,9 +614,8 @@ def hist_of(n): """), ("M_BOUND_SAMPLE_DIVERGES: sapuan dan pindaian listing memakai sampel terbatas yang SAMA", - [("optimize.py", - "listing, uses, sess = skills_tool.scan(carry.bounded_paths(paths, a.max_files))", - "listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths)")], + [("optimize.py", " listing, uses, sess = skills_tool.scan(selected)", + " listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths)")], """ import subprocess d = tempfile.mkdtemp() diff --git a/tools/carry.py b/tools/carry.py index 22088f1..d527da1 100755 --- a/tools/carry.py +++ b/tools/carry.py @@ -282,7 +282,7 @@ def sweep_label(facts): return "COMPLETE" -def accumulate(paths, min_turns=50, max_files=0): +def accumulate(paths, min_turns=50, max_files=0, selected=None): carry, size, usage = collections.Counter(), collections.Counter(), collections.Counter() runtimes, models = collections.Counter(), collections.Counter() turns = sessions = 0 @@ -303,7 +303,10 @@ def accumulate(paths, min_turns=50, max_files=0): usage_conflicted_transcripts = identity_conflicted_transcripts = 0 usage_exact_measurement_excluded = identity_exact_measurement_excluded = 0 conflicted_sources = records_rejected = 0 - paths = bounded_paths(paths, max_files) + # `selected` lets a caller that must hand the SAME sample to another consumer choose once and + # pass it in; without it the sweep selects for itself. `total_paths` above is still what + # discovery found, so the bound is reported against the whole population either way. + paths = list(selected) if selected is not None else bounded_paths(paths, max_files) for p in paths: scanned += 1 try: diff --git a/tools/optimize.py b/tools/optimize.py index 5a88a5d..af3ce91 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -203,6 +203,12 @@ def sweep_quality(live): claimed = "COMPLETE" if not live.get("sessions"): return "INVALID" if live.get("scanned") else "EMPTY" + if not live.get("turns") or not live.get("carry_bytes", 1): + # The same impossibility the record law refuses: sessions were counted, so turns were + # counted. A live dict that says otherwise is not a quiet sweep. `carry_bytes` is absent + # from the sweep dict itself (the caller sums it), so its absence is not the claim. + # (cross-family review, round 4) + return "INVALID" derived = "COMPLETE" if any(live.get(k) for k in carry.LOSS_FIELDS): derived = "DEGRADED" @@ -318,7 +324,12 @@ def valid_record(o): return False, f"unsupported schema_version {sv!r}" sh = o.get("shares") if not isinstance(sh, dict) or not sh: - return False, "no shares" + # Two different lines, and the container quality depends on which one this is: a line that + # claims to be one of OUR records is a corrupted record (damage), a line that claims + # nothing is another tool's entry in a shared file (not ours to lose). + claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o + or "run_id" in o or "carry_bytes" in o) + return False, ("carry record without shares" if claims_ours else "no shares") for k, v in sh.items(): if not isinstance(k, str) or not isinstance(v, (int, float)) or isinstance(v, bool): return False, "non-numeric share" @@ -850,11 +861,20 @@ def emit_candidates(findings, outdir, status): try: os.makedirs(d, exist_ok=True) tmp = p + ".tmp-%d" % os.getpid() - with open(tmp, "w", encoding="utf-8") as fh: - fh.write(spec_text(f)) - fh.flush() - os.fsync(fh.fileno()) - os.replace(tmp, p) # a crash leaves the old file or the new one, never half + try: + with open(tmp, "w", encoding="utf-8") as fh: + fh.write(spec_text(f)) + fh.flush() + os.fsync(fh.fileno()) + os.replace(tmp, p) # a crash leaves the old file or the new one, never half + except OSError: + # A write that failed mid-way must not leave its half behind: the next reader of + # this directory would find a file nobody promised. (cross-family review, round 4) + try: + os.unlink(tmp) + except OSError: + pass + raise written.append(f["candidate_id"]) except OSError as e: failed.append((f["candidate_id"], safe_err(e))) @@ -1021,11 +1041,14 @@ def main(argv=None): except Exception: paths = [] if paths: - live = carry.accumulate(paths, min_turns=a.min_turns, max_files=a.max_files) + # Selected ONCE, then handed to both consumers. Calling bounded_paths() twice is two + # selections, and an mtime that becomes unreadable between them is two different samples + # again — the defect this function exists to close. (cross-family review, round 4) + selected = carry.bounded_paths(paths, a.max_files) + live = carry.accumulate(paths, min_turns=a.min_turns, max_files=a.max_files, + selected=selected) try: - # The SAME bounded sample the sweep used. Two selections of "the newest N" is two - # populations reported as one. - listing, uses, sess = skills_tool.scan(carry.bounded_paths(paths, a.max_files)) + listing, uses, sess = skills_tool.scan(selected) if listing: ent = skills_tool.parse_listing(listing) rows = skills_tool.tally(ent, uses, sess) From aadebfb196a7c547ca432e55486d3d0947933ba1 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:39:40 +0000 Subject: [PATCH 13/26] test: freeze the B1 epoch matrix, RED on this branch's current head MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An independent acceptance review BLOCKED the first head of this branch. The six original defects were closed, but the container-damage repair introduced a new one: damage was permanent. One torn line — the crash fragment docs/MULTI_AGENT.md calls an expected event, the one the reader is designed to "reject exactly that line and count it" — set the whole file DEGRADED forever, for every scope sharing it, with no flag able to adopt a loss and nothing in the product that expires or rotates a history. Measured on the current head, all of these stay PARTIAL_EVIDENCE with zero candidates: 6 good + torn; + 2 more; + 6 more; + 60 more; a sibling scope's records; a loss before any record; and a legitimate zero-carry record. This commit freezes what the repair must do, before it exists: thirteen rows in docs/V142_COUNTEREXAMPLES.md §5 with status, exit code, candidate files, comparable count, active-epoch size, history quality and boundary count, plus the two MEDIUMs the same review found (history.quality read COMPLETE with nothing comparable; a producer-valid zero-carry record read as corruption). The model being frozen: an unattributable loss cuts the promotion history at that line's PHYSICAL position. Evidence before the cut never joins evidence after it, the newest epoch is by construction damage-free, and the loss stays reported. No time window, no expiry, no ratio, no new override flag, and --accept-partial still adopts a chosen bound and never a loss. 21 assertions fail here, each naming its frozen expectation. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 73 ++++++++++++++++++ tests/test_evidence_integrity.py | 125 +++++++++++++++++++++++++++++++ 2 files changed, 198 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index a08fb30..c9fd5a8 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -88,3 +88,76 @@ Confirmation round, attacking the repair again: | **R142_15b** | a record carrying `malformed`, `malformed_lines`, `identity_changed`, `conflicted_sources` or `records_rejected` | `DEGRADED` — a loss must be honoured wherever the reader can see it, not only in the two counters the first cut looked at | | **R142_16** | history + one record from an unsupported schema 3, and one whose shares sum to 10 | `PARTIAL_EVIDENCE` / exit 40 / 0 files — a rejected line is damage unless it is a refusal by design, a deduplicated retry, or a line that was never a carry record (control: a foreign line still promotes) | | **R142_17** | ledger of 100 `checked` lines plus one `{"event": "garbage"}` | the unaccountable line counts as rejected and the guard only observes | + +## 5. B1 — the blocker an independent acceptance review found, and the epoch matrix + +The first head of this branch closed the six original defects (§1–§4) and introduced one of its own. +An **independent acceptance review of PR #13 BLOCKED it**: container damage was permanent. One torn +line — the crash fragment `docs/MULTI_AGENT.md` §Concurrency calls an expected event, the one the +reader is designed to "reject exactly that line and count it" — set the whole file to `DEGRADED` +forever. Nothing in the product expires, rotates or repairs a history (`docs/MULTI_AGENT.md`: "The +optimizer does not schedule, expire or rotate"), and `--accept-partial` cannot adopt `DEGRADED` by +design, so a single crash permanently disabled promotion for **every scope** sharing +`~/logs/carry_history.jsonl`. + +Measured on that head before the repair (every row: zero candidate files, exit 40, forever): + +```text +6 good + torn PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +6 good + torn + 2 good PARTIAL_EVIDENCE exit 40 comparable 8 hq DEGRADED +6 good + torn + 6 good PARTIAL_EVIDENCE exit 40 comparable 12 hq DEGRADED +6 good + torn + 60 good PARTIAL_EVIDENCE exit 40 comparable 66 hq DEGRADED +scope-a good + torn + scope-b good PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +torn first + 6 good PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +zero-carry record (shares {}) PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +6 good + torn + 60 good, --accept-partial exit 40 (no flag can adopt a loss) +``` + +### The replacement: a history EPOCH, cut at the physical position of the loss + +An unattributable loss cuts the promotion history **at that line's position in the file**. Evidence +before the cut and evidence after it are never combined for a promotion; the loss stays reported; +the newest epoch is, by construction, free of damage. No time window, no expiry, no ratio, no new +override flag — and `--accept-partial` still adopts only a caller's chosen bound, never a loss. + +| case | fixture (physical order) | status | exit | files | comparable | active epoch | history.quality | boundaries | +|---|---|---|---|---|---|---|---|---| +| **B1_01** | 6 good, torn | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | `EMPTY` | 1 | +| **B1_02** | 6 good, torn, 2 good | `INSUFFICIENT_DATA` | 20 | 0 | 2 | 2 | `COMPLETE` | 1 | +| **B1_03** | 6 good, torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | +| **B1_04** | 6 good, torn, 60 good | `CANDIDATE` | 10 | 1 | 60 | 60 | `COMPLETE` | 1 | +| **B1_05** | scope-a 6 good, torn, scope-b 6 good | `CANDIDATE` (scope b) · `INSUFFICIENT_DATA` with `--scope-id a` | 10 · 20 | 1 · 0 | 6 · 0 | 6 · 0 | `COMPLETE` · `EMPTY` | 1 | +| **B1_06** | 6 good, torn, 6 good, torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 2 | +| **B1_07** | torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | +| **B1_08** | 6 good, truncated final line | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | `EMPTY` | 1 | +| **B1_09** | 6 good, foreign JSON line | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 0 | +| **B1_10** | 6 good, schema-4 envelope line | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 0 | +| **B1_11** | 6 good, zero-carry record (`shares {}`, `carry_bytes 0`) | `CANDIDATE` | 10 | 1 | 6 | 7 | `COMPLETE` | 0 | +| **B1_12** | 6 records the law calls INVALID | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | 6 | `EMPTY` | 0 | +| **B1_13** | 5 good, `run_id=X` good, torn, `run_id=X` good ×6 | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | + +Rules the matrix encodes: + +* **Unattributable loss → file-global boundary.** An unparseable line, an oversized line or an + unreadable file cannot name a scope, so the cut applies to every scope. +* **Parseable rejection carrying a readable `scope_id` → scope-local boundary.** Only that scope's + continuity is cut; a sibling agent is not punished for a neighbour's corrupted record. Attribution + trusts the record's own `scope_id` (a string of 1–64 characters, no control bytes) and falls back + to file-global whenever it cannot be read. +* **Not damage, so not a boundary:** a well-formed foreign JSON line, current-generation (schema-4) + evidence refused by design, and a deduplicated retry. +* **The damage is still reported** — `history.rejected` counts it and `history.damage` says how many + boundaries the file holds — while `history.quality` describes only the evidence eligible for the + CURRENT analysis. A historical gap stays true while the post-gap evidence is independently + complete. +* **`history.quality` with nothing comparable is `EMPTY`, never `COMPLETE`** (the acceptance + review's MEDIUM M1). +* **A zero-carry sweep is readable evidence, not corruption** (MEDIUM M2). `carry.py` says it in its + own report — "A session whose every item lands on its final turn carries nothing" — the 1.3 writer + emits `shares: {}` for it and the current writer guards `if C else {}`. Measured on this branch: + `carry.accumulate()` on such a transcript returns `sessions=1 turns=6 carry_total=0`. The record + is `EMPTY` for the quality law: nothing to compare, and nothing wrong with the file. + +### 5.1 Deviations from this frozen matrix + +None. diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index dd52868..5daa9e2 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -438,6 +438,131 @@ def main(): check("and the bound is still reported against the whole population", swept["skipped_by_limit"], 1) + # ============================================================ B1: the history epoch + # An unattributable loss cuts the promotion history at that physical position. Evidence before + # the cut never joins evidence after it; the loss stays reported; the newest epoch is, by + # construction, free of damage. docs/V142_COUNTEREXAMPLES.md §5 froze every row below. + print("\nB1 - a damaged history recovers, and the loss is still reported") + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 5' + FOREIGN = json.dumps({"note": "another tool's line in a shared file"}) + ENVELOPE = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, "payload": {}, + "certificate": {}}) + + def series(n, first=0, scope="default", per_week=3.0, base=30.0): + return [json.dumps(rec(first + i, base + per_week * i, scope=scope)) for i in range(n)] + + def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, extra=()): + rc, j, n = run(rows, extra=extra) + got = (j["status"], rc, n, j["history"]["comparable"], + j["scope"].get("records_in_epoch"), j["history"].get("quality"), + (j["history"].get("damage") or {}).get("boundaries")) + check(label, got, (status, code, files, comparable, epoch, quality, boundaries)) + + b1("B1_01 loss at the tail leaves no epoch to promote from", + series(6) + [TORN], "INSUFFICIENT_DATA", 20, 0, 0, 0, "EMPTY", 1) + b1("B1_02 an epoch too small to carry a trend", + series(6) + [TORN] + series(2, 20), "INSUFFICIENT_DATA", 20, 0, 2, 2, "COMPLETE", 1) + b1("B1_03 a sufficient post-loss epoch promotes on its own", + series(6) + [TORN] + series(6, 20), "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_04 and it still promotes sixty records later", + series(6) + [TORN] + series(60, 20, per_week=1.0, base=20.0), + "CANDIDATE", 10, 1, 60, 60, "COMPLETE", 1) + b1("B1_05 an unattributable loss cuts every scope (analysing the recovered one)", + series(6, scope="agent-a") + [TORN] + series(6, 20, scope="agent-b"), + "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_05b ...and the scope with nothing after the loss says so", + series(6, scope="agent-a") + [TORN] + series(6, 20, scope="agent-b"), + "INSUFFICIENT_DATA", 20, 0, 0, 0, "EMPTY", 1, extra=["--scope-id", "agent-a"]) + b1("B1_06 two boundaries: only the newest epoch is analysed", + series(6) + [TORN] + series(6, 20) + [TORN] + series(6, 40), + "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 2) + b1("B1_07 a loss before any record does not stop the file", + [TORN] + series(6), "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_09 a foreign line is not a loss", + series(6) + [FOREIGN], "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 0) + b1("B1_10 current-generation evidence refused by design is not a loss", + series(6) + [ENVELOPE], "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 0) + b1("B1_12 records the law calls INVALID are refused, not a loss", + [json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)], + "PARTIAL_EVIDENCE", 40, 0, 0, 6, "EMPTY", 0) + + # B1_08: a truncated final line, written the way a crash leaves one + trunc_dir = tempfile.mkdtemp(dir=d) + trunc = os.path.join(trunc_dir, "history.jsonl") + with open(trunc, "w", encoding="utf-8") as fh: + fh.write("\n".join(series(6)) + "\n") + fh.write('{"schema_version": 2, "record_type": "carry_run", "ts": 17500000') + out_dir = os.path.join(trunc_dir, "cand") + proc = subprocess.run([sys.executable, OPT, "--history", trunc, "--ledger", + os.path.join(trunc_dir, "none.jsonl"), "--scan", "--emit-candidate", + out_dir, "--json", "--strict-exit"], capture_output=True, text=True, + timeout=300) + jt = json.loads(proc.stdout) + # the emit directory itself is created by the LOCK, before any status is known; what the + # matrix froze is the candidate-FILE count + spec_files = [f for _b, _dd, fs in os.walk(out_dir) for f in fs if f != ".optimize.lock"] + check("B1_08 a truncated final line is the same loss", + (jt["status"], proc.returncode, len(spec_files), + (jt["history"].get("damage") or {}).get("boundaries")), + ("INSUFFICIENT_DATA", 20, 0, 1)) + + print("\nB1_11 / M2 - a zero-carry sweep is readable evidence, not corruption") + zero_carry = dict(rec(9, 0.0), shares={}, bpt={}, carry_bytes=0) + ok, why = optimize.valid_record(zero_carry) + check("the reader accepts the shape the producer writes", (ok, why), (True, "")) + check("and the quality law calls it EMPTY, not INVALID", + optimize.record_quality(zero_carry), "EMPTY") + b1("B1_11 one zero-carry record does not stop a healthy population", + series(6) + [json.dumps(zero_carry)], "CANDIDATE", 10, 1, 6, 7, "COMPLETE", 0) + b1("B1_11b a history of nothing but zero-carry records is not damage", + [json.dumps(dict(rec(i, 0.0), shares={}, bpt={}, carry_bytes=0)) for i in range(6)], + "INSUFFICIENT_DATA", 20, 0, 0, 6, "EMPTY", 0) + check("carry with no shares is still a contradiction", + optimize.record_quality(dict(rec(9, 40.0), carry_bytes=0)), "INVALID") + check("shares with no carry is still a contradiction", + optimize.record_quality(dict(rec(9, 40.0), shares={}, bpt={})), "INVALID") + # the producer's own words, executed: a session whose items all land on its last turn + zt = os.path.join(d, "zero_carry.jsonl") + with open(zt, "w", encoding="utf-8") as fh: + for i in range(5): + fh.write(json.dumps({"type": "assistant", "message": { + "id": f"t{i}", "usage": {"output_tokens": 3}, "content": []}}) + "\n") + fh.write(json.dumps({"type": "assistant", "message": { + "id": "last", "usage": {"output_tokens": 3}, + "content": [{"type": "tool_use", "name": "Bash", "input": {"command": "ls"}}]}}) + "\n") + zsweep = carry.accumulate([zt], min_turns=1) + check("the producer really can measure zero carry", + (zsweep["sessions"] > 0, sum(zsweep["carry"].values())), (True, 0)) + + print("\nB1_13 - one run_id on both sides of a boundary") + dup_before = series(5) + [json.dumps(dict(json.loads(series(1, 5)[0]), run_id="carried"))] + dup_after = [json.dumps(dict(json.loads(r), run_id="carried" if i == 0 else None)) + for i, r in enumerate(series(6, 20))] + dup_after = [json.dumps({k: v for k, v in json.loads(r).items() if v is not None}) + for r in dup_after] + b1("B1_13 the post-loss epoch keeps its own copy", + dup_before + [TORN] + dup_after, "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + same_epoch = series(5) + [json.dumps(dict(json.loads(series(1, 5)[0]), run_id="twin")), + json.dumps(dict(json.loads(series(1, 6)[0]), run_id="twin", + evidence_quality="PARTIAL", skipped_by_limit=9))] + rc, j, n = run(same_epoch) + check("B1_13b a retry inside one epoch still lowers the survivor", + (j["history"]["quality"], j["status"], n), ("PARTIAL", "PARTIAL_EVIDENCE", 0)) + + print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") + rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) + check("nothing comparable is never reported as COMPLETE", + (j["history"]["comparable"], j["history"]["quality"]), (0, "EMPTY")) + + print("\nthe damage is reported even while the current epoch is clean") + rc, j, n = run(series(6) + [TORN] + series(6, 20)) + check("the rejection is still counted", j["history"]["rejected"].get("unparseable line"), 1) + check("and the file's historical damage is named", + ((j["history"].get("damage") or {}).get("boundaries"), + (j["history"].get("damage") or {}).get("file_global")), (1, 1)) + check("while the analysed epoch is clean and promotes", + (j["history"]["quality"], j["status"]), ("COMPLETE", "CANDIDATE")) + # ---------------------------------------------------------------- positive controls print("\npositive controls - a gate that refuses everything is not a gate") good = good_rows From a3067dc65c18de7e814a78e783ee8336041fd4e5 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:46:43 +0000 Subject: [PATCH 14/26] fix: a loss cuts the history at its position instead of poisoning the file MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit B1, the blocker an independent acceptance review found in this branch's first head. Container damage was permanent: one torn line — the crash fragment docs/MULTI_AGENT.md calls an expected event — set the whole file DEGRADED, and since nothing here expires, rotates or repairs a history and no flag adopts a loss, a single crash disabled promotion for every scope sharing ~/logs/carry_history.jsonl, forever. Measured before this commit: 6 good + torn + 60 good still returned PARTIAL_EVIDENCE with zero candidates. A loss is now a BOUNDARY at the rejected line's physical position in the file, not a verdict on the file: - `damage_boundary()` is the one classifier every rejection goes through, including the ones that never became an object. A line that parses and carries a readable `scope_id` cuts that scope; a line that cannot say whose record it was — unparseable, oversized, a file that would not open — cuts every scope, because guessing would be the fail-open half. A foreign line, a refusal by design and a deduplicated retry are not losses and cut nothing. - Records are stamped at READ time with the epoch they were written in: (file-global losses before them, losses attributed to their own scope before them). Physical order, never the clock, because the clock is what a damaged history cannot be trusted about. - `active_records()` is the analysed population: what came after the newest loss that applies to its scope. It is selected BEFORE anything is anchored, so a pre-loss record cannot choose the scope, the workload class, the corpus size or the trend for the population after it. - The active epoch is therefore damage-free by construction, which is why the container no longer gates: `analyse()` and `overall_status()` ask only about the evidence eligible right now. - The loss stays visible: `history.rejected` counts it and the new `history.damage` says how many boundaries the file holds, in the JSON and in the human report. Deduplication is now per epoch. The same run_id after a loss is that population's own observation; inside one epoch the retry is still dropped, counted, and still cannot launder the survivor's quality. Two MEDIUMs from the same review, repaired here because the epoch model makes both sharper: - `history.quality` is the quality of the evidence eligible for THIS analysis, and is EMPTY when nothing is comparable. It read COMPLETE with zero comparable records. `worst_quality([])` still means COMPLETE: a finding resting on no sampled evidence is not degraded by sampling it never used. - A zero-carry sweep is readable evidence, not corruption. tools/carry.py says so in its own report — "A session whose every item lands on its final turn carries nothing" — the 1.3 writer emits `shares: {}` for it and the current one guards `if C else {}`. Proved by running it: accumulate() on such a transcript returns sessions=1 turns=6 carry=0. The record reads EMPTY, is left out of comparable, and cuts nothing. The contradictions stay INVALID: carry with no shares, shares with no carry, sessions without turns, sessions beyond what was scanned. --accept-partial is unchanged and still adopts only a caller's chosen bound, never a loss. No time window, no expiry, no ratio, no override flag, no new status. Schema-4 stays unsupported and its encoded bytes are unchanged (clean, malformed and bounded sweeps all hash identical to main). 1460 assertions, 53 mutants. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- docs/V142_COUNTEREXAMPLES.md | 23 +++ experiments/aivos/longrun.py | 2 +- tests/test_evidence_integrity.py | 37 +++-- tests/test_evidence_phase2.py | 2 +- tests/test_multiagent.py | 10 +- tests/test_mutation.py | 168 ++++++++++++++++++--- tests/test_optimize.py | 19 ++- tools/optimize.py | 241 +++++++++++++++++++++++-------- 9 files changed, 397 insertions(+), 107 deletions(-) diff --git a/README.md b/README.md index dcf3676..a383238 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1408 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1460 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index c9fd5a8..fc8758b 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -161,3 +161,26 @@ Rules the matrix encodes: ### 5.1 Deviations from this frozen matrix None. + +### 5.2 Expectations the epoch model supersedes + +Three rows written for the first head's permanent-damage model are now wrong, and are replaced +rather than quietly re-run. In each, the damaged line sits at the END of the file, so there is no +post-loss epoch: the outcome is still **no promotion and no candidate file**, and what changes is +only the word for it and whether later evidence can ever clear it. + +```text +R142_11 torn line in the history PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +R142_15 unsupported schema, bad shares PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +R142_18 carry record with no shares KEY PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +``` + +Each keeps its protection and gains a companion case: the same rejected line placed BEFORE a healthy +population, where the population after the loss must stand on its own. The records before the loss +are never counted with the records after it — `R142_11` pins that as `comparable == 2` for +`6 good + torn + 2 good`. + +Dedup across a boundary is also new, and is stated rather than inherited: deduplication is +**per epoch**. The same `run_id` on the far side of a loss is that population's own observation and +is kept; inside one epoch the retry is still dropped, still counted, and still cannot launder the +survivor's quality (`tests/test_optimize.py` pins both). diff --git a/experiments/aivos/longrun.py b/experiments/aivos/longrun.py index c790b19..7eeed9d 100644 --- a/experiments/aivos/longrun.py +++ b/experiments/aivos/longrun.py @@ -158,7 +158,7 @@ def longrun(days, cycles, out, plateau=20): day_new = 0 for _cycle in range(cycles): - recs, _rej, _lines = optimize.load_history(hist) + recs, _rej, _lines, _ep = optimize.load_history(hist) scopes = optimize.by_scope(recs) for role in ROLES: keep, dropped = optimize.comparable(scopes.get(role, [])) diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 5daa9e2..096de43 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -262,7 +262,7 @@ def main(): worse["run_id"] = "dup" dup.append(worse) hist_path = write(os.path.join(d, "dup.jsonl"), dup) - recs, rejected, _lines = optimize.load_history(hist_path) + recs, rejected, _lines, _ep = optimize.load_history(hist_path) check("the retry is dropped as an observation", len(recs), 6) check("and counted", rejected["duplicate run_id (retry)"], 1) keep, _dropped = optimize.comparable(recs) @@ -315,11 +315,20 @@ def main(): "PARTIAL_EVIDENCE", 40, 0, comparable=0) print("\nR142_11 - a torn line in the history file is lost evidence, not a footnote") + # Expectation updated by the B1 repair (docs/V142_COUNTEREXAMPLES.md §5.1): the loss still + # refuses the records it followed — nothing is promoted and nothing is written — but it is a + # BOUNDARY, not a verdict on the file, so the status is "no epoch to analyse" rather than a + # permanent PARTIAL_EVIDENCE that no later evidence could ever clear. torn_hist = [json.dumps(r) for r in good_rows] + ['{"torn":'] - j = case("R142_11", torn_hist, "PARTIAL_EVIDENCE", 40, 0, history_quality="DEGRADED") + j = case("R142_11", torn_hist, "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") check("the torn line is still counted", j["history"]["rejected"].get("unparseable line"), 1) + check("and the file's loss is named", (j["history"].get("damage") or {}).get("boundaries"), 1) case("R142_11 (--accept-partial does not adopt a loss)", torn_hist, - "PARTIAL_EVIDENCE", 40, 0, extra=["--accept-partial"]) + "INSUFFICIENT_DATA", 20, 0, extra=["--accept-partial"]) + check("the records BEFORE the loss cannot support a promotion after it", + run([json.dumps(r) for r in good_rows] + ['{"torn":'] + + [json.dumps(rec(i, 30.0 + 3.0 * i)) for i in range(20, 22)])[1]["history"]["comparable"], + 2) envelope_line = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, "payload": {}, "certificate": {}}) foreign = json.dumps({"note": "another tool's line in a shared file"}) @@ -379,15 +388,20 @@ def main(): print("\nR142_15 - a rejected line is damage unless it is one of the named exceptions") for line, label, status, quality in ( ('{"schema_version":3,"record_type":"carry_run","shares":{"Bash":60.0,"Read":40.0}}', - "a record from a schema this reader does not know", "PARTIAL_EVIDENCE", "DEGRADED"), + "a record from a schema this reader does not know", "INSUFFICIENT_DATA", "EMPTY"), ('{"schema_version":2,"record_type":"carry_run","shares":{"Bash":10.0}}', - "a carry record whose shares do not sum to a population", "PARTIAL_EVIDENCE", - "DEGRADED"), + "a carry record whose shares do not sum to a population", "INSUFFICIENT_DATA", + "EMPTY"), ('{"note": "another tool\'s line in a shared file"}', "a line that was never a carry record", "CANDIDATE", "COMPLETE")): + code = {"INSUFFICIENT_DATA": 20, "PARTIAL_EVIDENCE": 40, "CANDIDATE": 10}[status] case(f"R142_15: {label}", [json.dumps(r) for r in good_rows] + [line], - status, 40 if status == "PARTIAL_EVIDENCE" else 10, - 0 if status == "PARTIAL_EVIDENCE" else 1, history_quality=quality) + status, code, 1 if status == "CANDIDATE" else 0, history_quality=quality) + if status != "CANDIDATE": + # the same rejected line BEFORE a healthy population: the loss cuts, it does not kill + case(f"R142_15: {label} — and a population after it still stands", + [line] + [json.dumps(r) for r in good_rows], "CANDIDATE", 10, 1, + comparable=6, history_quality="COMPLETE") print("\nR142_16 - a ledger line that is neither a write nor a denial is a line lost") led_dir = tempfile.mkdtemp(dir=d) @@ -410,9 +424,12 @@ def main(): optimize.sweep_quality(dict(live_impossible, turns=2000)), "COMPLETE") claims_ours = json.dumps({"schema_version": 2, "record_type": "carry_run", "ts": TS0, "sessions": 40, "turns": 1000, "carry_bytes": 10 ** 7}) - case("R142_18: a carry record with no shares is damage", + case("R142_18: a carry record with no shares key at all is a loss", [json.dumps(r) for r in good_rows] + [claims_ours], - "PARTIAL_EVIDENCE", 40, 0, history_quality="DEGRADED") + "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") + case("R142_18: ...and a population after that loss still stands", + [claims_ours] + [json.dumps(r) for r in good_rows], "CANDIDATE", 10, 1, + comparable=6, history_quality="COMPLETE") tmp_dir = tempfile.mkdtemp(dir=d) blocked_id = [f["candidate_id"] for f in optimize.analyse( None, {"comparable": good_rows, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], diff --git a/tests/test_evidence_phase2.py b/tests/test_evidence_phase2.py index edeed36..48afa82 100644 --- a/tests/test_evidence_phase2.py +++ b/tests/test_evidence_phase2.py @@ -334,7 +334,7 @@ def legacy_row(schema, scope="s", ts=None): # ---------------------------------------------------------------- OPTIMIZER ISOLATION p = os.path.join(d, "iso.jsonl") carry.history(p, facts(14), 100, scope_id="s") - records, rejected, _lines = optimize.load_history(p) + records, rejected, _lines, _ep = optimize.load_history(p) check("optimizer 1.3 tak menerima satu pun record generasi ini", len(records), 0) check("penolakannya terhitung, bukan senyap", sum(rejected.values()), 1) check("penolakannya menyebut skema", diff --git a/tests/test_multiagent.py b/tests/test_multiagent.py index 4507f7d..b8e6aab 100644 --- a/tests/test_multiagent.py +++ b/tests/test_multiagent.py @@ -117,7 +117,7 @@ def main(): for i in range(6)] mixed += [rec(100 + i * 86400, {"Bash": 80.0, "Read": 20.0}, scope="reviewer") for i in range(6)] p = write(os.path.join(d, "mixed.jsonl"), mixed) - recs, _, _ = optimize.load_history(p) + recs, _, _, _ = optimize.load_history(p) check("dua scope terbaca utuh", len(recs), 12) groups = optimize.by_scope(recs) check("record dikelompokkan per scope", sorted(groups), ["builder", "reviewer"]) @@ -212,7 +212,7 @@ def main(): # WHY: phase 2 ports acquisition, not promotion. An optimizer that guessed at a # current-generation record would be reading fields whose meaning it does not know — # the exact "legacy gains current trust" failure, in the other direction. - recs, rej, _ = optimize.load_history(hp) + recs, rej, _, _ = optimize.load_history(hp) check(f"{n} penulis paralel: optimizer 1.3 menolak skema baru, fail closed", (len(recs), sum(rej.values())), (0, n)) @@ -503,11 +503,11 @@ def deny(*a, **k): h1 = [rec(100 + i * 86400, {"Bash": 40.0, "Read": 60.0}, scope="fleet") for i in range(3)] h2 = [rec(100 + i * 86400, {"Bash": 41.0, "Read": 59.0}, scope="fleet") for i in range(3)] merged = write(os.path.join(d, "fleet.jsonl"), h1 + h2) # dua host, satu scope, digabung - recs, rej, _ = optimize.load_history(merged) + recs, rej, _, _ = optimize.load_history(merged) check("record dua host dalam satu scope bisa digabung tanpa tabrakan", len(recs), 6) check("nol tolakan saat penggabungan", sum(rej.values()), 0) dup = write(os.path.join(d, "fleet_dup.jsonl"), h1 + h2 + h1) # rsync menyalin dua kali - recs, rej, _ = optimize.load_history(dup) + recs, rej, _, _ = optimize.load_history(dup) check("penggabungan ulang idempoten (run_id sama = satu observasi)", len(recs), 6) check("salinan ganda dihitung sebagai retry", rej["duplicate run_id (retry)"], 3) @@ -529,7 +529,7 @@ def deny(*a, **k): big = os.path.join(d, "budget.jsonl") write(big, [rec(100 + i * 3600, {"Bash": 50.0, "Read": 50.0}, scope="b") for i in range(20000)]) t0 = time.time() - recs, _, _ = optimize.load_history(big) + recs, _, _, _ = optimize.load_history(big) dt = time.time() - t0 check("20k record terbaca", len(recs), 20000) check(f"20k record dalam waktu terbatas (terukur {dt:.1f} s, plafon 120 s)", dt < 120, True) diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 1c68c00..6b98bb4 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -120,7 +120,7 @@ def mutant(pairs): """ p = w(os.path.join(D, "h.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}), rec(100, {"Bash": 50.0, "Read": 50.0})]) - recs, rej, _ = optimize.load_history(p) + recs, rej, _, _ = optimize.load_history(p) assert len(recs) == 2, "populasi menyusut: dua agen dihitung satu" """), @@ -350,7 +350,7 @@ def mutant(pairs): """ p = os.path.join(D, "h.jsonl") carry.history(p, facts(1), 100, scope_id="s") - recs, rej, _ = optimize.load_history(p) + recs, rej, _, _ = optimize.load_history(p) assert (len(recs), sum(rej.values())) == (0, 1), (len(recs), dict(rej)) """), @@ -685,7 +685,7 @@ def hist_of(n): bad["evidence_quality"] = "PARTIAL" bad["skipped_by_limit"] = 7 p = w(os.path.join(D, "dup.jsonl"), rows + [good, bad]) - recs, rej, _ = optimize.load_history(p) + recs, rej, _, _ = optimize.load_history(p) keep, _d = optimize.comparable(recs) q = optimize.history_quality(keep) assert q == "PARTIAL", q @@ -702,23 +702,22 @@ def hist_of(n): assert q == "INVALID", q """), - ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history menurunkan kualitas evidence", - [("optimize.py", - ' if count and not any(x in str(reason) for x in NOT_CONTAINER_DAMAGE):', - " if False:")], + ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history memotong epoch", + [("optimize.py", ' if any(x in str(reason) for x in NOT_A_LOSS):\n return None', + " if True:\n return None")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="c") for i in range(6)] p = w(os.path.join(D, "torn_hist.jsonl"), [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) - recs, rej, _ = optimize.load_history(p) - keep, dropped = optimize.comparable(recs) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 0, ("epoch tak terpotong", len(cur)) + keep, dropped = optimize.comparable(cur) h = {"comparable": keep, "total": len(recs), "in_scope": len(recs), "rejected": rej, "dropped": dropped, "time_order": "ok"} - f = optimize.analyse(None, h, None, None, scope="c") - assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] - st = optimize.overall_status(f, h, None, optimize.population(keep), True) - assert st == "PARTIAL_EVIDENCE", st + f = optimize.analyse(None, h, None, None, scope="c", accept_partial=True) + assert all(x["state"] != "CANDIDATE" for x in f), [x["state"] for x in f] """), ("M_LEDGER_TORN_PROMOTES: ledger yang kehilangan baris bukan sampel lapangan", @@ -788,22 +787,19 @@ def hist_of(n): assert q == "DEGRADED", (counter, q) """), - ("M_STRUCTURAL_REJECT_NOT_DAMAGE: penolakan struktural menutup emitter, bukan sekadar dicatat", + ("M_STRUCTURAL_REJECT_NOT_DAMAGE: penolakan struktural memotong epoch, bukan sekadar dicatat", [("optimize.py", - 'NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares")', - 'NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares", "unsupported schema_version", "shares sum to")')], + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares")', + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", "unsupported schema_version", "shares sum to")')], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="g") for i in range(6)] bad = chr(123) + '"schema_version":3,"record_type":"carry_run","shares":' \ + chr(123) + '"Bash":60.0,"Read":40.0' + chr(125) + chr(125) p = w(os.path.join(D, "struct.jsonl"), [json.dumps(r) for r in rows] + [bad]) - recs, rej, _ = optimize.load_history(p) - keep, dropped = optimize.comparable(recs) - h = {"comparable": keep, "total": len(recs), "in_scope": len(recs), "rejected": rej, - "dropped": dropped, "time_order": "ok"} - f = optimize.analyse(None, h, None, None, scope="g") - assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 0, ("penolakan struktural tak memotong epoch", len(cur)) """), ("M_LEDGER_UNKNOWN_EVENT: baris ledger yang tak terhitung adalah baris yang hilang", @@ -825,6 +821,134 @@ def hist_of(n): assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] """), + + # ------------------------------------------- B1: the history epoch (acceptance-review blocker) + ("M_DAMAGE_POISONS_FOREVER: kerusakan lama tak boleh memblokir epoch yang bersih", + [("optimize.py", " hist_q = history_quality(hist.get(\"comparable\", []))", + " hist_q = worst_quality([history_quality(hist.get(\"comparable\", [])),\n" + " \"DEGRADED\" if (hist.get(\"damage\") or {}).get(\"boundaries\") " + "else \"COMPLETE\"])")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + p = w(os.path.join(D, "recover.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + keep, dropped = optimize.comparable(cur) + h = {"comparable": keep, "total": len(recs), "in_scope": len(cur), "rejected": rej, + "dropped": dropped, "time_order": "ok", + "damage": optimize.damage_summary(ep)} + f = optimize.analyse(None, h, None, None, scope="p") + assert any(x["state"] == "CANDIDATE" for x in f), [x["state"] for x in f] + """), + + ("M_DAMAGE_IGNORED_COMPLETELY: kehilangan yang tak teratribusi tetap memotong", + [("optimize.py", " if cut is None:\n continue", + " if True:\n continue")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="q") + for i in range(6)] + p = w(os.path.join(D, "tail.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 0, "epoch tak terpotong" + """), + + ("M_EPOCH_MERGES_ACROSS_GAP: dua epoch tak boleh digabung", + [("optimize.py", " scoped = [r for r in current if scope_of(r) == scope]", + " scoped = [r for r in recs if scope_of(r) == scope]")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + empty = os.path.join(d, "none.jsonl") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["history"]["comparable"] == 6, j["history"]["comparable"] + assert j["scope"]["records_in_epoch"] == 6, j["scope"]["records_in_epoch"] + """), + + ("M_SCOPE_DAMAGE_GLOBALIZED: kerusakan milik satu scope tak memotong scope lain", + [("optimize.py", ' return ("scope", sid)', ' return ("file", None)')], + """ + good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + broken_b = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":' \ + + chr(123) + '"Bash":10.0' + chr(125) + chr(125) + p = w(os.path.join(D, "scoped.jsonl"), [json.dumps(r) for r in good_a] + [broken_b]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("scope a ikut terpotong", len(cur)) + """), + + ("M_UNATTRIBUTABLE_DAMAGE_SCOPED: baris robek memotong SEMUA scope", + [("optimize.py", ' return ("file", None)\n', + ' return ("scope", scope_of(rec) if isinstance(rec, dict) else "default")\n')], + """ + good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + p = w(os.path.join(D, "global.jsonl"), + [json.dumps(r) for r in good_a] + [chr(123) + '"torn":']) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 0, "scope a lolos dari potongan global" + """), + + ("M_EMPTY_COMPARABLE_REPORTS_COMPLETE: nol rekaman layak bukan COMPLETE", + [("optimize.py", ' return "EMPTY"\n return worst_quality([record_quality(r) for r in records])', + ' return "COMPLETE"\n return worst_quality([record_quality(r) for r in records])')], + """ + assert optimize.history_quality([]) == "EMPTY", optimize.history_quality([]) + """), + + ("M_ZERO_CARRY_BECOMES_DAMAGE: sapuan tanpa carry itu bukti terbaca, bukan korupsi", + [("optimize.py", ' if not claims_ours:\n return False, "no shares"\n sh = {}', + ' return False, "carry record without shares"')], + """ + zero = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}) + zero["shares"] = {} + zero["bpt"] = {} + zero["carry_bytes"] = 0 + ok, why = optimize.valid_record(zero) + assert ok, ("record zero-carry ditolak", why) + assert optimize.record_quality(zero) == "EMPTY", optimize.record_quality(zero) + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) for i in range(6)] + p = w(os.path.join(D, "zero.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(zero)]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + """), + + ("M_PRE_DAMAGE_RECORD_ANCHORS: jangkar datang dari epoch yang sedang dianalisis", + [("optimize.py", " anchor = (eligible_anchor(current) or eligible_anchor(recs)", + " anchor = (eligible_anchor(recs) or eligible_anchor(current)")], + """ + import subprocess + d = tempfile.mkdtemp() + # jam mundur: rekaman SEBELUM potongan punya ts paling baru, tapi epoch adalah POSISI FISIK + old_scope = [rec(9_000_000_000 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, + scope="stale") for i in range(6)] + new_scope = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, + scope="fresh") for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), + [json.dumps(r) for r in old_scope] + [chr(123) + '"torn":'] + + [json.dumps(r) for r in new_scope]) + empty = os.path.join(d, "none.jsonl") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["scope"]["analysed"] == "fresh", j["scope"]["analysed"] + assert j["status"] == "CANDIDATE", (j["status"], j["history"]["comparable"]) + """), + ] diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 9e332d0..1b90ad8 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -71,10 +71,21 @@ def main(): p = write(os.path.join(d, "corrupt.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="a" * 32), "{tidak lengkap", "", "null", rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="a" * 32)]) # penulisan ULANG run yang sama - recs, rejected, lines = optimize.load_history(p) - check("JSONL rusak: satu record sah bertahan", len(recs), 1) + recs, rejected, lines, epochs = optimize.load_history(p) + # Sejak perbaikan B1: baris robek MEMOTONG riwayat di posisi fisiknya. Record sesudah potongan + # adalah pengamatan milik epoch BARU — termasuk bila run_id-nya sama — jadi keduanya bertahan + # dan tak ada yang dihitung sebagai retry. Yang dijaga: record SEBELUM potongan tidak ikut + # dianalisis, dan duplikat DI DALAM satu epoch tetap dibuang + menurunkan kualitas survivor. + check("JSONL rusak: record sebelum & sesudah potongan sama-sama terbaca", len(recs), 2) check("baris tak terparse dihitung", rejected["unparseable line"], 1) - check("run_id sama (retry) dihitung sekali", rejected["duplicate run_id (retry)"], 1) + check("run_id sama di SISI LAIN potongan bukan retry", rejected["duplicate run_id (retry)"], 0) + check("hanya epoch terbaru yang aktif", len(optimize.active_records(recs, epochs)), 1) + p_same = write(os.path.join(d, "same_epoch.jsonl"), + [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32), + rec(101, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) + recs_same, rej_same, _l, _e = optimize.load_history(p_same) + check("run_id sama DI DALAM satu epoch tetap dihitung sekali", + (len(recs_same), rej_same["duplicate run_id (retry)"]), (1, 1)) # Dua AGEN boleh menghasilkan metrik identik. Tanpa run_id itu dua pengamatan, bukan duplikat: # membuang salah satunya akan mengecilkan populasi yang sedang diukur. p2 = write(os.path.join(d, "twin.jsonl"), @@ -87,7 +98,7 @@ def main(): p = write(os.path.join(d, "shrink.jsonl"), [rec(100, {"Bash": 40.0, "Read": 60.0}, turns=10000), rec(200, {"Bash": 60.0, "Read": 40.0}, turns=1000)]) - recs, _, _ = optimize.load_history(p) + recs, _, _, _ = optimize.load_history(p) keep, dropped = optimize.comparable(recs) check("korpus menyusut 10x -> record lama TIDAK dibandingkan", (len(keep), len(dropped)), (1, 1)) check("jam mundur terdeteksi", diff --git a/tools/optimize.py b/tools/optimize.py index af3ce91..5c42c2a 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -37,8 +37,15 @@ OPTIMIZER_VERSION = "1.0" OUTPUT_SCHEMA_VERSION = 2 # shape of --json; bump when a field's meaning changes - # 2 = `evidence_quality` is DERIVED (may read DEGRADED/UNKNOWN), - # and `history.quality` reports the eligible history's worst + # 2 (unreleased) = `evidence_quality` is DERIVED and may read + # DEGRADED/UNKNOWN · `history.quality` is the worst quality of + # the evidence ELIGIBLE FOR THIS ANALYSIS, and is EMPTY when + # nothing is comparable · `history.damage` counts the loss + # boundaries the file holds · `scope.records_in_scope` counts the + # whole scope while `scope.records_in_epoch` counts what the + # current epoch contributes · a CANDIDATE outranks + # PARTIAL_EVIDENCE, because each finding is gated on its own + # evidence before the run is summarised THRESHOLD_SCHEMA_VERSION = 1 # bump when any threshold below changes, with a reason and a test # ---------------------------------------------------------------- frozen thresholds @@ -126,10 +133,20 @@ def _derived_record_quality(rec, schema): # the producer's own terms: a sweep that looked and found nothing usable is INVALID; one # that had nothing to look at is EMPTY return "INVALID" if nums.get("scanned", 0) > 0 else "EMPTY" - if "sessions" in nums and (nums.get("turns", 1) == 0 or nums.get("carry_bytes", 1) == 0): - # Sessions were counted, so turns were counted and carry was measured. A record reporting - # sessions with neither is not a quiet sweep, it is an impossible one. + if "sessions" in nums and nums.get("turns", 1) == 0: + # Sessions were counted, so turns were counted. A record reporting sessions without them is + # not a quiet sweep, it is an impossible one. return "INVALID" + shares = rec.get("shares") + if isinstance(shares, dict) and "carry_bytes" in nums: + # Zero carry is a REAL outcome, not a corrupt record: "A session whose every item lands on + # its final turn carries nothing" (tools/carry.py's own report). The 1.3 writer emits + # `shares: {}` for it and the current writer guards `if C else {}`. What cannot both be + # true is a share vector with no carry behind it, or carry with nothing to distribute. + if nums["carry_bytes"] == 0: + return "EMPTY" if not shares else "INVALID" + if not shares: + return "INVALID" if "scanned" in nums and nums["scanned"] < nums.get("sessions", 0): # A session is a transcript that was scanned AND cleared the turn floor, so the producer # can never report more sessions than it scanned. Forty sessions out of one scanned file @@ -217,33 +234,76 @@ def sweep_quality(live): return worst_quality([claimed, derived]) -# A rejected line is DAMAGE by default: the history held something that could not be read as a +# A rejected line is a LOSS by default: the history held something that could not be read as a # record, and a trend built from the survivors is built over a gap. The exceptions are named, and # they are the only two things a rejection can be that are not a loss, plus the one shape that was # never our record to begin with: # * a refusal by design — current-generation evidence this optimizer does not read — which is a -# contract, not a loss, and would otherwise block every history in the middle of a migration; +# contract, not a loss, and would otherwise cut every history in the middle of a migration; # * a deduplicated retry, which is bookkeeping (its quality already travels via QUALITY_FLOOR); # * a well-formed JSON line that is not a carry record at all. In a shared file that is another -# tool's entry, and treating it as lost evidence would let one stray append block a real -# population forever. -# Listed this way round on purpose: a reason added to valid_record() later defaults to DAMAGE -# rather than slipping through an allowlist nobody updated. (cross-family review, confirmation round) -NOT_CONTAINER_DAMAGE = ("current-generation", "duplicate run_id", "not an object", "no shares") +# tool's entry, and treating it as lost evidence would cut a history nothing happened to. +# Listed this way round on purpose: a reason added to valid_record() later defaults to LOSS rather +# than slipping through an allowlist nobody updated. (cross-family review, confirmation round) +NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares") +# The epoch a record belongs to: (file-global losses seen before it, losses seen before it that were +# attributed to ITS scope). A private, in-memory annotation; nothing writes it back to a file. +EPOCH_KEY = "_history_epoch" -def container_quality(rejected): - """What the history FILE itself can attest, separately from the records inside it. - A torn line in the history is a record that was written and cannot be read — a loss, not a - bound, so no flag adopts it. It used to appear only as a line in `history.rejected` while the - trend built from the surviving records was promoted as if nothing had been lost. - (cross-family review, round 1) +def damage_boundary(reason, rec=None): + """Where a rejected line cuts the promotion history -> None | ("file", None) | ("scope", id). + + The first repair made damage permanent: one torn line set the whole file DEGRADED, and since + nothing in this product expires, rotates or repairs a history — and no flag adopts a loss — a + single crash fragment disabled promotion for every scope, forever. An independent acceptance + review blocked that, correctly. + + A loss is not a verdict on the file. It is a BOUNDARY at that line's physical position: the + records before it and the records after it are two populations, and only the newest one may + support a promotion. The gap stays visible in `history.rejected` and `history.damage`. + + Attribution: a line that parses and carries a readable `scope_id` says which population lost a + record, so it cuts that scope only. A line that cannot say — unparseable, oversized, a file + that would not open — cuts every scope, because guessing would be the fail-open half of this. """ - for reason, count in (rejected or {}).items(): - if count and not any(x in str(reason) for x in NOT_CONTAINER_DAMAGE): - return "DEGRADED" - return "COMPLETE" + if any(x in str(reason) for x in NOT_A_LOSS): + return None + if isinstance(rec, dict): + sid = rec.get("scope_id") + if isinstance(sid, str) and 0 < len(sid) <= 64 and not carry.SAFE_LABEL.search(sid): + return ("scope", sid) + return ("file", None) + + +def active_epoch_key(scope, epochs): + """The epoch a record of `scope` must carry to be part of the CURRENT analysis.""" + e = epochs or {} + return (e.get("file_global", 0), (e.get("scope_local") or {}).get(scope, 0)) + + +def active_records(recs, epochs): + """The records after the newest loss that applies to their own scope. + + A record built in memory rather than read from a file carries no epoch annotation; it belongs + to the current epoch, because no loss was observed around it. + """ + out = [] + for r in (recs or []): + key = active_epoch_key(scope_of(r), epochs) + if r.get(EPOCH_KEY, key) == key: + out.append(r) + return out + + +def damage_summary(epochs): + """What the FILE holds, as diagnostics — never a gate. A historical gap can stay true while the + evidence after it is independently complete.""" + e = epochs or {} + local = dict(e.get("scope_local") or {}) + return {"boundaries": e.get("file_global", 0) + sum(local.values()), + "file_global": e.get("file_global", 0), "scope_local": local} def history_quality(records): @@ -255,7 +315,14 @@ def history_quality(records): half matters as much as the fail-closed half: a gate that blocks on evidence a finding never used is not correct, it is merely stuck. """ - return worst_quality([record_quality(r) for r in (records or [])]) + if not records: + # M1 (acceptance review): a machine consumer must never read COMPLETE and conclude that + # usable history exists. Nothing eligible is EMPTY — the producer's own word for "there was + # nothing to look at" — and it is refused by the promotion gate like every other non- + # COMPLETE value. worst_quality([]) keeps meaning COMPLETE: a FINDING that rests on no + # sampled evidence is not degraded by sampling it never used. + return "EMPTY" + return worst_quality([record_quality(r) for r in records]) # machine-readable outcome. The CLI exits 0 for every VALID run by default (a scheduler must not # treat "nothing to do" as breakage); --strict-exit maps the status to the exit code instead. @@ -323,20 +390,28 @@ def valid_record(o): if not isinstance(sv, int) or isinstance(sv, bool) or sv not in SCHEMA_SUPPORTED: return False, f"unsupported schema_version {sv!r}" sh = o.get("shares") - if not isinstance(sh, dict) or not sh: - # Two different lines, and the container quality depends on which one this is: a line that - # claims to be one of OUR records is a corrupted record (damage), a line that claims + claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o + or "run_id" in o or "carry_bytes" in o) + if not isinstance(sh, dict): + # Two different lines, and the damage classification depends on which one this is: a line + # that claims to be one of OUR records is a corrupted record (a loss), a line that claims # nothing is another tool's entry in a shared file (not ours to lose). - claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o - or "run_id" in o or "carry_bytes" in o) return False, ("carry record without shares" if claims_ours else "no shares") + if not sh: + # An EMPTY share map is what the producer writes for a sweep that measured no carry. It is + # readable evidence with nothing to compare — record_quality() calls it EMPTY and + # comparable() leaves it out — and reading it as corruption would cut a history that is + # perfectly intact. A bare `{"shares": {}}` with no sign of being ours is still foreign. + if not claims_ours: + return False, "no shares" + sh = {} for k, v in sh.items(): if not isinstance(k, str) or not isinstance(v, (int, float)) or isinstance(v, bool): return False, "non-numeric share" if v < 0 or v > 100.5: return False, "share out of range" tot = sum(sh.values()) - if not (95.0 <= tot <= 105.0): + if sh and not (95.0 <= tot <= 105.0): return False, f"shares sum to {tot:.1f}, not ~100" for k in ("ts", "turns", "sessions", "carry_bytes"): v = o.get(k) @@ -355,39 +430,65 @@ def scope_of(r): def load_history(path): - """-> (records, rejected counter, lines). Deduplication is by run_id ONLY: two agents can + """-> (records, rejected counter, lines, epochs). Deduplication is by run_id within one epoch: two agents can legitimately produce the same timestamp, turn count and shares, and discarding one of them would undercount the population. A record without a run_id (schema 0/1) cannot be deduplicated and is kept as it is.""" recs, rejected = [], collections.Counter() + cuts_file, cuts_scope = 0, collections.Counter() if not path or not os.path.exists(path): - return recs, rejected, 0 + return recs, rejected, 0, {"file_global": 0, "scope_local": {}} lines = 0 try: fh = open(path, encoding="utf-8", errors="replace") except OSError as e: rejected[f"history unreadable: {safe_err(e)}"] += 1 - return recs, rejected, 0 + # A file that would not open is a loss nobody can attribute: every scope starts a new epoch + # with no records in it, which is the fail-closed answer. + return recs, rejected, 0, {"file_global": 1, "scope_local": {}} with fh: for line in fh: line = line.strip() if not line: continue lines += 1 + o, ok, why = None, False, "" if len(line) > carry.MAX_RECORD: - rejected["record above the size cap"] += 1 + why = "record above the size cap" + else: + try: + o = json.loads(line) + except Exception: + why = "unparseable line" # a torn line cannot say whose record it was + else: + ok, why = valid_record(o) + if ok: + # Stamped at READ time, in physical order: the epoch is a position in the file, + # never a timestamp, because the clock is exactly what a damaged history cannot + # be trusted about. + o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]) + recs.append(o) continue - try: - o = json.loads(line) - except Exception: - rejected["unparseable line"] += 1 + rejected[why] += 1 + # ONE classifier for every rejection, including the ones that never became an object: + # a reason plus whatever the line could show about itself. + cut = damage_boundary(why, o) + if cut is None: continue - ok, why = valid_record(o) - (recs.append(o) if ok else rejected.__setitem__(why, rejected[why] + 1)) + if cut[0] == "scope": + cuts_scope[cut[1]] += 1 + else: + cuts_file += 1 seen, uniq = {}, [] for r in recs: rid = r.get("run_id") if isinstance(rid, str) and rid: + # Deduplication is per EPOCH. Two epochs are two populations: the same id appearing + # after a loss is that population's own observation, and dropping it — or carrying the + # excluded copy's floor into it — would let an old gap poison a healthy epoch, which is + # the defect this repair exists to remove. Inside one epoch nothing changes: the retry + # is dropped, counted, and cannot launder the survivor's quality. + rid = (r.get(EPOCH_KEY), rid) if rid in seen: rejected["duplicate run_id (retry)"] += 1 # The retry is dropped as an OBSERVATION, never as provenance. One run_id that @@ -402,7 +503,7 @@ def load_history(path): continue seen[rid] = r uniq.append(r) - return uniq, rejected, lines + return uniq, rejected, lines, {"file_global": cuts_file, "scope_local": dict(cuts_scope)} def by_scope(recs): @@ -578,8 +679,7 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): # partial history it never read, and a finding drawing on history is not waved through because # today's sweep happened to be clean. quality = sweep_quality(live) if live else "COMPLETE" - hist_q = worst_quality([history_quality(hist.get("comparable", [])), - container_quality(hist.get("rejected"))]) + hist_q = history_quality(hist.get("comparable", [])) usable = may_promote(quality, accept_partial) # findings that rest on the live sweep hist_usable = may_promote(hist_q, accept_partial) # findings that rest on the history run_ids = [r.get("run_id") for r in hist.get("comparable", []) if r.get("run_id")] @@ -745,13 +845,13 @@ def overall_status(findings, hist, live, pop, accept_partial): if live: refused.append(sweep_quality(live)) comp = hist.get("comparable", []) - damage = container_quality(hist.get("rejected")) if comp: - refused.append(worst_quality([history_quality(comp), damage])) - elif any(record_quality(r) in ("INVALID", "EMPTY") for r, _why in hist.get("dropped", [])): + refused.append(history_quality(comp)) + elif any(record_quality(r) == "INVALID" for r, _why in hist.get("dropped", [])): + # Evidence that exists and is refused. EMPTY is NOT one of these: a record that measured + # nothing is missing evidence, not bad evidence, and "keep collecting" is the honest word + # for it. A loss does not appear here at all any more — it moved the epoch instead. refused.append("INVALID") - elif damage != "COMPLETE": - refused.append(damage) if any(not may_promote(q, accept_partial) for q in refused): return "PARTIAL_EVIDENCE" if not live and len(comp) < MIN_HISTORY_FOR_TREND: @@ -936,6 +1036,12 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop if hist["comparable"]: L.append(f" history : {history_quality(hist['comparable'])} " f"(worst of {len(hist['comparable'])} eligible records)") + dmg = hist.get("damage") or {} + if dmg.get("boundaries"): + L.append(f" damage : {dmg['boundaries']} loss boundary/ies in this file " + f"({dmg.get('file_global', 0)} unattributable, " + f"{sum((dmg.get('scope_local') or {}).values())} scope-local) — evidence from " + f"before the newest one is not combined with evidence after it") L.append("") if live and live["sessions"]: C = sum(live["carry"].values()) or 1 @@ -1009,27 +1115,32 @@ def main(argv=None): "governed repository: the optimizer has no authority to change code.") a = ap.parse_args(argv) - recs, rejected, lines = load_history(a.history) + recs, rejected, lines, epochs = load_history(a.history) scopes = by_scope(recs) scopes_seen = sorted(scopes) or ["default"] + # The epoch is selected BEFORE anything is anchored: a record from before the newest loss must + # not choose the scope, the workload class, the corpus size or the trend for the population + # that came after it. + current = active_records(recs, epochs) if a.scope_id: - scope, scoped = a.scope_id, scopes.get(a.scope_id, []) + scope = a.scope_id elif recs: # Which scope gets analysed is an anchor too: taking the newest record of ANY quality let a # single INVALID sweep in another agent's scope send the whole run to a population that was - # never going to be analysable. The newest record that CAN speak chooses. - # The fallback names a scope only when NOTHING in the file is eligible — and then there is - # no population to strand and nothing that can be promoted: the status is PARTIAL_EVIDENCE - # and the emitter is closed. Both halves are pinned by tests (a reviewer read the fallback - # as a way back in; it is a label on an empty run, verified by execution). - anchor = eligible_anchor(recs) or max(recs, key=lambda r: r.get("ts") or 0) + # never going to be analysable. The newest record that CAN speak, in the current epoch, + # chooses. The fallbacks only NAME a scope when nothing is eligible or nothing survives the + # newest loss — an empty run's label, with the emitter closed either way. + anchor = (eligible_anchor(current) or eligible_anchor(recs) + or max(recs, key=lambda r: r.get("ts") or 0)) scope = scope_of(anchor) - scoped = scopes[scope] else: - scope, scoped = "default", [] + scope = "default" + scoped = [r for r in current if scope_of(r) == scope] keep, dropped = comparable(scoped) - hist = {"total": len(recs), "in_scope": len(scoped), "comparable": keep, "dropped": dropped, - "rejected": rejected, "lines": lines, "time_order": time_order(keep)} + damage = damage_summary(epochs) + hist = {"total": len(recs), "in_scope": len(scopes.get(scope, [])), "in_epoch": len(scoped), + "comparable": keep, "dropped": dropped, "rejected": rejected, "lines": lines, + "time_order": time_order(keep), "damage": damage} ledger = load_ledger(a.ledger) live = cold = None @@ -1088,14 +1199,18 @@ def main(argv=None): "history_schema_supported": list(SCHEMA_SUPPORTED), "generated": int(time.time()), "status": status, "status_code": STATUS.get(status, STATUS["INTERNAL_ERROR"]), - "scope": {"analysed": scope, "known": scopes_seen, "records_in_scope": len(scoped), - "comparable": len(keep)}, + "scope": {"analysed": scope, "known": scopes_seen, + "records_in_scope": len(scopes.get(scope, [])), + "records_in_epoch": len(scoped), "comparable": len(keep)}, # DERIVED, not the producer's summary word: a sweep that lost records reads DEGRADED # here even though the 1.3 label for it is PARTIAL (output_schema_version 2). "evidence_quality": sweep_quality(live) if live else "NO_SCAN", "history": {"records": len(recs), "comparable": len(keep), - "quality": worst_quality([history_quality(keep), - container_quality(rejected)]), + # the evidence eligible for THIS analysis, not a verdict on the file + "quality": history_quality(keep), + # ...and what the file holds regardless: a historical gap stays true while + # the evidence after it is independently complete + "damage": damage, "rejected": dict(rejected), "time_order": hist["time_order"]}, "ledger": ledger, "live": ({"sessions": live["sessions"], "turns": live["turns"], "scanned": live["scanned"], From 9596b9ee51385b3380d4c86e624b689e4a46ad33 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:48:38 +0000 Subject: [PATCH 15/26] report: name the epoch in the human report too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A run over a damaged file now says how many records the scope holds, how many of them came after the newest loss, and what the loss was — the same distinction the JSON makes between what the file holds and what the current analysis rests on. Co-Authored-By: Claude Opus 5 --- tools/optimize.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tools/optimize.py b/tools/optimize.py index 5c42c2a..33efd06 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -1017,8 +1017,10 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop + (f" (history also holds: {', '.join(sorted(s for s in scopes_seen if s != scope))})" if len(scopes_seen) > 1 else "")) L.append("scope") + epoch_note = ("" if hist.get("in_epoch", hist["in_scope"]) == hist["in_scope"] + else f" ({hist.get('in_epoch', 0)} after the newest loss)") L.append(f" history : {sources['history'] or '(none)'} — {hist['total']} records, " - f"{hist['in_scope']} in this scope, {len(hist['comparable'])} comparable, " + f"{hist['in_scope']} in this scope{epoch_note}, {len(hist['comparable'])} comparable, " f"{sum(hist['rejected'].values())} rejected, time order {hist['time_order']}") for why, n in hist["rejected"].most_common(): L.append(f" rejected: {n} × {why}") From 302b20a99648398161b596e8aeab3081e18118ec Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:52:14 +0000 Subject: [PATCH 16/26] test: both line orders for a run_id that spans a loss, and three copies inside one epoch The chosen semantics, pinned rather than inherited: the copy that lands after a loss is judged on its own evidence (clean before, bounded after -> PARTIAL), the copy excluded on the far side of the loss does not poison the epoch that follows it (bounded before, clean after -> COMPLETE, and no refusal), and three copies inside ONE epoch still leave the worst of them on the survivor (DEGRADED, two retries counted). Which copy comes first is exactly what a crash decides, so both orders are tests now. README: 1463 assertions. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- tests/test_evidence_integrity.py | 23 +++++++++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index a383238..1e98df3 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1460 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1463 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 096de43..32a015e 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -565,6 +565,29 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, rc, j, n = run(same_epoch) check("B1_13b a retry inside one epoch still lowers the survivor", (j["history"]["quality"], j["status"], n), ("PARTIAL", "PARTIAL_EVIDENCE", 0)) + # both line orders, because which copy comes first is exactly what a crash decides + clean_then_worse = (series(5) + [json.dumps(dict(rec(5, 45.0), run_id="X"))] + [TORN] + + [json.dumps(dict(rec(6, 45.0), run_id="X", evidence_quality="PARTIAL", + skipped_by_limit=9))] + + series(5, 21)) + rc, j, n = run(clean_then_worse) + check("B1_13c the post-loss copy is judged on its OWN evidence", + j["history"]["quality"], "PARTIAL") + worse_then_clean = (series(5) + + [json.dumps(dict(rec(5, 45.0), run_id="X", evidence_quality="PARTIAL", + skipped_by_limit=9))] + [TORN] + + [json.dumps(dict(rec(6, 45.0), run_id="X"))] + series(5, 21)) + rc, j, n = run(worse_then_clean) + check("B1_13d and an excluded pre-loss copy does not poison it", + (j["history"]["quality"], j["status"] == "PARTIAL_EVIDENCE"), ("COMPLETE", False)) + three_in_one = series(4) + [json.dumps(dict(rec(4, 42.0), run_id="S")), + json.dumps(dict(rec(5, 45.0), run_id="S", unreadable=1)), + json.dumps(dict(rec(6, 48.0), run_id="S", + evidence_quality="PARTIAL", skipped_by_limit=4))] + rc, j, n = run(three_in_one) + check("B1_13e three copies inside one epoch: the worst of them survives", + (j["history"]["quality"], j["history"]["rejected"].get("duplicate run_id (retry)"), n), + ("DEGRADED", 2, 0)) print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) From 056e138361b8710cc828fe601a1cf136fececccd Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:58:42 +0000 Subject: [PATCH 17/26] fix: two ways the epoch repair still consulted the wrong population MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by a cross-family lane attacking the epoch model itself, and both reproduced before being believed. - A finding that rests on the ledger alone may promote after a loss — that is per-finding evidence working — but the SCOPE it was filed under still came from the records before the loss, and the scope travels into the candidate id. A history of six `stale` records followed by a torn line produced `noop-guard-retire-stale-…` and wrote it. The scope now comes from the current epoch or from nothing: `default` when the epoch is empty, never a population that no longer exists. - The epoch stamp is a pair of COUNTS, so a record of scope "a" and one of scope "b" can carry the same numbers while belonging to different epochs. Deduplication keyed on the stamp alone therefore treated scope "b"'s record as scope "a"'s retry and pushed its PARTIAL quality onto scope "a" through QUALITY_FLOOR, turning a clean population into PARTIAL_EVIDENCE. Identity now carries the scope. - The long-run simulator discarded the epoch it was handed, so its analysis could combine records across a loss. It uses active_records() and the same damage summary the CLI does. Two mutants (M_STALE_SCOPE_ANCHOR, M_DEDUP_IGNORES_SCOPE) and four assertions (B1_14, B1_15) pin all of it; the older anchor mutant moved with the code. 1469 assertions, 55 mutants, schema-4 bytes still equal to main, skill body untouched. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- experiments/aivos/longrun.py | 11 +++++-- tests/test_evidence_integrity.py | 37 ++++++++++++++++++++++++ tests/test_mutation.py | 49 +++++++++++++++++++++++++++++++- tools/optimize.py | 20 ++++++++----- 5 files changed, 107 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index 1e98df3..7866b87 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1463 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1469 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/experiments/aivos/longrun.py b/experiments/aivos/longrun.py index 7eeed9d..2579ace 100644 --- a/experiments/aivos/longrun.py +++ b/experiments/aivos/longrun.py @@ -158,12 +158,17 @@ def longrun(days, cycles, out, plateau=20): day_new = 0 for _cycle in range(cycles): - recs, _rej, _lines, _ep = optimize.load_history(hist) - scopes = optimize.by_scope(recs) + recs, _rej, _lines, ep = optimize.load_history(hist) + # the same epoch the CLI analyses: evidence from before a loss is not combined with + # evidence after it, here either (cross-family review of the B1 repair) + current = optimize.active_records(recs, ep) + scopes = optimize.by_scope(current) for role in ROLES: keep, dropped = optimize.comparable(scopes.get(role, [])) h = {"comparable": keep, "total": len(recs), "in_scope": len(scopes.get(role, [])), - "rejected": {}, "dropped": dropped, "time_order": optimize.time_order(keep)} + "in_epoch": len(scopes.get(role, [])), "rejected": dict(_rej), + "dropped": dropped, "time_order": optimize.time_order(keep), + "damage": optimize.damage_summary(ep)} lv = live(int(30 + min(day, plateau) * (30.0 / plateau))) f = optimize.analyse(lv, h, None, None, scope=role) st = optimize.overall_status(f, h, lv, optimize.population(keep), False) diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 32a015e..5abdc33 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -589,6 +589,43 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, (j["history"]["quality"], j["history"]["rejected"].get("duplicate run_id (retry)"), n), ("DEGRADED", 2, 0)) + print("\nB1_14/B1_15 - what a cross-family review of the epoch repair found") + # a finding that rests on the ledger alone may still promote after a loss — but nothing from + # before the loss may name it, because the scope travels into the candidate id + led_lines = ['{"event": "checked"}'] * 100 + stale = [json.dumps(dict(rec(i, 30.0 + 3.0 * i), scope_id="stale")) for i in range(6)] + ld = os.path.join(tempfile.mkdtemp(dir=d), "ledger.jsonl") + with open(ld, "w", encoding="utf-8") as fh: + fh.write("\n".join(led_lines) + "\n") + hd = tempfile.mkdtemp(dir=d) + hp = write(os.path.join(hd, "history.jsonl"), stale + [TORN]) + out14 = os.path.join(hd, "cand") + pr = subprocess.run([sys.executable, OPT, "--history", hp, "--ledger", ld, "--scan", + "--emit-candidate", out14, "--json", "--strict-exit"], + capture_output=True, text=True, timeout=300) + j14 = json.loads(pr.stdout) + check("B1_14 a stale population cannot name a candidate raised after the loss", + (j14["scope"]["analysed"], j14["scope"]["records_in_epoch"], + any("stale" in c for c in j14["candidate_ids"])), ("default", 0, False)) + check("B1_14 and the ledger finding itself still stands", + (j14["status"], [c.split("-")[0] for c in j14["candidate_ids"]]), ("CANDIDATE", ["noop"])) + + # two scopes can hold the same epoch NUMBERS; identity needs the scope as well + broken_a = '{"schema_version":2,"record_type":"carry_run","scope_id":"a","shares":{"Bash":10.0}}' + broken_b = '{"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":{"Bash":10.0}}' + rows_a = [json.dumps(dict(rec(i, 30.0 + 3.0 * i), scope_id="a", + run_id="shared" if i == 0 else f"a{i}")) for i in range(6)] + row_b = json.dumps(dict(rec(9, 50.0, quality="PARTIAL", skipped=1), scope_id="b", + run_id="shared")) + rc15, j15, n15 = run([broken_a] + rows_a + [broken_b, row_b], extra=["--scope-id", "a"]) + check("B1_15 a retry in another scope is not this scope's retry", + (j15["history"]["quality"], j15["history"]["comparable"], + j15["history"]["rejected"].get("duplicate run_id (retry)")), ("COMPLETE", 6, None)) + check("B1_15 and the loss in each scope cut only that scope", + (j15["history"]["damage"]["boundaries"], + sorted((j15["history"]["damage"]["scope_local"] or {}).items())), + (2, [("a", 1), ("b", 1)])) + print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) check("nothing comparable is never reported as COMPLETE", diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 6b98bb4..85d9969 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -927,7 +927,7 @@ def hist_of(n): """), ("M_PRE_DAMAGE_RECORD_ANCHORS: jangkar datang dari epoch yang sedang dianalisis", - [("optimize.py", " anchor = (eligible_anchor(current) or eligible_anchor(recs)", + [("optimize.py", " anchor = (eligible_anchor(current)", " anchor = (eligible_anchor(recs) or eligible_anchor(current)")], """ import subprocess @@ -949,6 +949,53 @@ def hist_of(n): assert j["status"] == "CANDIDATE", (j["status"], j["history"]["comparable"]) """), + + ("M_STALE_SCOPE_ANCHOR: populasi pra-loss tak boleh menamai kandidat pasca-loss", + [("optimize.py", + " anchor = (eligible_anchor(current)\n" + " or (max(current, key=lambda r: r.get(\"ts\") or 0) if current else None))\n" + " scope = scope_of(anchor) if anchor is not None else \"default\"", + " anchor = (eligible_anchor(current) or eligible_anchor(recs)\n" + " or max(recs, key=lambda r: r.get(\"ts\") or 0))\n" + " scope = scope_of(anchor)")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="stale") + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + led = os.path.join(d, "led.jsonl") + with open(led, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", led, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["scope"]["analysed"] == "default", j["scope"]["analysed"] + assert not any("stale" in c for c in j["candidate_ids"]), j["candidate_ids"] + """), + + ("M_DEDUP_IGNORES_SCOPE: dua scope bisa punya nomor epoch yang sama", + [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY), rid)', + " rid = (r.get(EPOCH_KEY), rid)")], + """ + broken_a = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"a","shares":' \ + + chr(123) + '"Bash":10.0' + chr(125) + chr(125) + broken_b = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":' \ + + chr(123) + '"Bash":10.0' + chr(125) + chr(125) + rows_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a", + run_id="shared" if i == 0 else "a%d" % i) for i in range(6)] + row_b = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="b", run_id="shared") + row_b["evidence_quality"] = "PARTIAL" + row_b["skipped_by_limit"] = 1 + p = w(os.path.join(D, "twoscope.jsonl"), + [broken_a] + [json.dumps(r) for r in rows_a] + [broken_b, json.dumps(row_b)]) + recs, rej, _l, ep = optimize.load_history(p) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "a"] + q = optimize.history_quality(optimize.comparable(cur)[0]) + assert q == "COMPLETE", (q, rej) + """), + ] diff --git a/tools/optimize.py b/tools/optimize.py index 33efd06..324e8be 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -488,7 +488,11 @@ def load_history(path): # excluded copy's floor into it — would let an old gap poison a healthy epoch, which is # the defect this repair exists to remove. Inside one epoch nothing changes: the retry # is dropped, counted, and cannot launder the survivor's quality. - rid = (r.get(EPOCH_KEY), rid) + # The stamp is a pair of COUNTS, so a record of scope "a" and one of scope "b" can + # carry the same numbers while belonging to different epochs. Identity therefore needs + # the scope: without it, one scope's retry lowered another scope's record through + # QUALITY_FLOOR. (cross-family review of the B1 repair) + rid = (scope_of(r), r.get(EPOCH_KEY), rid) if rid in seen: rejected["duplicate run_id (retry)"] += 1 # The retry is dropped as an OBSERVATION, never as provenance. One run_id that @@ -1129,12 +1133,14 @@ def main(argv=None): elif recs: # Which scope gets analysed is an anchor too: taking the newest record of ANY quality let a # single INVALID sweep in another agent's scope send the whole run to a population that was - # never going to be analysable. The newest record that CAN speak, in the current epoch, - # chooses. The fallbacks only NAME a scope when nothing is eligible or nothing survives the - # newest loss — an empty run's label, with the emitter closed either way. - anchor = (eligible_anchor(current) or eligible_anchor(recs) - or max(recs, key=lambda r: r.get("ts") or 0)) - scope = scope_of(anchor) + # never going to be analysable. The newest record that CAN speak, IN THE CURRENT EPOCH, + # chooses; if none can, the newest record of that epoch still names it. + # Nothing from before the newest loss is consulted, not even as a label: the scope travels + # into every candidate id, so a stale population naming a ledger finding would attach a + # promotion to a population that no longer exists. (cross-family review of the B1 repair) + anchor = (eligible_anchor(current) + or (max(current, key=lambda r: r.get("ts") or 0) if current else None)) + scope = scope_of(anchor) if anchor is not None else "default" else: scope = "default" scoped = [r for r in current if scope_of(r) == scope] From e6c1f4fb6b4502e0d2f3d09e3e688296dda80a40 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 07:59:02 +0000 Subject: [PATCH 18/26] docs: the two findings the cross-family lane made in the epoch repair B1_14 (a stale population naming a post-loss candidate) and B1_15 (two scopes sharing one epoch stamp), in the same post-freeze section as the rest, with the long-run simulator's missing epoch named too. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index fc8758b..5bf1df8 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -184,3 +184,16 @@ Dedup across a boundary is also new, and is stated rather than inherited: dedupl **per epoch**. The same `run_id` on the far side of a loss is that population's own observation and is kept; inside one epoch the retry is still dropped, still counted, and still cannot launder the survivor's quality (`tests/test_optimize.py` pins both). + +### 5.3 What a cross-family lane found in the epoch repair itself + +Two more places where the repair still consulted the population it had just cut, both reproduced +before they were believed and both now pinned by a case and a mutant: + +| case | fixture | expectation | +|---|---|---| +| **B1_14** | six `stale` records, a torn line, and a ledger of 100 writes | the ledger finding may still promote — its evidence is the ledger — but the scope, and therefore the candidate id, comes from the current epoch or from nothing: `default`, never `stale` | +| **B1_15** | scope-local loss in `a`, six clean `a` records, scope-local loss in `b`, one `b` record sharing a `run_id` with `a` | the epoch stamp is a pair of counts, so two scopes can hold the same numbers; deduplication identity carries the scope, and `a` stays `COMPLETE` instead of inheriting `b`'s `PARTIAL` through the retry floor | + +The long-run simulator was discarding the epoch it was handed, so it could combine records across a +loss; it now analyses `active_records()` like the CLI. From 443abaf89546d64858b622ee421cd9ce974a26ad Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 08:54:39 +0000 Subject: [PATCH 19/26] docs(v1.4.2): freeze the trusted damage-boundary matrix before implementing it The adversarial review of the epoch model found attribution taken from the line that had just failed validation: invalid UTF-8 in scope_id survived errors="replace" as U+FFFD, read as a readable label, and cut a scope that does not exist while the real population kept crossing the loss. Frozen here, before any code: rejected lines are never an authority for their own scope (every loss is file-global), the history is decoded strictly per physical line, and the one trusted source of a scope-local boundary is a record that passed validation and whose own canonical loss counters prove the loss. Section 6.3 records the hypothesis reproduced first, with a matched control: VALID_DEGRADED_FAIL_STUCK=YES. Section 6.6 records the two B1 expectations this trust model supersedes, and what replaces each. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 180 +++++++++++++++++++++++++++++++++++ 1 file changed, 180 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index 5bf1df8..330eb64 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -197,3 +197,183 @@ before they were believed and both now pinned by a case and a mutant: The long-run simulator was discarding the epoch it was handed, so it could combine records across a loss; it now analyses `active_records()` like the CLI. + +## 6. Trusted damage boundaries (frozen before the second repair) + +Frozen against the branch head `e6c1f4f`, **before** any of it was implemented. The adversarial +review of the epoch model found that attribution was taken from the very line that had just failed +validation: a `scope_id` holding invalid UTF-8 survived `errors="replace"` as `U+FFFD`, read as a +"readable" label, and cut a scope that does not exist — while the real population kept crossing the +loss. §2 of that review reproduced it as `CANDIDATE`, one candidate file, twelve comparable records +spanning both sides of the gap. + +The rule this section freezes is about provenance, not about Unicode: + +> **A record that failed validation is not a trustworthy authority for its own scope attribution.** + +Two consequences, and one deliberate trade: + +* **Every rejection that is a loss is a FILE-GLOBAL boundary.** No rejected line may name a scope, + whatever its `scope_id` looks like — `rev`, `agent-b`, a path, a 64-character label, or bytes that + never decoded. This over-blocks: one corrupt line cuts scopes that were never damaged. That is the + chosen half of the trade (§23 of the task), because epochs recover and a fail-open crossing of a + real loss does not. +* **The history is read as BYTES and decoded strictly, per physical line.** `errors="replace"` + destroyed the evidence that decoding had failed; a line that cannot decode is now a named + rejection (`line is not valid UTF-8`) and a file-global boundary, and the reader continues at the + next line rather than abandoning the file. +* **A scope-local boundary now has exactly one trusted source** (§6.2): a record that PASSED + validation and whose own canonical loss counters prove that evidence was lost. + +### 6.1 TUTF — invalid UTF-8 (all FILE_GLOBAL, no ghost scope) + +Fixture unless stated: `6 clean scope=rev` + the bad line + `6 clean scope=rev`. The bad line is +SameWrite-shaped, holds invalid UTF-8 in the named field, and independently fails `valid_record()` +(its shares sum to 20, not ~100). + +| case | bad line | boundary | `damage.scope_local` | active epoch | status | +|---|---|---|---|---|---| +| **TUTF_01** | invalid UTF-8 inside `scope_id` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_02** | invalid UTF-8 inside `shares` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_03** | invalid UTF-8 inside `run_id` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_04** | invalid UTF-8 inside `workload_class` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_05** | one undecodable line before every record (`bad + 6 clean`) | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_06** | one undecodable line at EOF (`6 clean + bad`) | FILE_GLOBAL ×1 | `{}` | 0 | `INSUFFICIENT_DATA` | +| **TUTF_07** | two undecodable lines (`6 + bad + 6 + bad + 6`) | FILE_GLOBAL ×2 | `{}` | 6 | `CANDIDATE` | +| **TUTF_08** | one undecodable line between two scopes (`6×a + bad + 6×b`) | FILE_GLOBAL ×1 | `{}` | 6 (`b`); 0 with `--scope-id a` | `CANDIDATE`; `INSUFFICIENT_DATA` | + +In every row: no candidate may name a `run_id` from before the boundary, `scope.known` may not +contain a label that came from the rejected line, and no raw undecodable byte may appear anywhere in +the JSON output, the human output or a candidate file. + +### 6.2 TSCOPE — a rejected record does not authenticate its own `scope_id` + +Fixture: `6 clean scope=rev` + one rejected record carrying a syntactically perfect +`scope_id="ghost"` + `6 clean scope=rev`. Every row below is a loss. + +| case | rejected record | boundary | `damage.scope_local` | active epoch | status | +|---|---|---|---|---|---| +| **TSCOPE_01** | shares sum to 20, not ~100 | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_02** | a non-numeric share value | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_03** | `sessions` negative (implausible) | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_04** | `evidence_quality` outside the vocabulary | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_05** | `schema_version: 3`, a legacy schema this reader cannot name | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_06** | a carry record with no `shares` key at all | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_07** | `record_type` is not `carry_run` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | + +Still **not** a loss, so still no boundary at all: a well-formed foreign JSON line, a bare object +with no `shares` and no sign of being ours, and current-generation (schema-4) evidence refused by +design. A migration must not read as file damage. + +Canaries, each used as the rejected record's `scope_id`, each of which must leave +`damage.scope_local == {}` and must not appear in `scope.known`: `/etc/passwd.d/synthetic`, +`agent-b`, `rev`, a 63-character label, a 64-character label, a label holding control bytes, and a +label that never decoded. **Cardinality:** 2000 rejected lines carrying 2000 distinct `scope_id` +values produce `file_global == 2000`, `scope_local == {}` — attacker-controlled strings cannot add a +single public map key. + +### 6.3 The hypothesis this repair had to test first: a VALID record that is DEGRADED + +Reproduced on `e6c1f4f` before anything was designed for it, with a matched control: + +```text +6 clean + 1 valid schema-2 record with unreadable=3 + N clean, scope agent-a + N=2 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files + N=6 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files + N=60 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files +control (identical populations, no degraded record) + N=2,6,60 CANDIDATE / history.quality COMPLETE / 1 candidate file +``` + +`VALID_DEGRADED_FAIL_STUCK=YES`. The record passes `valid_record()`, so no rejection and no +boundary was ever considered; it stays in `comparable()` forever, and `history_quality()` is the +worst of that set. No amount of later healthy evidence clears it — the same fail-stuck shape the +epoch model was built to remove, reached through the one door the epoch model did not watch. + +So §9 of the task applies, and narrowly. A **fully validated** record whose OWN canonical loss +counters prove degraded evidence creates a **scope-local** recovery boundary after itself. This is +trusted where a rejected line is not: the record passed structural validation, its `scope_id` is the +same field every accepted record publishes through `scope.known`, and the damage fact comes from +`RECORD_LOSS_COUNTERS`, not from guessing what a corrupt line meant. + +| case | fixture (scope `a` unless stated) | boundary | active epoch | status | `history.quality` | +|---|---|---|---|---|---| +| **VD_01** | 6 clean + DEGRADED + 2 clean | SCOPE_LOCAL `{a: 1}` | 2 | `INSUFFICIENT_DATA` | `COMPLETE` | +| **VD_02** | 6 clean + DEGRADED + 6 clean | SCOPE_LOCAL `{a: 1}` | 6 | `CANDIDATE` | `COMPLETE` | +| **VD_03** | 6 clean + DEGRADED + 60 clean | SCOPE_LOCAL `{a: 1}` | 60 | `CANDIDATE` | `COMPLETE` | +| **VD_04** | VD_02, reading the candidate | — | — | the candidate names none of the pre-boundary `run_id`s, and not the degraded record's own | | +| **VD_05** | 6 clean + DEGRADED at EOF | SCOPE_LOCAL `{a: 1}` | 0 | `INSUFFICIENT_DATA` | `EMPTY` | +| **VD_06** | `6×a`, `6×b`, DEGRADED `a`, `6×a`, `6×b` | SCOPE_LOCAL `{a: 1}` | `a`: 6 · `b`: 12 | `CANDIDATE` both | `COMPLETE` | +| **VD_07** | 6 clean + valid `PARTIAL` by `skipped_by_limit=3` + 6 clean | **none** | 13 | `PARTIAL_EVIDENCE`; `CANDIDATE` with `--accept-partial` | `PARTIAL` | +| **VD_08** | 6 clean + a schema-1 record (`UNKNOWN`, cannot attest) + 6 clean | **none** | 13 | `PARTIAL_EVIDENCE` | `UNKNOWN` | +| **VD_09** | 6 clean + a valid `INVALID` record + a valid `EMPTY` (zero-carry) record + 6 clean | **none** | 14 | `CANDIDATE` | `COMPLETE` | + +The degraded record belongs to the OLD epoch (§14): it sits before its own boundary, in physical +order, never in the population that recovers. Its degradation stays reported in `history.damage`. + +`PARTIAL`, `UNKNOWN`, `INVALID` and `EMPTY` are deliberately excluded. Only an actual loss counter +cuts: a bound the caller asked for is an intentional population, a legacy schema that cannot attest +completeness is not proof that a line was lost, and an ineligible record is not a damaged one. + +### 6.4 Retry and a trusted boundary (§15) + +Deduplication identity becomes `(scope, FILE-GLOBAL epoch, run_id)` — the scope-local component is +deliberately **not** part of it. Across a file-global loss the reader cannot tell whether a repeated +`run_id` is the same run, so both copies stand (B1_13). Across a *trusted* scope-local boundary the +file is intact and the reader knows exactly what happened, so a repeated `run_id` is the same +logical run retrying, and the later copy is bookkeeping. + +| case | physical order (scope `a`) | expectation | +|---|---|---| +| **VD_10** | `clean X`, `degraded retry X`, 6 clean | boundary `{a: 1}` after the degraded copy; both copies pre-boundary; `duplicate run_id (retry)` = 1; active epoch 6, `CANDIDATE`, `COMPLETE` | +| **VD_11** | `degraded X`, `clean retry X`, 6 clean | boundary `{a: 1}` after the degraded copy; the clean retry is dropped as the same run's bookkeeping and never becomes evidence in the recovered epoch; `duplicate run_id (retry)` = 1; active epoch 6, `CANDIDATE`, `COMPLETE` | + +Neither order launders the loss into the recovered epoch, and neither poisons it forever. + +### 6.5 File-global recovery is unchanged (§20, §21) + +| case | fixture | expectation | +|---|---|---| +| **MSG_01** | `6×a`, `6×b`, torn line, `6×a`, `2×b` | `a`: epoch 6, `CANDIDATE`; `b`: epoch 2, `INSUFFICIENT_DATA`; nothing from before the cut helps either | +| **MSG_02** | the same with the sufficiencies reversed (`2×a`, `6×b` after the cut) | `a`: `INSUFFICIENT_DATA`; `b`: `CANDIDATE` | + +File-global means every scope is cut **at that position**, never that a scope is disabled forever: +`B1_01`–`B1_04` still pin `+0 / +2 / +6 / +60`. + +### 6.6 Frozen-decision amendments to §5 (§27) + +Two rows of the B1 matrix were written against the old attribution rule and are wrong under this +one. Both are replaced rather than deleted, and the property each was protecting keeps a case. + +```text +B1_13e three copies of one run_id in one epoch, the middle one carrying unreadable=1 + was: history.quality DEGRADED, 2 duplicates, 0 files + now: the middle copy is a VALID record proving a loss, so it cuts scope-locally. + The property "the worst copy survives inside one epoch" moves to a fixture whose + copies are PARTIAL/COMPLETE (no loss counter); the degraded-copy behaviour is + VD_10/VD_11 above. + +B1_15 two scope-local boundaries taken from two REJECTED lines + was: damage.scope_local == {a: 1, b: 1} from rejected records + now: rejected lines are file-global, so the same fixture yields file_global == 2. + The property it protected — two scopes can hold the same epoch NUMBERS, so the + deduplication identity must carry the scope — is re-pinned with the same shape + built from TRUSTED sources: a valid DEGRADED record in each scope. +``` + +No other B1 or R142 expectation changes. `R142_01`–`R142_06` and `B1_01`–`B1_14` are re-run +unchanged. + +### 6.7 The residual limit this repair does not close (§24) + +`LEGACY_STRUCTURALLY_VALID_CORRUPTION_LIMITATION=YES`. A legacy flat record carries no integrity +tag. A corruption that turns one valid record into a *different* valid record — `scope_id` flipped +from `agent-a` to `agent-b`, a share vector rewritten to another vector that still sums to ~100 — is +indistinguishable from a record the producer meant to write. Nothing in this repair detects it, and +nothing can: the information needed to tell them apart is absent from the format. What the repair +does guarantee is narrower and checkable: a line that *fails* validation never supplies attribution, +and a line that cannot be decoded is never mistaken for one that can. + +### 6.8 Deviations from this frozen section + +None. From 24c81d441e7fa0e3d2f7dc045d4de0e25465aee1 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 08:56:19 +0000 Subject: [PATCH 20/26] =?UTF-8?q?test(v1.4.2):=20RED=20=E2=80=94=20TUTF/TS?= =?UTF-8?q?COPE/VD/MSG=20cases=20for=20the=20trusted=20damage=20boundary?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 38 failures on e6c1f4f, one per row of docs/V142_COUNTEREXAMPLES.md §6: TUTF_01..08 invalid UTF-8 in a physical line is an unattributable loss. Today errors="replace" turns it into U+FFFD, the reader calls that a readable scope_id, and a scope that does not exist is cut while the real population keeps crossing the gap — the candidate names run ids from both sides of the loss. TSCOPE_01..07 a record that failed validation is not an authority for its own scope_id, whatever the string looks like. Canaries and a 2000-label cardinality attack pin that no rejected label can become a public map key. VD_01..11 a record that PASSED validation and whose own canonical loss counters prove a loss is the one trusted source of a scope-local boundary. Reproduced first, with a matched control: today that record keeps history.quality DEGRADED forever, at +2, +6 and +60 healthy records. MSG_01/02 file-global recovery, already correct, pinned against regress. The write() helper now takes bytes: a line that is not valid UTF-8 cannot be written through a text handle, and that line is the input under test. Co-Authored-By: Claude Opus 5 --- tests/test_evidence_integrity.py | 240 ++++++++++++++++++++++++++++++- 1 file changed, 238 insertions(+), 2 deletions(-) diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 5abdc33..94bfff1 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -57,9 +57,13 @@ def rec(i, share, quality="COMPLETE", schema=2, counters=True, sessions=40, turn def write(path, rows): - with open(path, "w", encoding="utf-8") as fh: + """Rows may be bytes: a history holding a line that is not valid UTF-8 is exactly the input + the decode rule has to be tested against, and it cannot be written through a text handle.""" + with open(path, "wb") as fh: for r in rows: - fh.write((r if isinstance(r, str) else json.dumps(r)) + "\n") + if not isinstance(r, bytes): + r = (r if isinstance(r, str) else json.dumps(r)).encode("utf-8") + fh.write(r + b"\n") return path @@ -121,6 +125,236 @@ def transcript(path, turns=80, listing=False, torn=False): "time_order": "ok"} + +def trusted_boundaries(): + """docs/V142_COUNTEREXAMPLES.md §6 — a line that failed validation is not an authority. + + The epoch model's attribution came from the very line that had just been rejected: invalid + UTF-8 in `scope_id` survived `errors="replace"` as U+FFFD, read as a "readable" label, and cut + a scope that does not exist while the real population kept crossing the loss. Every case below + was frozen before this repair existed and is RED on e6c1f4f. + """ + d = tempfile.mkdtemp(prefix="sw-142-tb-") + + def ser(n, first=0, scope="rev", per_week=3.0, base=30.0): + return [json.dumps(rec(first + i, base + per_week * i, scope=scope)) for i in range(n)] + + def later(n=6, scope="rev"): + """The population AFTER the boundary: its own climb, so it can promote on its own.""" + return ser(n, 20, scope=scope, base=48.0) + + def bad_utf8(field, scope="ghost"): + """A SameWrite-shaped line holding invalid UTF-8, which also fails validation on its own.""" + r = dict(rec(99, 50.0, scope=scope), shares={"Bash": 10.0, "Read": 10.0}) + if field == "shares": + r["shares"] = {"Bash": 10.0, "@@M@@": 10.0} + else: + r[field] = "@@M@@" + return json.dumps(r).encode().replace(b"@@M@@", b"\xff\xfe\x80") + + def rejected_rec(scope="ghost", **over): + r = dict(rec(99, 50.0, scope=scope), shares={"Bash": 10.0, "Read": 10.0}) + r.update(over) + return json.dumps(r) + + def dmg(j): + x = j["history"].get("damage") or {} + return x.get("file_global"), x.get("scope_local") + + def shape(rc, j, n): + return ((j["status"], rc, n, j["scope"]["records_in_epoch"], j["history"]["comparable"]) + + dmg(j)) + + def run_paths(rows, extra=()): + dd = tempfile.mkdtemp(dir=d) + hist = write(os.path.join(dd, "history.jsonl"), rows) + out = os.path.join(dd, "cand") + p = subprocess.run([sys.executable, OPT, "--history", hist, "--ledger", + os.path.join(dd, "none.jsonl"), "--emit-candidate", out, "--json", + "--strict-exit", "--scan"] + list(extra), + capture_output=True, text=True, timeout=300) + files = [os.path.join(b, f) for b, _sub, fs in os.walk(out) for f in fs + if f != ".optimize.lock"] + return p.returncode, json.loads(p.stdout), files + + # ---------------------------------------------------------------- TUTF (§6.1) + print("\nTUTF - a line that cannot be decoded cannot name a scope") + for name, field in (("TUTF_01", "scope_id"), ("TUTF_02", "shares"), + ("TUTF_03", "run_id"), ("TUTF_04", "workload_class")): + rc, j, n = run(ser(6) + [bad_utf8(field)] + later()) + check(f"{name} invalid UTF-8 in {field} is an unattributable loss", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + check(f"{name} the decode failure is counted by name", + j["history"]["rejected"].get("line is not valid UTF-8"), 1) + check(f"{name} nothing from the rejected line reaches the report", + ("�" in json.dumps(j, ensure_ascii=False), "ghost" in j["scope"]["known"]), + (False, False)) + + rc, j, n = run([bad_utf8("scope_id")] + ser(6)) + check("TUTF_05 an undecodable line before every record", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + rc, j, n = run(ser(6) + [bad_utf8("scope_id")]) + check("TUTF_06 an undecodable line at EOF leaves no epoch", + shape(rc, j, n), ("INSUFFICIENT_DATA", 20, 0, 0, 0, 1, {})) + rc, j, n = run(ser(6) + [bad_utf8("scope_id")] + ser(6, 20, base=48.0) + + [bad_utf8("run_id")] + ser(6, 40, base=66.0)) + check("TUTF_07 two undecodable lines are two boundaries", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 2, {})) + between = ser(6, 0, "a") + [bad_utf8("scope_id")] + later(6, "b") + rc, j, n = run(between) + check("TUTF_08 an undecodable line between two scopes cuts both", + shape(rc, j, n) + (j["scope"]["analysed"],), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, "b")) + rc, j, n = run(between, extra=["--scope-id", "a"]) + check("TUTF_08 ...and the scope before it has no epoch left", + shape(rc, j, n), ("INSUFFICIENT_DATA", 20, 0, 0, 0, 1, {})) + + rc, j, files = run_paths(ser(6) + [bad_utf8("scope_id")] + later()) + body = open(files[0], encoding="utf-8").read() if files else "" + ids = sorted(x.strip() for line in body.splitlines() if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check("TUTF_01 the candidate rests only on the recovered epoch", + ids, sorted("rrev%d" % i for i in range(20, 26))) + + # ---------------------------------------------------------------- TSCOPE (§6.2) + print("\nTSCOPE - a rejected record does not authenticate its own scope_id") + for label, bad in ( + ("TSCOPE_01 shares that do not sum to a population", rejected_rec()), + ("TSCOPE_02 a non-numeric share", + rejected_rec(shares={"Bash": "lots", "Read": 50.0})), + ("TSCOPE_03 an impossible session count", rejected_rec(sessions=-1)), + ("TSCOPE_04 a quality word outside the vocabulary", + rejected_rec(evidence_quality="SPLENDID")), + ("TSCOPE_05 a legacy schema this reader cannot name", rejected_rec(schema_version=3)), + ("TSCOPE_06 a carry record with no shares key", + json.dumps({k: v for k, v in rec(99, 50.0, scope="ghost").items() + if k != "shares"})), + ("TSCOPE_07 a record_type this reader does not know", + rejected_rec(record_type="carry_note"))): + rc, j, n = run(ser(6) + [bad] + later()) + check(label, shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + check(label + " — and no ghost scope is published", "ghost" in j["scope"]["known"], False) + + for label, line in (("a foreign JSON line", json.dumps({"note": "another tool's entry"})), + ("a bare object that never claimed to be ours", + json.dumps({"note": "x", "shares": None})), + ("current-generation evidence refused by design", + json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}))): + rc, j, n = run(ser(6) + [line] + later()) + check("TSCOPE control: " + label + " is not damage", + shape(rc, j, n), ("CANDIDATE", 10, 1, 12, 12, 0, {})) + + print("\nTSCOPE canaries - no rejected label may become a public map key") + for canary in ("/etc/passwd.d/synthetic", "agent-b", "rev", "x" * 63, "y" * 64, + "ctl\x01label", "shares: 100"): + rc, j, n = run(ser(6, scope="real") + [rejected_rec(scope=canary)] + + later(6, "real")) + check("TSCOPE canary %r stays out of the public report" % canary[:18], + (dmg(j), canary in j["scope"]["known"]), ((1, {}), False)) + + flood = [rejected_rec(scope="s%04d" % i) for i in range(2000)] + rc, j, n = run(flood + later()) + check("TSCOPE cardinality: 2000 rejected labels add no public key", + (dmg(j), j["scope"]["known"], len(json.dumps(j["history"]["damage"])) < 200), + ((2000, {}), ["rev"], True)) + + # ---------------------------------------------------------------- VD (§6.3) + print("\nVD - a VALID record whose own counters prove a loss cuts its own scope") + deg = json.dumps(rec(6, 48.0, scope="a", unreadable=3)) + check("the fail-stuck hypothesis names a real quality", + optimize.record_quality(json.loads(deg)), "DEGRADED") + for label, tail, epoch, status, code, files, quality in ( + ("VD_01 an epoch too small to carry a trend", ser(2, 20, "a", base=48.0), 2, + "INSUFFICIENT_DATA", 20, 0, "COMPLETE"), + ("VD_02 a sufficient epoch promotes on its own", ser(6, 20, "a", base=48.0), 6, + "CANDIDATE", 10, 1, "COMPLETE"), + ("VD_03 and it still promotes sixty records later", + ser(60, 20, "a", per_week=1.0, base=20.0), 60, "CANDIDATE", 10, 1, "COMPLETE")): + rc, j, n = run(ser(6, 0, "a") + [deg] + tail) + check(label, (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"].get("quality")) + dmg(j), + (status, code, files, epoch, quality, 0, {"a": 1})) + + rc, j, files = run_paths(ser(6, 0, "a") + [deg] + ser(6, 20, "a", base=48.0)) + body = open(files[0], encoding="utf-8").read() if files else "" + ids = sorted(x.strip() for line in body.splitlines() if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check("VD_04 the degraded record is not evidence in the epoch it opened", + ids, sorted("ra%d" % i for i in range(20, 26))) + + rc, j, n = run(ser(6, 0, "a") + [deg]) + check("VD_05 a degradation at EOF leaves no epoch to promote from", + (j["status"], rc, n, j["scope"]["records_in_epoch"], j["history"].get("quality")) + + dmg(j), ("INSUFFICIENT_DATA", 20, 0, 0, "EMPTY", 0, {"a": 1})) + + both = (ser(6, 0, "a") + ser(6, 0, "b") + [deg] + ser(6, 20, "a", base=48.0) + + ser(6, 20, "b", base=48.0)) + rc, j, n = run(both, extra=["--scope-id", "a"]) + check("VD_06 the degraded scope starts again after its own loss", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 6, 0, {"a": 1})) + rc, j, n = run(both, extra=["--scope-id", "b"]) + check("VD_06 ...and the neighbour keeps every record it ever had", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 12, 0, {"a": 1})) + + bounded = json.dumps(rec(6, 48.0, scope="a", quality="PARTIAL", skipped=3)) + rows = ser(6, 0, "a") + [bounded] + ser(6, 20, "a", base=48.0) + rc, j, n = run(rows) + check("VD_07 an intentional bound is not damage", + (j["status"], j["scope"]["records_in_epoch"], j["history"].get("quality")) + dmg(j), + ("PARTIAL_EVIDENCE", 13, "PARTIAL", 0, {})) + rc, j, n = run(rows, extra=["--accept-partial"]) + check("VD_07 ...and the flag still adopts it", (j["status"], n), ("CANDIDATE", 1)) + + rc, j, n = run(ser(6, 0, "a") + [json.dumps(rec(6, 48.0, scope="a", schema=1))] + + ser(6, 20, "a", base=48.0)) + check("VD_08 a schema that cannot attest completeness is not a loss", + (j["status"], j["scope"]["records_in_epoch"], j["history"].get("quality")) + dmg(j), + ("PARTIAL_EVIDENCE", 13, "UNKNOWN", 0, {})) + + rc, j, n = run(ser(6, 0, "a") + + [json.dumps(rec(6, 48.0, scope="a", sessions=0, scanned=40)), + json.dumps(dict(rec(7, 48.0, scope="a"), shares={}, bpt={}, carry_bytes=0))] + + ser(6, 20, "a", base=48.0)) + check("VD_09 an ineligible record is not a damaged one", + (j["status"], j["scope"]["records_in_epoch"], j["history"]["comparable"]) + dmg(j), + ("CANDIDATE", 14, 12, 0, {})) + + # ---------------------------------------------------------------- retry (§6.4) + print("\nVD_10/VD_11 - a retry may neither launder a loss nor poison the epoch after it") + clean_x = json.dumps(dict(rec(5, 45.0, scope="a"), run_id="X")) + deg_x = json.dumps(dict(rec(6, 48.0, scope="a", unreadable=2), run_id="X")) + for label, rows in (("VD_10 clean first, the retry reports the loss", + ser(5, 0, "a") + [clean_x, deg_x] + ser(6, 20, "a", base=48.0)), + ("VD_11 the loss first, the retry reports clean", + ser(5, 0, "a") + [deg_x, clean_x] + ser(6, 20, "a", base=48.0))): + rc, j, n = run(rows) + check(label, (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"].get("quality"), + j["history"]["rejected"].get("duplicate run_id (retry)")) + dmg(j), + ("CANDIDATE", 10, 1, 6, "COMPLETE", 1, 0, {"a": 1})) + + # ---------------------------------------------------------------- file-global (§6.5) + print("\nMSG - a file-global cut applies to every scope at that position, and only there") + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 5' + msg1 = (ser(6, 0, "a") + ser(6, 0, "b") + [TORN] + ser(6, 20, "a", base=48.0) + + ser(2, 20, "b", base=48.0)) + for scope, status, code, files, epoch in (("a", "CANDIDATE", 10, 1, 6), + ("b", "INSUFFICIENT_DATA", 20, 0, 2)): + rc, j, n = run(msg1, extra=["--scope-id", scope]) + check(f"MSG_01 scope {scope} after an unattributable loss", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + (status, code, files, epoch, 1, {})) + msg2 = (ser(6, 0, "a") + ser(6, 0, "b") + [TORN] + ser(2, 20, "a", base=48.0) + + ser(6, 20, "b", base=48.0)) + for scope, status, code, files, epoch in (("a", "INSUFFICIENT_DATA", 20, 0, 2), + ("b", "CANDIDATE", 10, 1, 6)): + rc, j, n = run(msg2, extra=["--scope-id", scope]) + check(f"MSG_02 scope {scope} when the sufficiencies are reversed", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + (status, code, files, epoch, 1, {})) + + def main(): d = tempfile.mkdtemp(prefix="sw-142-fx-") good_rows = [rec(i, 30.0 + 3.0 * i) for i in range(6)] @@ -626,6 +860,8 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, sorted((j15["history"]["damage"]["scope_local"] or {}).items())), (2, [("a", 1), ("b", 1)])) + trusted_boundaries() + print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) check("nothing comparable is never reported as COMPLETE", From 7f37713fac707839a21b12e910e7ed5141efedb3 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 09:12:25 +0000 Subject: [PATCH 21/26] fix(optimizer): a rejected line is never an authority for its own scope MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes the HIGH blocker the adversarial review of the epoch model found, and the MEDIUM and two LOWs that shared its root cause. The defect was not Unicode. It was provenance: attribution was read off the very line that had just failed validation. Reproduced first, on e6c1f4f: six clean records, one SameWrite-shaped line whose `scope_id` holds invalid UTF-8 and which also fails validation, six more clean records. errors="replace" turned the bytes into U+FFFD, damage_boundary() called that a readable label and cut a scope nobody has, the real population was never cut, and the run wrote a candidate whose evidence_run_ids span both sides of the loss (CANDIDATE / exit 10 / epoch 12 / comparable 12). Three changes, one rule: A record that failed validation is not a trustworthy authority for its own scope attribution. * The history is read as BYTES and decoded strictly, one physical line at a time. A line that cannot decode is a named rejection ("line is not valid UTF-8") and a loss; the reader continues at the next line rather than abandoning the file. The size cap now measures the bytes it rejects. * Every loss a rejected line represents is FILE-GLOBAL. No rejected line may name a scope, whatever its scope_id looks like. This over-blocks — one corrupt line cuts scopes that were never damaged — and that is the chosen half of the trade: epochs recover, a fail-open crossing of a real loss does not. It also removes the ghost keys and the unbounded cardinality in one move rather than three patches (2000 rejected labels now add zero public map keys). * `history.damage.scope_local` keeps exactly one source, found by testing the hypothesis rather than assuming it: a record that PASSED validation and whose own canonical loss counters prove evidence was lost. Measured on the old head, such a record kept history.quality DEGRADED at +2, +6 and +60 healthy records, while the identical populations without it promoted every time — the same fail-stuck shape the epoch model exists to remove, through the one door it did not watch. It now opens a scope-local recovery boundary after itself, and belongs to the epoch it closes rather than the one it opens. Deduplication identity drops the scope-local half of the epoch stamp: across a file-global loss the reader cannot tell whether a repeated run_id is the same run, so both copies stand; across a trusted boundary the file is intact and the repeat is the same run retrying, so its cleaner copy cannot enter the recovered epoch and launder the loss its twin reported. Also found while verifying privacy: valid_record() echoed a rejected line's own `schema_version` VALUE into `history.rejected`, which reaches --json, the human report and any log that keeps them. Truncating it is not a bound — thirty characters of a credential is still the credential — so only a number is echoed and anything else is named by type. PARTIAL (a bound the caller asked for), UNKNOWN (a schema that cannot attest) and INVALID/EMPTY (readable but not comparable) deliberately create no boundary: the repair target is loss continuity, not every ineligible record. Two frozen B1 expectations are superseded rather than quietly re-run, with the reason and the replacement recorded in docs/V142_COUNTEREXAMPLES.md §6.6. 17 suites, 1545 assertions, 0 failures. 66 mutants, all RED on the mutation and GREEN on the real source. Six planted bad implementations — lenient decode, trusted rejected scope, ignored loss, permanent global cut, no candidate ever, no recovery boundary — are each caught by the suites. Schema-4 encoded bytes are identical to base main for clean, malformed and bounded acquisitions, and the legacy optimizer still refuses schema-4 by name. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- docs/V142_COUNTEREXAMPLES.md | 22 +++ tests/test_evidence_integrity.py | 49 ++++++- tests/test_mutation.py | 241 ++++++++++++++++++++++++++++--- tools/optimize.py | 140 ++++++++++++------ 5 files changed, 385 insertions(+), 69 deletions(-) diff --git a/README.md b/README.md index 7866b87..02e6c5a 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1469 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1545 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index 330eb64..eba605e 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -377,3 +377,25 @@ and a line that cannot be decoded is never mistaken for one that can. ### 6.8 Deviations from this frozen section None. + +### 6.9 Added after the freeze (neither a deviation nor a replacement) + +Two cases were added while the repair was built. Neither changes a frozen expectation; each pins a +property the frozen rows implied but did not state. + +```text +TUTF_09 a line that would be a perfectly good record BUT FOR its bytes. + Every TUTF_01..08 fixture also fails validation on its own, so a lenient decode and a + strict one could in principle agree on the outcome by accident. Here they cannot: with + errors="replace" the line is ACCEPTED, publishes U+FFFD through scope.known and creates no + boundary; strictly it is a loss like any other. Expectation: 12 records read (not 13), + file-global boundary ×1, scope.known == ["rev"]. + +TPRIV a rejected line carrying synthetic secret- and path-shaped strings. + Found while verifying §35: valid_record() echoed the rejected line's own + `schema_version` VALUE into `history.rejected`, which reaches --json, the human report and + every log that keeps them. Truncating it is not a bound — thirty characters of a + credential is still the credential — so only a NUMBER is echoed now, and any other value + is named by type ("unsupported schema_version of type str"). A real schema number is still + named in full. +``` diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 94bfff1..41ea65d 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -216,6 +216,35 @@ def run_paths(rows, extra=()): check("TUTF_01 the candidate rests only on the recovered epoch", ids, sorted("rrev%d" % i for i in range(20, 26))) + # A line that would be a perfectly good record BUT FOR its bytes: with a lenient decode it is + # accepted and publishes U+FFFD as a scope; strictly, it is a loss like any other. (post-freeze + # addition, docs §6.9 — it strengthens TUTF_01 rather than changing any frozen expectation.) + whole = dict(rec(99, 50.0, scope="@@M@@")) + intact_but_undecodable = json.dumps(whole).encode().replace(b"@@M@@", b"\xff\xfe\x80") + rc, j, n = run(ser(6) + [intact_but_undecodable] + later()) + check("TUTF_09 an otherwise-valid record with undecodable bytes is a loss, not a record", + shape(rc, j, n) + (j["history"]["records"], j["scope"]["known"]), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, 12, ["rev"])) + + # ---------------------------------------------------------------- privacy (§35) + print("\nTPRIV - a rejected line's own content is never echoed into public output") + secret = "sk-synthetic-NOTAREALKEY-0123456789" + leaky = dict(rec(99, 50.0, scope="/home/synthetic/.ssh/id_ed25519"), + shares={"Bash": 10.0, "Read": 10.0}, workload_class=secret, + run_id=secret + "-run", schema_version=secret) + rc, j, files = run_paths(ser(6) + [json.dumps(leaky)] + later()) + blob = json.dumps(j, ensure_ascii=False) + spec = "".join(open(f, encoding="utf-8").read() for f in files) + check("TPRIV nothing from the rejected line reaches --json or a candidate file", + (secret[:16] in blob, "/home/synthetic" in blob, + secret[:16] in spec, "/home/synthetic" in spec, dmg(j)), + (False, False, False, False, (1, {}))) + check("TPRIV the reason names the value's TYPE, never the value", + sorted(j["history"]["rejected"]), ["unsupported schema_version of type str"]) + check("TPRIV a real schema number is still named", + optimize.valid_record({"schema_version": 3, "shares": {"Bash": 100.0}})[1], + "unsupported schema_version 3") + # ---------------------------------------------------------------- TSCOPE (§6.2) print("\nTSCOPE - a rejected record does not authenticate its own scope_id") for label, bad in ( @@ -814,14 +843,18 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, rc, j, n = run(worse_then_clean) check("B1_13d and an excluded pre-loss copy does not poison it", (j["history"]["quality"], j["status"] == "PARTIAL_EVIDENCE"), ("COMPLETE", False)) + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §6.6): the middle copy used to carry + # unreadable=1, which under the trust model is a VALID record proving a loss — it now opens an + # epoch rather than sitting inside one. The property this case protects is unchanged and the + # fixture states it with copies that lose nothing; the degraded-copy behaviour is VD_10/VD_11. three_in_one = series(4) + [json.dumps(dict(rec(4, 42.0), run_id="S")), - json.dumps(dict(rec(5, 45.0), run_id="S", unreadable=1)), - json.dumps(dict(rec(6, 48.0), run_id="S", - evidence_quality="PARTIAL", skipped_by_limit=4))] + json.dumps(dict(rec(5, 45.0), run_id="S", + evidence_quality="PARTIAL", skipped_by_limit=4)), + json.dumps(dict(rec(6, 48.0), run_id="S"))] rc, j, n = run(three_in_one) check("B1_13e three copies inside one epoch: the worst of them survives", (j["history"]["quality"], j["history"]["rejected"].get("duplicate run_id (retry)"), n), - ("DEGRADED", 2, 0)) + ("PARTIAL", 2, 0)) print("\nB1_14/B1_15 - what a cross-family review of the epoch repair found") # a finding that rests on the ledger alone may still promote after a loss — but nothing from @@ -845,8 +878,12 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, (j14["status"], [c.split("-")[0] for c in j14["candidate_ids"]]), ("CANDIDATE", ["noop"])) # two scopes can hold the same epoch NUMBERS; identity needs the scope as well - broken_a = '{"schema_version":2,"record_type":"carry_run","scope_id":"a","shares":{"Bash":10.0}}' - broken_b = '{"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":{"Bash":10.0}}' + # Frozen-decision amendment (§6.6): the two boundaries used to be taken from REJECTED lines, + # which no longer name a scope at all. The shape this case needs — one scope-local cut in each + # of two scopes, so both carry the same epoch numbers — is built from the only trusted source + # there is: a validated record whose own loss counter proves it lost evidence. + broken_a = json.dumps(dict(rec(0, 20.0), scope_id="a", run_id="dega", unreadable=1)) + broken_b = json.dumps(dict(rec(0, 20.0), scope_id="b", run_id="degb", unreadable=1)) rows_a = [json.dumps(dict(rec(i, 30.0 + 3.0 * i), scope_id="a", run_id="shared" if i == 0 else f"a{i}")) for i in range(6)] row_b = json.dumps(dict(rec(9, 50.0, quality="PARTIAL", skipped=1), scope_id="b", diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 85d9969..c1b9ece 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -49,9 +49,11 @@ def rec(ts, shares, scope="default", turns=1000, run_id=None): "scanned": 100, "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "shares": shares, "bpt": {{k: 1.0 for k in shares}}}} def w(path, rows): - with open(path, "w", encoding="utf-8") as fh: + with open(path, "wb") as fh: for r in rows: - fh.write((r if isinstance(r, str) else json.dumps(r)) + chr(10)) + if not isinstance(r, bytes): + r = (r if isinstance(r, str) else json.dumps(r)).encode() + fh.write(r + chr(10).encode()) return path import evidence_acquire, evidence_history from evidence.absence import Absence @@ -703,8 +705,8 @@ def hist_of(n): """), ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history memotong epoch", - [("optimize.py", ' if any(x in str(reason) for x in NOT_A_LOSS):\n return None', - " if True:\n return None")], + [("optimize.py", " return not any(x in str(reason) for x in NOT_A_LOSS)", + " return False")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="c") for i in range(6)] @@ -846,8 +848,8 @@ def hist_of(n): """), ("M_DAMAGE_IGNORED_COMPLETELY: kehilangan yang tak teratribusi tetap memotong", - [("optimize.py", " if cut is None:\n continue", - " if True:\n continue")], + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if False:\n cuts_file += 1")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="q") for i in range(6)] @@ -878,21 +880,22 @@ def hist_of(n): """), ("M_SCOPE_DAMAGE_GLOBALIZED: kerusakan milik satu scope tak memotong scope lain", - [("optimize.py", ' return ("scope", sid)', ' return ("file", None)')], + [("optimize.py", " cuts_scope[scope_of(o)] += 1", " cuts_file += 1")], """ good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(6)] - broken_b = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":' \ - + chr(123) + '"Bash":10.0' + chr(125) + chr(125) - p = w(os.path.join(D, "scoped.jsonl"), [json.dumps(r) for r in good_a] + [broken_b]) + deg_b = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="b") + deg_b["unreadable"] = 2 + p = w(os.path.join(D, "scoped.jsonl"), [json.dumps(r) for r in good_a] + [json.dumps(deg_b)]) recs, rej, _l, ep = optimize.load_history(p) - cur = optimize.active_records(recs, ep) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "a"] assert len(cur) == 6, ("scope a ikut terpotong", len(cur)) """), ("M_UNATTRIBUTABLE_DAMAGE_SCOPED: baris robek memotong SEMUA scope", - [("optimize.py", ' return ("file", None)\n', - ' return ("scope", scope_of(rec) if isinstance(rec, dict) else "default")\n')], + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " cuts_scope[scope_of(o) if isinstance(o, dict) else 'default'] += 1")], """ good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(6)] @@ -976,13 +979,15 @@ def hist_of(n): """), ("M_DEDUP_IGNORES_SCOPE: dua scope bisa punya nomor epoch yang sama", - [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY), rid)', - " rid = (r.get(EPOCH_KEY), rid)")], + [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid)', + " rid = (r.get(EPOCH_KEY, (0, 0))[0], rid)")], """ - broken_a = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"a","shares":' \ - + chr(123) + '"Bash":10.0' + chr(125) + chr(125) - broken_b = chr(123) + '"schema_version":2,"record_type":"carry_run","scope_id":"b","shares":' \ - + chr(123) + '"Bash":10.0' + chr(125) + chr(125) + def degraded(scope): + x = rec(100, {"Bash": 20.0, "Read": 80.0}, scope=scope) + x["unreadable"] = 1 + return json.dumps(x) + broken_a = degraded("a") + broken_b = degraded("b") rows_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a", run_id="shared" if i == 0 else "a%d" % i) for i in range(6)] row_b = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="b", run_id="shared") @@ -996,6 +1001,204 @@ def hist_of(n): assert q == "COMPLETE", (q, rej) """), + # --------------------------- the trusted damage boundary (adversarial-review blocker B-UTF8) + ("M_UTF8_REPLACEMENT_ATTRIBUTED: byte yang tak ter-decode bukan record, dan bukan label", + [("optimize.py", ' text = raw.decode("utf-8")', + ' text = raw.decode("utf-8", "replace")')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + # sah dalam segala hal KECUALI byte-nya: dengan decode longgar ia jadi record diterima + ghost = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="@@M@@") + bad = json.dumps(ghost).encode().replace("@@M@@".encode(), bytes([255, 254, 128])) + p = w(os.path.join(D, "utf8.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("line is not valid UTF-8") == 1, dict(rej) + assert len(recs) == 6, ("baris tak ter-decode diterima sebagai record", len(recs)) + assert optimize.damage_summary(ep)["file_global"] == 1, optimize.damage_summary(ep) + assert all(chr(65533) not in optimize.scope_of(r) for r in recs), "U+FFFD masuk sebagai scope" + """), + + ("M_REJECTED_VALUE_ECHOED: isi baris yang ditolak tak boleh masuk output publik", + [("optimize.py", ' if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)):\n' + ' return repr(v)\n return "of type " + type(v).__name__', + " return repr(v)")], + """ + secret = "sk-synthetic-NOTAREALKEY-0123456789" + bad = rec(100, {"Bash": 10.0, "Read": 10.0}, scope="/home/synthetic/.ssh/id_ed25519") + bad["schema_version"] = secret + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + p = w(os.path.join(D, "leak.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + assert all(secret[:12] not in k for k in rej), list(rej) + assert list(rej) == ["unsupported schema_version of type str"], list(rej) + """), + + ("M_REJECTED_SCOPE_TRUSTED: record yang gagal validasi bukan otoritas atas scope-nya sendiri", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " sid = (o or {}).get('scope_id')\n" + " if isinstance(sid, str) and 0 < len(sid) <= 64 and not carry.SAFE_LABEL.search(sid):\n" + " cuts_scope[sid] += 1\n" + " else:\n" + " cuts_file += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + bad = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="ghost") + p = w(os.path.join(D, "trusted.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, d + assert d["file_global"] == 1, d + assert len(optimize.active_records(recs, ep)) == 0, "populasi rev tak ikut terpotong" + """), + + ("M_REJECTED_SCOPE_GHOST_KEY: label dari baris yang ditolak tak boleh jadi kunci publik", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " cuts_scope[str((o or {}).get('scope_id') or 'default')] += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + for label in ("/etc/passwd.d/synthetic", "agent-b", "x" * 64): + bad = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope=label) + p = w(os.path.join(D, "ghost.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, (label, d) + assert d["file_global"] == 1, (label, d) + """), + + ("M_REJECTED_SCOPE_CARDINALITY: regex 'label yang tampak aman' tetap mempercayai baris ditolak", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " sid = (o or {}).get('scope_id')\n" + " if isinstance(sid, str) and sid.isalnum() and len(sid) <= 32:\n" + " cuts_scope[sid] += 1\n" + " else:\n" + " cuts_file += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + flood = [json.dumps(rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="s%04d" % i)) + for i in range(2000)] + p = w(os.path.join(D, "flood.jsonl"), flood + [json.dumps(r) for r in rows]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, len(d["scope_local"]) + assert d["file_global"] == 2000, d["file_global"] + """), + + ("M_GLOBAL_DAMAGE_NOT_CUT: kegagalan decode adalah kehilangan, bukan catatan kaki", + [("optimize.py", + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares")', + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", "not valid UTF-8")')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + ghost = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="@@M@@") + bad = json.dumps(ghost).encode().replace("@@M@@".encode(), bytes([255, 254, 128])) + p = w(os.path.join(D, "notcut.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 1, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 0, "epoch tak terpotong" + """), + + ("M_GLOBAL_DAMAGE_POISONS_FOREVER: potongan global memotong di POSISI, bukan selamanya", + [("optimize.py", " if r.get(EPOCH_KEY, key) == key:", + " if r.get(EPOCH_KEY, key) == key and not (epochs or {}).get('file_global'):")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="rev") for i in range(6)] + p = w(os.path.join(D, "poison.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 6, len(optimize.active_records(recs, ep)) + """), + + ("M_VALID_DEGRADED_POISONS_SCOPE_FOREVER: record sah yang kehilangan bukti membuka epoch", + [("optimize.py", " if degrades_scope(o):", " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a") + deg["unreadable"] = 3 + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "stuck.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg)] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("epoch pemulihan tak terbuka", len(cur)) + assert optimize.history_quality(optimize.comparable(cur)[0]) == "COMPLETE", "masih DEGRADED" + """), + + ("M_VALID_DEGRADED_CUTS_ALL_SCOPES: kehilangan milik satu scope hanya memotong scope itu", + [("optimize.py", " cuts_scope[scope_of(o)] += 1", + " cuts_file += 1")], + """ + a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + b = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="b") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a") + deg["unreadable"] = 3 + p = w(os.path.join(D, "neighbour.jsonl"), + [json.dumps(r) for r in a] + [json.dumps(r) for r in b] + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "b"] + assert len(cur) == 6, ("tetangga kehilangan epoch-nya", len(cur)) + """), + + ("M_DEGRADED_RECORD_INCLUDED_POST_BOUNDARY: record yang melaporkan kehilangan ada di epoch LAMA", + [("optimize.py", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])\n" + " recs.append(o)\n" + " if degrades_scope(o):", + " if degrades_scope(o):\n" + " cuts_scope[scope_of(o)] += 1\n" + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])\n" + " recs.append(o)\n" + " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a", run_id="degraded-one") + deg["unreadable"] = 3 + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "order.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg)] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert "degraded-one" not in [r.get("run_id") for r in cur], "record DEGRADED masuk epoch baru" + assert len(cur) == 6, len(cur) + """), + + ("M_DEGRADED_RETRY_NO_RECOVERY: retry tak boleh mencuci kehilangan yang dilaporkan kembarannya", + [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid)', + " rid = (scope_of(r), r.get(EPOCH_KEY), rid)")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(5)] + deg = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + clean_retry = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a", run_id="X") + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "retry.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg), json.dumps(clean_retry)] + + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("salinan bersih dari run yang sama ikut jadi bukti", len(cur)) + """), + ] diff --git a/tools/optimize.py b/tools/optimize.py index 324e8be..5e6c830 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -41,9 +41,13 @@ # DEGRADED/UNKNOWN · `history.quality` is the worst quality of # the evidence ELIGIBLE FOR THIS ANALYSIS, and is EMPTY when # nothing is comparable · `history.damage` counts the loss - # boundaries the file holds · `scope.records_in_scope` counts the - # whole scope while `scope.records_in_epoch` counts what the - # current epoch contributes · a CANDIDATE outranks + # boundaries the file holds: `file_global` is every loss a + # rejected line represents, and `scope_local` maps a scope to + # the losses its OWN VALIDATED RECORDS reported — never a label + # read off a line that failed validation, so its keys are always + # scopes that also appear in `scope.known` · `scope.records_in_scope` + # counts the whole scope while `scope.records_in_epoch` counts what + # the current epoch contributes · a CANDIDATE outranks # PARTIAL_EVIDENCE, because each finding is gated on its own # evidence before the run is summarised THRESHOLD_SCHEMA_VERSION = 1 # bump when any threshold below changes, with a reason and a test @@ -252,29 +256,37 @@ def sweep_quality(live): EPOCH_KEY = "_history_epoch" -def damage_boundary(reason, rec=None): - """Where a rejected line cuts the promotion history -> None | ("file", None) | ("scope", id). - - The first repair made damage permanent: one torn line set the whole file DEGRADED, and since - nothing in this product expires, rotates or repairs a history — and no flag adopts a loss — a - single crash fragment disabled promotion for every scope, forever. An independent acceptance - review blocked that, correctly. +def rejection_is_loss(reason): + """Did this rejected line cost the history a record? -> bool. A loss is not a verdict on the file. It is a BOUNDARY at that line's physical position: the records before it and the records after it are two populations, and only the newest one may - support a promotion. The gap stays visible in `history.rejected` and `history.damage`. + support a promotion. The gap stays visible in `history.rejected` and `history.damage`. (The + first repair made damage permanent instead, and an independent acceptance review blocked it: + nothing in this product expires, rotates or repairs a history, and no flag adopts a loss, so a + single crash fragment disabled promotion for every scope, forever.) + """ + return not any(x in str(reason) for x in NOT_A_LOSS) + + +def degrades_scope(rec): + """Does this VALIDATED record's own canonical evidence prove that it lost records? -> bool. + + The one trusted source of a scope-local boundary, and the reason it is trusted is provenance, + not syntax: the record passed valid_record(), its `scope_id` is the same field every accepted + record already publishes through `scope.known`, and the damage fact comes from + RECORD_LOSS_COUNTERS rather than from guessing what a corrupt line meant. - Attribution: a line that parses and carries a readable `scope_id` says which population lost a - record, so it cuts that scope only. A line that cannot say — unparseable, oversized, a file - that would not open — cuts every scope, because guessing would be the fail-open half of this. + Only an actual loss qualifies. A bound the caller asked for (PARTIAL), a legacy schema that + cannot attest completeness (UNKNOWN) and a readable record nothing can be compared with + (INVALID/EMPTY) are not proof that a line went missing, and must not open an epoch. """ - if any(x in str(reason) for x in NOT_A_LOSS): - return None - if isinstance(rec, dict): - sid = rec.get("scope_id") - if isinstance(sid, str) and 0 < len(sid) <= 64 and not carry.SAFE_LABEL.search(sid): - return ("scope", sid) - return ("file", None) + if not isinstance(rec, dict): + return False + schema = rec.get("schema_version", 0) + if isinstance(schema, bool) or not isinstance(schema, int): + schema = 0 + return _derived_record_quality(rec, schema) == "DEGRADED" def active_epoch_key(scope, epochs): @@ -299,7 +311,14 @@ def active_records(recs, epochs): def damage_summary(epochs): """What the FILE holds, as diagnostics — never a gate. A historical gap can stay true while the - evidence after it is independently complete.""" + evidence after it is independently complete. + + `file_global` is every loss a rejected line represents: a rejected line cannot say whose record + it was, so it cuts every scope at its position. `scope_local` therefore has exactly one source — + records that PASSED validation and whose own canonical counters reported a loss — and its keys + are always scopes that also appear in `scope.known`. No attacker-controlled string from a + rejected line can add a key here, whatever it looks like, and the map's cardinality is bounded + by the real scopes in the file rather than by the corrupt lines in it.""" e = epochs or {} local = dict(e.get("scope_local") or {}) return {"boundaries": e.get("file_global", 0) + sum(local.values()), @@ -371,6 +390,20 @@ def safe_err(exc): # ---------------------------------------------------------------- history: load and validate +def safe_label(v): + """A value read off a rejected line, before it may appear in a public reason string. + + Only a number can be a schema version, so only a number is echoed. Anything else is a corrupt + line's own content, it carries no diagnostic value beyond its type, and a reason string is + public output: `history.rejected` keys reach --json, the human report and any log that keeps + them. Truncating such a value is not a bound — thirty characters of a credential is still the + credential — so it is named by TYPE and never by content. + """ + if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)): + return repr(v) + return "of type " + type(v).__name__ + + def valid_record(o): """-> (ok, reason). Malformed input must not poison a trend; it must be counted and dropped.""" if not isinstance(o, dict): @@ -381,14 +414,16 @@ def valid_record(o): # the exact "a record gains trust by defaulting" failure, in the direction nobody watches. # Refuse it by NAME, and count the refusal, until the optimizer is ported. if isinstance(o.get("envelope"), dict): - return False, ("unsupported schema_version %r: current-generation (v1.4) evidence, " + return False, ("unsupported schema_version %s: current-generation (v1.4) evidence, " "not read by this optimizer" - % o["envelope"].get("schema_version")) + % safe_label(o["envelope"].get("schema_version"))) if o.get("record_type") not in (None, "carry_run"): return False, "unknown record_type" sv = o.get("schema_version", 0) if not isinstance(sv, int) or isinstance(sv, bool) or sv not in SCHEMA_SUPPORTED: - return False, f"unsupported schema_version {sv!r}" + # The reason string is public: it becomes a key of `history.rejected`. A rejected line's + # own content therefore goes through safe_label() before it can be echoed there. + return False, f"unsupported schema_version {safe_label(sv)}" sh = o.get("shares") claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o or "run_id" in o or "carry_bytes" in o) @@ -440,44 +475,58 @@ def load_history(path): return recs, rejected, 0, {"file_global": 0, "scope_local": {}} lines = 0 try: - fh = open(path, encoding="utf-8", errors="replace") + # BYTES, decoded strictly per physical line. `errors="replace"` destroyed the evidence that + # a decode had failed: undecodable bytes inside `scope_id` arrived as U+FFFD, passed for a + # readable label, and cut a scope that does not exist — while the real population kept + # crossing the loss. A line the reader cannot decode is named and counted as a loss; the + # reader then continues at the next line rather than abandoning the file. + fh = open(path, "rb") except OSError as e: rejected[f"history unreadable: {safe_err(e)}"] += 1 # A file that would not open is a loss nobody can attribute: every scope starts a new epoch # with no records in it, which is the fail-closed answer. return recs, rejected, 0, {"file_global": 1, "scope_local": {}} with fh: - for line in fh: - line = line.strip() - if not line: + for raw in fh: + raw = raw.strip() + if not raw: continue lines += 1 o, ok, why = None, False, "" - if len(line) > carry.MAX_RECORD: + if len(raw) > carry.MAX_RECORD: # the cap is on BYTES; so is the line that hit it why = "record above the size cap" else: try: - o = json.loads(line) - except Exception: - why = "unparseable line" # a torn line cannot say whose record it was + text = raw.decode("utf-8") + except UnicodeDecodeError: + why = "line is not valid UTF-8" else: - ok, why = valid_record(o) + try: + o = json.loads(text) + except Exception: + why = "unparseable line" # a torn line cannot say whose record it was + else: + ok, why = valid_record(o) if ok: # Stamped at READ time, in physical order: the epoch is a position in the file, # never a timestamp, because the clock is exactly what a damaged history cannot # be trusted about. o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]) recs.append(o) + if degrades_scope(o): + # The record belongs to the epoch it closes, never to the one it opens: it is + # stamped first and the counter moves after it. A population that recovers is + # not founded on the observation that reported the loss. + cuts_scope[scope_of(o)] += 1 continue rejected[why] += 1 - # ONE classifier for every rejection, including the ones that never became an object: - # a reason plus whatever the line could show about itself. - cut = damage_boundary(why, o) - if cut is None: - continue - if cut[0] == "scope": - cuts_scope[cut[1]] += 1 - else: + # EVERY loss a rejected line represents is file-global. A record that failed validation + # is not a trustworthy authority for its own scope attribution: the field that would + # name the population is part of the line this reader just refused to believe. Reading + # it anyway is the fail-open half — it invents a scope that may not exist and leaves + # the damaged one uncut. This over-blocks on purpose, and the over-block is temporary + # because epochs recover; a crossing of a real loss is not. + if rejection_is_loss(why): cuts_file += 1 seen, uniq = {}, [] for r in recs: @@ -492,7 +541,12 @@ def load_history(path): # carry the same numbers while belonging to different epochs. Identity therefore needs # the scope: without it, one scope's retry lowered another scope's record through # QUALITY_FLOOR. (cross-family review of the B1 repair) - rid = (scope_of(r), r.get(EPOCH_KEY), rid) + # ...and only the FILE-GLOBAL half of the stamp. Across a file-global loss the reader + # cannot tell whether a repeated id is the same run, so both copies stand. Across a + # TRUSTED scope-local boundary the file is intact and the reader knows exactly what + # happened: the repeat is the same logical run retrying, and letting its cleaner copy + # into the recovered epoch would launder the loss its own twin reported. + rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid) if rid in seen: rejected["duplicate run_id (retry)"] += 1 # The retry is dropped as an OBSERVATION, never as provenance. One run_id that From 189db57d02b2851f447e9672e9c1272bfee98b56 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 09:26:06 +0000 Subject: [PATCH 22/26] fix(optimizer): one boundary per logical run, and a foreign record_type is not our loss MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two findings, both reproduced before they were believed. A cross-family review of the trust repair found a HIGH in the recovery mechanism itself. The scope-local boundary was opened at READ time, before deduplication, so a copy that was then discarded as `duplicate run_id (retry)` still moved the counter — and this reader's own rule calls a deduplicated retry bookkeeping rather than a loss. Its fixture, measured on the previous head: DEGRADED X · 6 healthy records · DEGRADED X again -> INSUFFICIENT_DATA, records_in_epoch=0, scope_local={"a": 2} Six healthy records sat between the two copies and none of them survived, and repeating `6 healthy + one more copy of X` held the scope down indefinitely: the fail-stuck shape this repair exists to remove, rebuilt out of its own recovery mechanism. A boundary is now opened at most once per `(scope, file-global epoch, run_id)`. A record with no `run_id` cannot be shown to be a retry and stays its own observation; across a file-global loss the identity differs, because there the reader cannot tell whether a repeated id is the same run at all. A genuinely different second loss still cuts (`VD_13`). The second was found while verifying §6 of the task — a migration must not read as file damage — rather than reported by anyone. A line carrying another tool's `record_type` in a shared history was classified `unknown record_type` and counted as a LOSS, so a foreign entry cut the file for every scope. The distinction `no shares` already draws one check further down now applies here too: a line that ALSO carries our fields (schema_version / run_id / carry_bytes) with an unknown record_type is a corrupted record of ours and stays a loss; a line that carries none of them is another tool's entry and is counted without a boundary. New frozen cases VD_12, VD_12b, VD_13, VD_13b, VD_13c and a foreign-record_type control; new mutants M_DEGRADED_RETRY_CUTS_TWICE and M_FOREIGN_RECORD_TYPE_IS_DAMAGE, both with their positive control in the same body. docs/V142_COUNTEREXAMPLES.md §6.10 records the finding and its fixture. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- docs/V142_COUNTEREXAMPLES.md | 33 ++++++++++++++++++++ tests/test_evidence_integrity.py | 32 ++++++++++++++++++++ tests/test_mutation.py | 52 +++++++++++++++++++++++++++++--- tools/optimize.py | 33 +++++++++++++++++--- 5 files changed, 142 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 02e6c5a..9f57d8f 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1545 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1553 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index eba605e..0a55af2 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -399,3 +399,36 @@ TPRIV a rejected line carrying synthetic secret- and path-shaped strings. is named by type ("unsupported schema_version of type str"). A real schema number is still named in full. ``` + +Third addition, found while verifying §6 of the task ("a migration must not read as file damage") +rather than reported by anyone: + +```text +TSCOPE control: another tool's record_type in a shared history. + `{"record_type": "hermes_run", "ts": ..., "note": ...}` was classified + "unknown record_type" and counted as a LOSS, so a foreign entry cut the file. The + distinction `no shares` already draws one check further down now applies here too: a line + that ALSO carries our fields (schema_version / run_id / carry_bytes) with an unknown + record_type is a corrupted record of ours and stays a loss (TSCOPE_07); a line that carries + none of them is another tool's entry and is counted without a boundary. +``` + +### 6.10 What a cross-family review found in the trust repair itself + +| case | fixture (scope `a`) | expectation | +|---|---|---| +| **VD_12** | `DEGRADED X`, 6 clean, `DEGRADED X` again at EOF | one boundary, not two: active epoch 6, `CANDIDATE`, `scope_local == {a: 1}`, one counted retry | +| **VD_12b** | the same pattern repeated (`DEG X`, 6 clean, `DEG X`, 6 clean, `DEG X`) | still one boundary; active epoch 12, `CANDIDATE` | +| **VD_13** | `DEGRADED X`, 6 clean, `DEGRADED Y` (a different run) | two boundaries — a real second loss still cuts | +| **VD_13b** | two `DEGRADED` records carrying no `run_id` at all | two boundaries: a record that cannot be shown to be a retry is its own observation | +| **VD_13c** | `DEGRADED X` in scope `a` and `DEGRADED X` in scope `b` | `{a: 1, b: 1}` — the same id in another scope is that scope's own loss | + +The boundary was opened before deduplication, so a copy that was then discarded as +`duplicate run_id (retry)` still moved the counter. Measured before the fix, the reviewer's own +fixture gave `INSUFFICIENT_DATA`, `records_in_epoch=0`, `scope_local={"a": 2}` with six healthy +records sitting between the two copies — and repeating `6 healthy + one more copy of X` held the +scope down indefinitely: **the fail-stuck shape this repair exists to remove, rebuilt out of its own +recovery mechanism.** A boundary is now opened at most once per `(scope, file-global epoch, run_id)`. +A record with no `run_id` cannot be shown to be a retry and stays its own observation; across a +file-global loss the identity differs, because there the reader cannot tell whether a repeated id is +the same run at all. diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 41ea65d..f3896ea 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -265,6 +265,8 @@ def run_paths(rows, extra=()): check(label + " — and no ghost scope is published", "ghost" in j["scope"]["known"], False) for label, line in (("a foreign JSON line", json.dumps({"note": "another tool's entry"})), + ("another tool's record_type in a shared history", + json.dumps({"record_type": "hermes_run", "ts": TS0, "note": "not ours"})), ("a bare object that never claimed to be ours", json.dumps({"note": "x", "shares": None})), ("current-generation evidence refused by design", @@ -363,6 +365,36 @@ def run_paths(rows, extra=()): j["history"]["rejected"].get("duplicate run_id (retry)")) + dmg(j), ("CANDIDATE", 10, 1, 6, "COMPLETE", 1, 0, {"a": 1})) + # VD_12/VD_13: what a cross-family review of THIS repair found. A retry reports the same loss + # its twin already reported, and a deduplicated retry is bookkeeping by this reader's own rule. + # Letting a late copy open a SECOND boundary let "six healthy records, then one more copy of X" + # erase a recovered epoch — on repeat, forever: the fail-stuck shape this repair exists to + # remove, rebuilt out of its own recovery mechanism. (docs §6.10) + print("\nVD_12/VD_13 - one boundary per logical run, and still one per real loss") + healthy = ser(6, 20, "a", base=48.0) + rc, j, n = run([deg] + healthy + [deg]) + check("VD_12 a late copy of the SAME degraded run does not cut again", + (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"]["rejected"].get("duplicate run_id (retry)")) + dmg(j), + ("CANDIDATE", 10, 1, 6, 1, 0, {"a": 1})) + rc, j, n = run([deg] + healthy + [deg] + ser(6, 40, "a", base=66.0) + [deg]) + check("VD_12 ...and repeating the pattern cannot hold the scope down forever", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 12, 0, {"a": 1})) + other = json.dumps(dict(json.loads(deg), run_id="ra-other", ts=TS0 + 30 * 604800)) + rc, j, n = run([deg] + healthy + [other]) + check("VD_13 a genuinely different second loss still cuts", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + ("INSUFFICIENT_DATA", 20, 0, 0, 0, {"a": 2})) + anon = json.dumps({k: v for k, v in json.loads(deg).items() if k != "run_id"}) + rc, j, n = run([anon, anon] + healthy) + check("VD_13 a degraded record with no run_id cannot be shown to be a retry", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 6, 0, {"a": 2})) + twin_b = json.dumps(dict(json.loads(deg), scope_id="b")) + rc, j, n = run([deg, twin_b] + healthy, extra=["--scope-id", "a"]) + check("VD_13 the same run_id in another scope is that scope's own loss", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), + ("CANDIDATE", 6, 0, {"a": 1, "b": 1})) + # ---------------------------------------------------------------- file-global (§6.5) print("\nMSG - a file-global cut applies to every scope at that position, and only there") TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 5' diff --git a/tests/test_mutation.py b/tests/test_mutation.py index c1b9ece..5eaaba6 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -791,8 +791,8 @@ def hist_of(n): ("M_STRUCTURAL_REJECT_NOT_DAMAGE: penolakan struktural memotong epoch, bukan sekadar dicatat", [("optimize.py", - 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares")', - 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", "unsupported schema_version", "shares sum to")')], + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares",\n "not our record_type")', + 'NOT_A_LOSS = ("unsupported schema_version", "shares sum to")')], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="g") for i in range(6)] @@ -1019,6 +1019,26 @@ def degraded(scope): assert all(chr(65533) not in optimize.scope_of(r) for r in recs), "U+FFFD masuk sebagai scope" """), + ("M_FOREIGN_RECORD_TYPE_IS_DAMAGE: entri alat lain di berkas bersama bukan kehilangan kita", + [("optimize.py", + ' return False, ("unknown record_type" if claims_ours else "not our record_type")', + ' return False, "unknown record_type"')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + foreign = json.dumps({"record_type": "hermes_run", "ts": 100, "note": "bukan record kita"}) + p = w(os.path.join(D, "foreign_type.jsonl"), [json.dumps(r) for r in rows] + [foreign]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "populasi rev ikut terpotong" + # kontrol: baris yang MENGAKU record kita dengan record_type asing TETAP kehilangan + ours = json.dumps({"record_type": "carry_note", "schema_version": 2, "run_id": "x", + "shares": {"Bash": 100.0}}) + p2 = w(os.path.join(D, "ours_type.jsonl"), [json.dumps(r) for r in rows] + [ours]) + recs2, rej2, _l2, ep2 = optimize.load_history(p2) + assert optimize.damage_summary(ep2)["file_global"] == 1, optimize.damage_summary(ep2) + """), + ("M_REJECTED_VALUE_ECHOED: isi baris yang ditolak tak boleh masuk output publik", [("optimize.py", ' if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)):\n' ' return repr(v)\n return "of type " + type(v).__name__', @@ -1093,8 +1113,8 @@ def degraded(scope): ("M_GLOBAL_DAMAGE_NOT_CUT: kegagalan decode adalah kehilangan, bukan catatan kaki", [("optimize.py", - 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares")', - 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", "not valid UTF-8")')], + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares",\n "not our record_type")', + 'NOT_A_LOSS = ("not valid UTF-8",)')], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") for i in range(6)] @@ -1179,6 +1199,30 @@ def degraded(scope): assert len(cur) == 6, len(cur) """), + ("M_DEGRADED_RETRY_CUTS_TWICE: satu boundary per run logis, bukan per salinan", + [("optimize.py", " if ident is None or ident not in opened:", + " if True:")], + """ + def deg(seq, bash, run_id): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = 2 + return json.dumps(r) + healthy = [json.dumps(rec(100 + (20 + i) * 604800, + {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "retrycut.jsonl"), [deg(0, 30.0, "X")] + healthy + [deg(0, 30.0, "X")]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {"a": 1}, d + assert len(optimize.active_records(recs, ep)) == 6, len(optimize.active_records(recs, ep)) + # kontrol: kehilangan kedua dari run yang BENAR-BENAR lain tetap memotong + p2 = w(os.path.join(D, "realsecond.jsonl"), + [deg(0, 30.0, "X")] + healthy + [deg(30, 30.0, "Y")]) + recs2, rej2, _l2, ep2 = optimize.load_history(p2) + assert optimize.damage_summary(ep2)["scope_local"] == {"a": 2}, optimize.damage_summary(ep2) + """), + ("M_DEGRADED_RETRY_NO_RECOVERY: retry tak boleh mencuci kehilangan yang dilaporkan kembarannya", [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid)', " rid = (scope_of(r), r.get(EPOCH_KEY), rid)")], diff --git a/tools/optimize.py b/tools/optimize.py index 5e6c830..dafd765 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -249,7 +249,8 @@ def sweep_quality(live): # tool's entry, and treating it as lost evidence would cut a history nothing happened to. # Listed this way round on purpose: a reason added to valid_record() later defaults to LOSS rather # than slipping through an allowlist nobody updated. (cross-family review, confirmation round) -NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares") +NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", + "not our record_type") # The epoch a record belongs to: (file-global losses seen before it, losses seen before it that were # attributed to ITS scope). A private, in-memory annotation; nothing writes it back to a file. @@ -417,16 +418,23 @@ def valid_record(o): return False, ("unsupported schema_version %s: current-generation (v1.4) evidence, " "not read by this optimizer" % safe_label(o["envelope"].get("schema_version"))) + # Computed here and used twice: a rejection's damage class depends on whether the line claimed + # to be one of OUR records at all. + claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o + or "run_id" in o or "carry_bytes" in o) if o.get("record_type") not in (None, "carry_run"): - return False, "unknown record_type" + # A record_type this reader cannot name, on a line that ALSO carries our fields, is a + # corrupted record of ours and therefore a loss. On a line that carries none of them it is + # another tool's entry in a shared history, and reading it as lost evidence would cut a + # history nothing happened to — the same distinction `no shares` already makes one check + # further down. (§6 of the trusted-boundary task: a migration must not read as damage.) + return False, ("unknown record_type" if claims_ours else "not our record_type") sv = o.get("schema_version", 0) if not isinstance(sv, int) or isinstance(sv, bool) or sv not in SCHEMA_SUPPORTED: # The reason string is public: it becomes a key of `history.rejected`. A rejected line's # own content therefore goes through safe_label() before it can be echoed there. return False, f"unsupported schema_version {safe_label(sv)}" sh = o.get("shares") - claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o - or "run_id" in o or "carry_bytes" in o) if not isinstance(sh, dict): # Two different lines, and the damage classification depends on which one this is: a line # that claims to be one of OUR records is a corrupted record (a loss), a line that claims @@ -471,6 +479,7 @@ def load_history(path): and is kept as it is.""" recs, rejected = [], collections.Counter() cuts_file, cuts_scope = 0, collections.Counter() + opened = set() if not path or not os.path.exists(path): return recs, rejected, 0, {"file_global": 0, "scope_local": {}} lines = 0 @@ -517,7 +526,21 @@ def load_history(path): # The record belongs to the epoch it closes, never to the one it opens: it is # stamped first and the counter moves after it. A population that recovers is # not founded on the observation that reported the loss. - cuts_scope[scope_of(o)] += 1 + # ONE boundary per logical run, though. A retry reports the SAME loss its twin + # already reported, and this reader's own rule calls a deduplicated retry + # bookkeeping rather than a loss; letting a late copy open a second boundary + # let `6 healthy records + one more copy of X` erase a recovered epoch, on + # repeat, forever — the fail-stuck shape this repair exists to remove, rebuilt + # out of its own recovery mechanism. (cross-family review of the trust repair) + # A record with no run_id cannot be shown to be a retry and stays its own + # observation; across a file-global loss the identity differs, because there + # the reader cannot tell whether a repeated id is the same run at all. + rid = o.get("run_id") + ident = ((scope_of(o), cuts_file, rid) if isinstance(rid, str) and rid + else None) + if ident is None or ident not in opened: + opened.add(ident) + cuts_scope[scope_of(o)] += 1 continue rejected[why] += 1 # EVERY loss a rejected line represents is file-global. A record that failed validation From 71016ee0e931c36ad30df6edf725d68dbece9054 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 09:39:54 +0000 Subject: [PATCH 23/26] fix(optimizer): scope_id is an attribution authority, so it is checked before acceptance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round 2 of author review returned NEEDS-FIX with a HIGH that goes to the root of the trust model's own rationale. §6.2 justified the one trusted scope-local boundary by saying "the record passed structural validation and its scope_id is the same field every accepted record publishes". The first half was true. The second was an assumption: valid_record() never checked scope_id at all, and scope_of() is str(r.get("scope_id") or "default"). Reproduced before it was believed: 6 clean rev · one valid DEGRADED record with scope_id = ["rev"] · 6 clean rev CANDIDATE / exit 10 / one candidate file records_in_epoch=12, comparable=12 damage {"file_global": 0, "scope_local": {"['rev']": 1}} candidate names rid-rev-0..5 AND rid-rev-20..25 — both sides of the loss That is the blocked B-UTF8 defect rebuilt through the one door this repair opened: attribution taken from a field nobody had checked, a cut landing on a population that does not exist, the real one crossing the loss. The rule is the producer's own contract, not a new invention. tools/carry.py writes exactly one shape — `str(scope_id or "default")[:64]` — so a value of another type, or a string longer than that cap, was not written by it. Such a record is refused with a STATIC reason (its content must never be echoed) and is a file-global loss like any other unattributable one. Measured after the fix, the reviewer's fixture gives file_global=1, an active epoch of six, scope.known == ["rev"], and a candidate naming only rid-rev-20..25. A control byte inside a label stays ACCEPTED, deliberately: the producer's cap truncates length but does not strip control characters, such a label is already published through scope.known on every head of this branch, and calling it corruption would invent damage where the file is intact. Stated in docs/V142_COUNTEREXAMPLES.md §6.11 rather than hidden. Frozen as TSCOPE_08 with four forged shapes and five producer-writable controls; mutant M_SCOPE_LABEL_UNCHECKED carries its own positive control. 17 suites, 1563 assertions, 0 failures. 69 mutants, 0 survivors. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- docs/V142_COUNTEREXAMPLES.md | 39 ++++++++++++++++++++++++++++++++ tests/test_evidence_integrity.py | 24 ++++++++++++++++++++ tests/test_mutation.py | 27 ++++++++++++++++++++++ tools/optimize.py | 14 ++++++++++++ 5 files changed, 105 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 9f57d8f..b847232 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1553 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1563 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index 0a55af2..b0efbad 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -432,3 +432,42 @@ recovery mechanism.** A boundary is now opened at most once per `(scope, file-gl A record with no `run_id` cannot be shown to be a retry and stays its own observation; across a file-global loss the identity differs, because there the reader cannot tell whether a repeated id is the same run at all. + +### 6.11 The second finding of that review: `scope_id` was made an authority without a contract + +Round 2 returned `NEEDS-FIX` with a HIGH that goes to the root of §6.2's own rationale. The trust +model says a scope-local boundary is safe because "the record passed structural validation and its +`scope_id` is the same field every accepted record publishes". The first half was true; the second +was an assumption. `valid_record()` never checked `scope_id` at all, and `scope_of()` is +`str(r.get("scope_id") or "default")` — so a record that passes validation with +`scope_id = ["rev"]` becomes the scope `"['rev']"`. + +Reproduced before it was believed, on `189db57`: + +```text +6 clean rev · one valid DEGRADED record with scope_id = ["rev"] · 6 clean rev + status CANDIDATE, exit 10, one candidate file + records_in_epoch=12, comparable=12 + damage {"file_global": 0, "scope_local": {"['rev']": 1}} + candidate names rid-rev-0..5 AND rid-rev-20..25 — both sides of the loss +``` + +That is the blocked B-UTF8 defect rebuilt through the one door this repair opened: attribution taken +from a field nobody had checked, a cut landing on a population that does not exist, and the real +population crossing the loss. + +**The fix is the producer's own contract, not a new invention.** `tools/carry.py` writes exactly one +shape: `"scope_id": str(scope_id or "default")[:64]`. So a value that is not a string, or a string +longer than 64 characters, was not written by it — the record is corrupt and is refused with a +STATIC reason (its content must never be echoed), which makes it a file-global loss like any other +unattributable one. + +| case | `scope_id` | expectation | +|---|---|---| +| **TSCOPE_08** | `["rev"]` · `5` · `{"s": 1}` · a 200-character string | file-global ×1, `scope_local == {}`, active epoch 6, `scope.known == ["rev"]`, candidate names only the post-loss run ids | +| **TSCOPE_08 controls** | absent · `"rev"` · `""` · exactly 64 characters · a label holding a control byte | still a record — the producer can write all of these | + +The control byte stays accepted deliberately: the producer's cap truncates length but does not strip +control characters, such a label is already published through `scope.known` on every head of this +branch, and calling it corruption would invent damage where the file is intact. That is the residual +this repair states rather than hides. diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index f3896ea..2d8dc70 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -226,6 +226,30 @@ def run_paths(rows, extra=()): shape(rc, j, n) + (j["history"]["records"], j["scope"]["known"]), ("CANDIDATE", 10, 1, 6, 6, 1, {}, 12, ["rev"])) + # TSCOPE_08: `scope_id` is an ATTRIBUTION AUTHORITY, so it is checked before a record is + # accepted. The producer writes exactly one shape — `str(scope_id or "default")[:64]` — and a + # value of another type let `["rev"]` become the scope `"['rev']"`: the cut landed on a + # population nobody has while the real `rev` records kept crossing the loss. (docs §6.11) + print("\nTSCOPE_08 - a scope label the producer could not have written") + for label, sid in (("a list", ["rev"]), ("an integer", 5), ("an object", {"s": 1}), + ("a label longer than the producer's own cap", "x" * 200)): + forged = dict(rec(6, 48.0, scope="rev", unreadable=2), run_id="LOSS-X") + forged["scope_id"] = sid + rc, j, n = run(ser(6) + [json.dumps(forged)] + later()) + check(f"TSCOPE_08 {label} is a loss nobody can attribute", + shape(rc, j, n) + (j["scope"]["known"],), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, ["rev"])) + for label, sid in (("absent", None), ("a plain label", "rev"), ("the empty string", ""), + ("exactly 64 characters", "y" * 64), + ("a control byte the producer can write", "a\u0001b")): + ok = dict(rec(6, 48.0, scope="rev"), run_id="OK-X") + if sid is None: + ok.pop("scope_id") + else: + ok["scope_id"] = sid + check(f"TSCOPE_08 control: {label} is still a record", + optimize.valid_record(ok), (True, "")) + # ---------------------------------------------------------------- privacy (§35) print("\nTPRIV - a rejected line's own content is never echoed into public output") secret = "sk-synthetic-NOTAREALKEY-0123456789" diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 5eaaba6..f97292e 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -1039,6 +1039,33 @@ def degraded(scope): assert optimize.damage_summary(ep2)["file_global"] == 1, optimize.damage_summary(ep2) """), + ("M_SCOPE_LABEL_UNCHECKED: scope_id adalah otoritas atribusi, jadi diperiksa sebelum diterima", + [("optimize.py", + " if sid is not None and (not isinstance(sid, str) or len(sid) > 64):", + " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + forged = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, run_id="LOSS-X") + forged["unreadable"] = 2 + forged["scope_id"] = ["rev"] + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="rev") for i in range(6)] + p = w(os.path.join(D, "forged_scope.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(forged)] + + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["file_global"] == 1, d + assert d["scope_local"] == {}, d + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("populasi rev menyeberangi loss", len(cur)) + assert sorted({optimize.scope_of(r) for r in recs}) == ["rev"], "scope hantu terbit" + # kontrol: label yang MEMANG ditulis produser tetap diterima + fine = rec(100, {"Bash": 50.0, "Read": 50.0}, scope="y" * 64) + assert optimize.valid_record(fine) == (True, ""), optimize.valid_record(fine) + """), + ("M_REJECTED_VALUE_ECHOED: isi baris yang ditolak tak boleh masuk output publik", [("optimize.py", ' if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)):\n' ' return repr(v)\n return "of type " + type(v).__name__', diff --git a/tools/optimize.py b/tools/optimize.py index dafd765..13869a0 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -465,6 +465,20 @@ def valid_record(o): q = o.get("evidence_quality") if q is not None and q not in ("COMPLETE", "PARTIAL", "INVALID", "EMPTY"): return False, "unknown evidence_quality" + sid = o.get("scope_id") + if sid is not None and (not isinstance(sid, str) or len(sid) > 64): + # `scope_id` is an ATTRIBUTION AUTHORITY: a validated record's own scope decides which + # population a loss cut applies to, so the field has to be checked before the record is + # accepted rather than after. The producer writes exactly one shape — + # `str(scope_id or "default")[:64]` (tools/carry.py) — so another type or another length + # was not written by it. Accepting one let `["rev"]` become the scope `"['rev']"`: the cut + # landed on a population nobody has and the real `rev` records kept crossing the loss, + # which is the blocked defect rebuilt through the one door this repair opened. + # (cross-family author review of the trust repair, round 2) + # The reason is STATIC on purpose: it is a rejected line's own content and must not be + # echoed. A control character inside a label the PRODUCER wrote is still accepted — it is + # already published through `scope.known` and calling it corruption would invent damage. + return False, "scope_id is not a label this producer writes" return True, "" From c0749e8037662fd9039be6dbe6e943cfca794881 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 10:49:41 +0000 Subject: [PATCH 24/26] docs(v1.4.2): freeze the run-id conflict matrix before implementing it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The author-adversarial review of 71016ee found that two materially different valid observations sharing one run_id were collapsed into a single retry, so the second loss never opened a boundary and a candidate was written whose evidence spanned it. Reproduced on that head first, with four controls. Frozen here, before any code: run_id is an identity CLAIM, not proof of semantic equality. Two records are one retry only when their identity AND their canonical persisted observation match; a same-identity pair whose observations differ is a RUN_ID_CONFLICT — an integrity event that opens a recoverable scope-local boundary at the later record's own position and is never reported as a retry. Section 7.3 argues the complete-versus-complete row rather than inferring it, and 7.4 records the limit no reading of this format can close: identical content under a reused run_id stays indistinguishable from one identical retry. Co-Authored-By: Claude Opus 5 --- docs/V142_COUNTEREXAMPLES.md | 102 +++++++++++++++++++++++++++++++++++ 1 file changed, 102 insertions(+) diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index b0efbad..1068789 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -471,3 +471,105 @@ The control byte stays accepted deliberately: the producer's cap truncates lengt control characters, such a label is already published through `scope.known` on every head of this branch, and calling it corruption would invent damage where the file is intact. That is the residual this repair states rather than hides. + +## 7. Run-id conflict: identity is a claim, not proof (frozen before the repair) + +The author-adversarial review of `71016ee` found that two **materially different** valid observations +carrying the same `run_id` were collapsed into one retry, so the second loss never opened a +boundary. Reproduced on that head before anything was designed for it: + +```text +DEGRADED X1(run_id=SHARED, ts 0, share 30, unreadable 2, sessions 40, turns 1000, scanned 80) +6 healthy +DEGRADED X2(run_id=SHARED, ts 40, share 66, unreadable 9, sessions 91, turns 2400, scanned 150) +6 healthy + + CANDIDATE / exit 10 / 1 candidate file + records_in_epoch 12, comparable 12, scope_local boundaries 1 (for TWO reported losses) + "duplicate run_id (retry)" = 1 + candidate evidence = 6 records from before X2 and 6 from after it +``` + +Nine persisted fields differ between X1 and X2 (`ts`, `shares`, `sessions`, `turns`, `carry_bytes`, +`scanned`, `unreadable`, `runtimes`, `models`). Both pass `valid_record()`; both are independently +`DEGRADED`. Controls on the same head: a different `run_id` gives two boundaries and an active epoch +of six; an identical duplicate, a key-order-only difference and a whitespace-only difference all give +one boundary and an epoch of twelve; a file-global loss between the copies gives two scope-local +boundaries plus one file-global. + +### 7.1 The rule this section freezes + +> **`run_id` is an identity CLAIM, not proof of semantic equality.** +> Two records may be treated as one retry only when their identity AND their persisted observation +> are equivalent. + +`SAME_RUN_ID_ONLY_IS_RETRY=NO`. Three cases, and one classification decides all of them — boundary +opening, deduplication, the quality floor and the diagnostics read the same verdict: + +* **Case A — TRUE_RETRY.** Same scope, same file-global epoch, same `run_id`, **and the same + canonical persisted observation.** Deduplicated exactly as before: the later copy is dropped, it + is counted, its quality still travels to the survivor, and it does **not** open a second boundary. +* **Case B — RUN_ID_CONFLICT.** Same identity, **different** canonical persisted observation. This + is an observable integrity event, not bookkeeping. The later record is **kept** (it is a valid + observation), it is **not** reported as a retry, and it opens a recoverable **scope-local** + boundary at its own physical position. It belongs to the epoch it closes. +* **Case C — across a file-global boundary.** Unchanged: the reader cannot establish continuity + across an unattributable loss, so the later occurrence is a fresh identity. No retry, no conflict, + no imported quality floor. + +Equivalence is decided on the **whole persisted record**, not a hand-picked subset — the defect being +repaired exists precisely because the previous identity was too weak. The comparison is a digest of +the parsed object with keys sorted, so JSON key order and whitespace cannot make two semantically +identical observations differ, and the reader's own private annotations (`_history_epoch`, +`_evidence_quality_floor`) are excluded, because reading a file must not change what a record is. A +difference in any other persisted field — including one this reader does not know — conservatively +makes the observations non-equivalent: that is fail-closed, and epochs recover. + +### 7.2 The frozen matrix + +Pieces: `A` = 6 healthy COMPLETE records, `B` = 6 more, all scope `s1`, one file-global epoch unless +stated. `scope_local` is the total count of scope-local boundaries. + +| case | fixture | status | exit | files | epoch | comparable | scope_local | file_global | retries | conflicts | history.quality | +|---|---|---|---|---|---|---|---|---|---|---|---| +| **RID_01** | `DEG X1 · A · DEG X1 (identical) · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_02** | `DEG X1 · A · DEG X2 (different)` | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | 2 | 0 | 0 | 1 | `EMPTY` | +| **RID_03** | `DEG X1 · A · DEG X2 · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_04** | `CLEAN X · A · DEG X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_05** | `DEG X · A · CLEAN X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_06** | `COMPLETE X · A · COMPLETE X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_07** | `PARTIAL X · A · PARTIAL X' (different) · B`, with and without `--accept-partial` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_08** | `DEG X scope a · DEG X scope b · 6×a · 6×b`, analysing `a` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 0 | `COMPLETE` | +| **RID_09** | `DEG X1 · A · torn line · DEG X1 (identical) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 1 | 0 | 0 | `COMPLETE` | +| **RID_10** | `DEG X1 · A · DEG X1 with the keys in reverse order · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_11** | `DEG X1 · A · DEG X1 with different separators · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_12** | `DEG X1 · A · DEG X1 with another `workload_class` · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_13** | `DEG X1 · A · DEG X1 with other `shares` · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_14** | `DEG X1 · A · DEG X1 with another loss counter · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_15** | `DEG X1 · 3 healthy · DEG X2 · 3 healthy · DEG X3 · B` (all three different) | `CANDIDATE` | 10 | 1 | 6 | 6 | 3 | 0 | 0 | 2 | `COMPLETE` | + +`RID_03`, `RID_04`, `RID_05`, `RID_06`, `RID_07`, `RID_12`–`RID_15` additionally require the +candidate's `evidence_run_ids` to name **only** records after the last boundary. + +### 7.3 Complete-versus-complete is a boundary, and why + +`RID_06` is the row that needed an argument rather than an inference. Neither record reports an +acquisition loss, so nothing was lost — but the file states two different things under one identity, +and the reader has no way to tell which one the population it is about to compare actually belongs +to. A trend drawn across that point is drawn over a file whose identity discipline has already +failed. The choice is therefore a **recoverable scope-local boundary**: it costs the pre-conflict +records, it costs nothing permanently, and the alternative — silently keeping one of the two and +calling it a retry — is the exact statement the repair exists to stop making. The diagnostic stays +honest either way: the conflict is counted as a conflict, never as a retry. + +### 7.4 The limit this repair cannot close + +`IDENTICAL_REUSED_RUNID_LIMITATION=YES`. Two physically distinct losses in one scope and one +file-global epoch that carry the same `run_id` **and identical persisted content** are +indistinguishable from one run retried identically. The flat legacy record has no other identity to +read. Line position is deliberately NOT used as identity: doing so would make every true retry open +a fresh boundary and recreate the fail-stuck behaviour the previous round removed. + +### 7.5 Deviations from this frozen section + +None. From b0624a7f1ad0e860917bd9978b70832f0e7f531d Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 10:50:50 +0000 Subject: [PATCH 25/26] =?UTF-8?q?test(v1.4.2):=20RED=20=E2=80=94=20RID=5F0?= =?UTF-8?q?1..RID=5F15,=20the=20run-id=20conflict=20matrix?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fourteen failures on 71016ee, one per row of docs/V142_COUNTEREXAMPLES.md §7. The rows that already pass are the ones the repair must NOT change: an exact duplicate (RID_01), a key-order-only difference (RID_10), a whitespace-only difference (RID_11), the same id in another scope (RID_08) and the same id across a file-global loss (RID_09). They are the positive controls against reintroducing the fail-stuck behaviour the previous round removed. The rows that fail are the defect: a materially different observation under one run_id is reported as "duplicate run_id (retry)", opens no boundary, and lets a candidate rest on evidence from both sides of the second loss. The last three checks pin the identity itself: a reader annotation (_history_epoch, _evidence_quality_floor) may never change what a record is, while any persisted field — including one this reader does not know — must. optimize.observation_digest does not exist yet, which is the point. Co-Authored-By: Claude Opus 5 --- tests/test_evidence_integrity.py | 123 +++++++++++++++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index 2d8dc70..a526186 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -440,6 +440,128 @@ def run_paths(rows, extra=()): (status, code, files, epoch, 1, {})) +def run_id_conflicts(): + """docs/V142_COUNTEREXAMPLES.md §7 — run_id is an identity CLAIM, not proof of equality. + + Two materially different valid observations sharing one run_id were collapsed into a single + retry on 71016ee, so the second loss never opened a boundary and a candidate was written whose + evidence spanned it. Every row below was frozen before the repair existed and is RED on that + head.""" + d = tempfile.mkdtemp(prefix="sw-142-rid-") + + def A(n, first, base, scope="s1", **kw): + return [json.dumps(rec(first + i, base + 3.0 * i, scope=scope, **kw)) for i in range(n)] + + def X(run_id="SHARED", **kw): + """One persisted observation under a chosen identity.""" + return dict(rec(0, 30.0, scope="s1", **kw), run_id=run_id) + + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 4' + MID, TAIL = A(6, 10, 36.0), A(6, 50, 72.0) + X1 = X(unreadable=2) + X2 = dict(X(unreadable=9), ts=TS0 + 40 * 604800, sessions=91, turns=2400, + carry_bytes=3 * 10 ** 7, scanned=150, + shares={"Bash": 66.0, "Read": 34.0}) + X3 = dict(X2, ts=TS0 + 41 * 604800, unreadable=4, sessions=55, + shares={"Bash": 70.0, "Read": 30.0}) + + def rid(label, rows, status, code, files, epoch, comparable, scope_local, file_global, + retries, conflicts, quality, extra=(), post_only=None): + rc, j, n = run(rows, extra=extra) + h, sc = j["history"], j["scope"] + dm = h["damage"] + got = (j["status"], rc, n, sc["records_in_epoch"], h["comparable"], + sum(dm["scope_local"].values()), dm["file_global"], + (h["rejected"] or {}).get("duplicate run_id (retry)", 0), + h.get("run_id_conflicts", 0), h["quality"]) + check(label, got, (status, code, files, epoch, comparable, scope_local, file_global, + retries, conflicts, quality)) + + rid("RID_01 an exact duplicate is one retry and one boundary", + [json.dumps(X1)] + MID + [json.dumps(dict(X1))] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + rid("RID_02 a materially different copy under one id is a conflict", + [json.dumps(X1)] + MID + [json.dumps(X2)], + "INSUFFICIENT_DATA", 20, 0, 0, 0, 2, 0, 0, 1, "EMPTY") + rid("RID_03 ...and the population after it recovers on its own", + [json.dumps(X1)] + MID + [json.dumps(X2)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_04 a clean record and a degraded one under one id conflict", + [json.dumps(X(unreadable=0))] + MID + [json.dumps(X2)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE") + rid("RID_05 ...and so do a degraded one and a clean one", + [json.dumps(X1)] + MID + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_06 two different COMPLETE records under one id still contradict", + [json.dumps(X(unreadable=0))] + MID + + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE") + partial_a = json.dumps(X(quality="PARTIAL", skipped=4)) + partial_b = json.dumps(dict(X2, unreadable=0, evidence_quality="PARTIAL", skipped_by_limit=9)) + for flag in ((), ("--accept-partial",)): + rid(f"RID_07 a bounded pair under one id conflicts too {list(flag)}", + [partial_a] + MID + [partial_b] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE", extra=flag) + rid("RID_08 the same id in another scope is another observation", + [json.dumps(dict(X1, scope_id="a")), json.dumps(dict(X1, scope_id="b"))] + + A(6, 20, 48.0, scope="a") + A(6, 20, 48.0, scope="b"), + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 0, "COMPLETE", extra=["--scope-id", "a"]) + rid("RID_09 a file-global loss makes the later copy a fresh identity", + [json.dumps(X1)] + MID + [TORN, json.dumps(dict(X1))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 1, 0, 0, "COMPLETE") + rid("RID_10 key order alone is not a different observation", + [json.dumps(X1)] + MID + + [json.dumps({k: X1[k] for k in reversed(list(X1))})] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + rid("RID_11 whitespace alone is not a different observation", + [json.dumps(X1)] + MID + [json.dumps(X1, separators=(" , ", " : "))] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + for label, other in (("RID_12 another workload class", dict(X1, workload_class="other")), + ("RID_13 other shares", dict(X1, shares={"Bash": 31.0, "Read": 69.0})), + ("RID_14 another loss counter", dict(X1, unreadable=5))): + rid(label + " is a conflict", [json.dumps(X1)] + MID + [json.dumps(other)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_15 three different copies under one id are three boundaries", + [json.dumps(X1)] + A(3, 10, 36.0) + [json.dumps(X2)] + A(3, 20, 48.0) + + [json.dumps(X3)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 3, 0, 0, 2, "COMPLETE") + + # the candidate may rest only on the population after the last boundary + for label, rows in (("RID_03", [json.dumps(X1)] + MID + [json.dumps(X2)] + TAIL), + ("RID_06", [json.dumps(X(unreadable=0))] + MID + + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL), + ("RID_15", [json.dumps(X1)] + A(3, 10, 36.0) + [json.dumps(X2)] + + A(3, 20, 48.0) + [json.dumps(X3)] + TAIL)): + dd = tempfile.mkdtemp(dir=d) + hist = write(os.path.join(dd, "history.jsonl"), rows) + out = os.path.join(dd, "cand") + subprocess.run([sys.executable, OPT, "--history", hist, "--ledger", + os.path.join(dd, "none.jsonl"), "--emit-candidate", out, "--json", + "--strict-exit", "--scan"], capture_output=True, text=True, timeout=300) + body = "".join(open(os.path.join(b, f), encoding="utf-8").read() + for b, _s, fs in os.walk(out) for f in fs if f != ".optimize.lock") + names = sorted(x.strip() for line in body.splitlines() + if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check(f"{label} the candidate names only the recovered population", + names, sorted("rs1%d" % i for i in range(50, 56))) + + # §18: a reader annotation may never change what a record IS + a = dict(X1) + b = dict(X1) + a[optimize.EPOCH_KEY] = (3, 7) + b[optimize.QUALITY_FLOOR] = "DEGRADED" + check("a private annotation does not change the observation", + (optimize.observation_digest(a), optimize.observation_digest(b)), + (optimize.observation_digest(dict(X1)), optimize.observation_digest(dict(X1)))) + check("but a persisted field does", + optimize.observation_digest(dict(X1, ts=X1["ts"] + 1)) + != optimize.observation_digest(X1), True) + check("an extra persisted field this reader does not know makes them differ", + optimize.observation_digest(dict(X1, future_field=1)) + != optimize.observation_digest(X1), True) + + def main(): d = tempfile.mkdtemp(prefix="sw-142-fx-") good_rows = [rec(i, 30.0 + 3.0 * i) for i in range(6)] @@ -954,6 +1076,7 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, (2, [("a", 1), ("b", 1)])) trusted_boundaries() + run_id_conflicts() print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) From 6ef998869e3c44229b45eb3a130b1f3c0c8679f9 Mon Sep 17 00:00:00 2001 From: Peter Jackson Date: Tue, 22 Sep 2026 11:04:34 +0000 Subject: [PATCH 26/26] fix(optimizer): a run_id is an identity claim, not proof of semantic equality MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The author-adversarial review of 71016ee found the last collapse this reader still made: two materially different valid observations carrying one run_id were treated as a single retry, so the second loss never opened a boundary. Reproduced first, with four controls: DEGRADED X1(ts 0, share 30, unreadable 2, sessions 40, turns 1000, scanned 80) 6 healthy DEGRADED X2(ts 40, share 66, unreadable 9, sessions 91, turns 2400, scanned 150) 6 healthy [one scope, one file-global epoch] CANDIDATE / exit 10 / 1 candidate file records_in_epoch 12, comparable 12, scope_local boundaries 1 for TWO losses "duplicate run_id (retry)" = 1 candidate evidence = 6 records from before X2 and 6 from after it Nine persisted fields differ; both records pass validation and are independently DEGRADED. The controls: a different run_id already gave two boundaries and an epoch of six, an identical duplicate gave one and twelve, and a file-global loss between the copies gave two plus one. Two records are the same run only when their identity AND their persisted observation match. Equivalence is a digest of the WHOLE parsed record with keys sorted — a hand-picked subset would be the same mistake in a new spelling — so key order and whitespace cannot make identical observations differ, a persisted field this reader does not know still can, and the reader's own annotations (_history_epoch, _evidence_quality_floor) are excluded, because reading a file must not change what a record is. A same-identity pair whose observations differ is a RUN_ID_CONFLICT: the later record is kept, it is never reported as a retry, and it opens a recoverable scope-local boundary at its own physical position, belonging to the epoch it closes. Complete-versus-complete counts too, and §7.3 of the counterexample document argues that row rather than inferring it: nothing was lost, but the file states two things under one identity and a trend drawn across that point is drawn over a file whose identity discipline has already failed. One classification drives the boundary, the deduplication, the quality floor and the diagnostics. The separate second pass is gone: a boundary layer and a dedup layer holding two notions of identity is exactly how they came to disagree about one pair. Across a file-global loss nothing changes — the reader cannot establish continuity there, so the later copy is a fresh identity. history.run_id_conflicts is a bounded integer, deliberately outside history.rejected (a conflicting record is accepted, not a line that failed to become one) and deliberately not a map: 600 distinct conflicting ids produce the integer 600 and zero new keys. output_schema_version stays 2. Five earlier cases used one run_id for two different observations and asserted a retry. Each is replaced rather than quietly re-run, with the reason recorded in §7.6 before the edit, and each keeps the property it protected — now enforced by a boundary instead of by a quality floor, which is strictly stronger. A consequence stated rather than hidden: a true retry now has the same quality by construction, so QUALITY_FLOOR is a guard rather than a live path. §7.4 records the limit no reading of this format can close: identical persisted content under a reused run_id stays indistinguishable from one identical retry. Line position is deliberately not used as identity — that would make every true retry open a fresh boundary and rebuild the fail-stuck behaviour just removed. 17 suites, 1601 assertions, 0 failures. 77 mutants, 0 survivors. Co-Authored-By: Claude Opus 5 --- README.md | 2 +- docs/V142_COUNTEREXAMPLES.md | 52 ++++++++ tests/test_evidence_integrity.py | 69 +++++++---- tests/test_mutation.py | 207 ++++++++++++++++++++++++++----- tests/test_optimize.py | 21 +++- tools/optimize.py | 170 ++++++++++++++++--------- 6 files changed, 408 insertions(+), 113 deletions(-) diff --git a/README.md b/README.md index b847232..8d8a040 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1563 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1601 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md index 1068789..d562683 100644 --- a/docs/V142_COUNTEREXAMPLES.md +++ b/docs/V142_COUNTEREXAMPLES.md @@ -573,3 +573,55 @@ a fresh boundary and recreate the fail-stuck behaviour the previous round remove ### 7.5 Deviations from this frozen section None. + +### 7.6 Frozen-decision amendments the identity law forces (§21 order: recorded before editing) + +Five existing cases were written under the rule this repair overturns — they use **one `run_id` for +two different observations** and assert that the pair is a retry. Under §7.1 such a pair is a +conflict, so each is replaced rather than quietly re-run, and each keeps the property it was +protecting. + +```text +R142_07 "a retry cannot upgrade what its own run_id saw" + was: run_id "dup" appears twice, the second claiming PARTIAL with skipped_by_limit=7, + and the survivor is lowered to PARTIAL through QUALITY_FLOOR + -> PARTIAL_EVIDENCE / 40 / 0 files + now: the two copies differ, so they are a CONFLICT: the later one cuts and nothing + before it can promote -> INSUFFICIENT_DATA / 20 / 0 files + The property is unchanged and the guarantee is STRICTER: no repeated id can upgrade what + it saw. It is now enforced by a boundary rather than by a floor. The case gains its + companion: two IDENTICAL copies are a true retry, counted, no boundary, quality untouched. + +B1_13b "a retry inside one epoch still lowers the survivor" -> same cause, same replacement. +B1_13e "three copies inside one epoch: the worst of them survives" -> the three copies differ, + so they are two conflicts; RID_15 is the frozen row for that shape. +VD_10 "clean first, the retry reports the loss" frozen in §6.4 on the assumption that one +VD_11 "the loss first, the retry reports clean" run_id means one run. That assumption is + exactly what this repair removes. Both pairs differ materially, so each later copy is a + conflict that cuts at its own position; the later population still recovers, and neither + copy can launder anything. VD_11 now shows two scope-local boundaries instead of one. +``` + +**A consequence worth stating rather than hiding:** under the new law a TRUE_RETRY has, by +construction, the same persisted observation and therefore the same `record_quality`, so +`QUALITY_FLOOR` can no longer change anything. It is kept as a guard, not as a live path, and the +anti-laundering property it used to carry is now carried by the conflict boundary. No test claims to +exercise a floor that cannot fire. + +### 7.7 What the machine output gains, and where it is NOT + +`history.run_id_conflicts` is a bounded integer beside `history.rejected`, and `output_schema_version` +stays **2** — v2 is unreleased and this is it evolving, not a second contract. + +It is deliberately **not** in `history.rejected`: that map counts lines that failed to become +records, and a conflicting record is accepted and kept. Reporting it there would be the same false +statement as calling it a retry. It is also not a map keyed by anything the file chose — six hundred +distinct conflicting ids produce the integer `600` and zero new keys, so a corrupt history cannot +grow the diagnostics. The human report names the count and what it means, never a value from either +record. + +`history.damage` is unchanged and its invariant still holds: +`boundaries == file_global + sum(scope_local)`. A conflict cut is a scope-local boundary like any +other, so a conflict that lands on a record which was ALSO degraded produces one boundary, not two — +a boundary is a position, not a tally of reasons — while `run_id_conflicts` keeps counting the +reasons separately. diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py index a526186..095fc97 100644 --- a/tests/test_evidence_integrity.py +++ b/tests/test_evidence_integrity.py @@ -379,15 +379,21 @@ def run_paths(rows, extra=()): print("\nVD_10/VD_11 - a retry may neither launder a loss nor poison the epoch after it") clean_x = json.dumps(dict(rec(5, 45.0, scope="a"), run_id="X")) deg_x = json.dumps(dict(rec(6, 48.0, scope="a", unreadable=2), run_id="X")) - for label, rows in (("VD_10 clean first, the retry reports the loss", - ser(5, 0, "a") + [clean_x, deg_x] + ser(6, 20, "a", base=48.0)), - ("VD_11 the loss first, the retry reports clean", - ser(5, 0, "a") + [deg_x, clean_x] + ser(6, 20, "a", base=48.0))): + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §7.6): both pairs differ materially, + # so under the identity law each later copy is a CONFLICT, not a retry. The population after it + # still recovers and neither copy can launder anything — the guarantee these rows exist for. + for label, rows, loc in (("VD_10 clean first, then something else under the same id", + ser(5, 0, "a") + [clean_x, deg_x] + ser(6, 20, "a", base=48.0), + {"a": 1}), + ("VD_11 the loss first, then something else under the same id", + ser(5, 0, "a") + [deg_x, clean_x] + ser(6, 20, "a", base=48.0), + {"a": 2})): rc, j, n = run(rows) check(label, (j["status"], rc, n, j["scope"]["records_in_epoch"], j["history"].get("quality"), - j["history"]["rejected"].get("duplicate run_id (retry)")) + dmg(j), - ("CANDIDATE", 10, 1, 6, "COMPLETE", 1, 0, {"a": 1})) + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0), + j["history"]["run_id_conflicts"]) + dmg(j), + ("CANDIDATE", 10, 1, 6, "COMPLETE", 0, 1, 0, loc)) # VD_12/VD_13: what a cross-family review of THIS repair found. A retry reports the same loss # its twin already reported, and a deduplicated retry is bookkeeping by this reader's own rule. @@ -696,22 +702,35 @@ def main(): # ---------------------------------------------------------------- a retry cannot launder print("\nR142_07 - a retry cannot upgrade what its own run_id saw") + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §7.6): the two copies below differ, + # so under the identity law they are a CONFLICT rather than a retry. The property is the same + # and the guarantee is stricter — no repeated id can upgrade what it saw — but it is now + # enforced by a boundary instead of by a quality floor. + same = [rec(i, 30.0 + 3.0 * i) for i in range(5)] + twice = dict(rec(5, 45.0), run_id="dup") + same += [dict(twice), dict(twice)] + hist_path = write(os.path.join(d, "same.jsonl"), same) + recs, rejected, _lines, ep = optimize.load_history(hist_path) + check("an identical copy is dropped as an observation", len(recs), 6) + check("and counted as a retry", rejected["duplicate run_id (retry)"], 1) + check("and it opens no boundary", optimize.damage_summary(ep)["boundaries"], 0) dup = [rec(i, 30.0 + 3.0 * i) for i in range(5)] - dup.append(rec(5, 45.0)) - dup[-1]["run_id"] = "dup" + dup.append(dict(twice)) worse = rec(5, 45.0, quality="PARTIAL", skipped=7) worse["run_id"] = "dup" dup.append(worse) hist_path = write(os.path.join(d, "dup.jsonl"), dup) - recs, rejected, _lines, _ep = optimize.load_history(hist_path) - check("the retry is dropped as an observation", len(recs), 6) - check("and counted", rejected["duplicate run_id (retry)"], 1) - keep, _dropped = optimize.comparable(recs) - check("but its quality travels with the record that survived", - optimize.history_quality(keep), "PARTIAL") - case("R142_07", dup, "PARTIAL_EVIDENCE", 40, 0, history_quality="PARTIAL") - case("R142_07 (--accept-partial adopts the bound)", dup, "CANDIDATE", 10, 1, - extra=["--accept-partial"]) + recs, rejected, _lines, ep = optimize.load_history(hist_path) + check("a copy that says something else is kept, not merged", len(recs), 7) + check("and it is NEVER called a retry", + rejected.get("duplicate run_id (retry)", 0), 0) + check("it is counted as an identity conflict", ep["run_id_conflicts"], 1) + check("and it cuts where it appears", optimize.damage_summary(ep)["boundaries"], 1) + check("so nothing before it can promote", + len(optimize.active_records(recs, ep)), 0) + case("R142_07", dup, "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") + case("R142_07 (--accept-partial is not an identity override)", dup, + "INSUFFICIENT_DATA", 20, 0, extra=["--accept-partial"]) check("a schema_version this reader cannot name is not a newer one", optimize.record_quality(dict(rec(0, 40.0), schema_version="2")), "UNKNOWN") check("nor is a float one", @@ -1004,8 +1023,11 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, json.dumps(dict(json.loads(series(1, 6)[0]), run_id="twin", evidence_quality="PARTIAL", skipped_by_limit=9))] rc, j, n = run(same_epoch) - check("B1_13b a retry inside one epoch still lowers the survivor", - (j["history"]["quality"], j["status"], n), ("PARTIAL", "PARTIAL_EVIDENCE", 0)) + # Frozen-decision amendment (§7.6): the twins differ, so they are an identity conflict. + check("B1_13b a copy that says something else cuts instead of lowering", + (j["history"]["quality"], j["status"], n, j["history"]["run_id_conflicts"], + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0)), + ("EMPTY", "INSUFFICIENT_DATA", 0, 1, 0)) # both line orders, because which copy comes first is exactly what a crash decides clean_then_worse = (series(5) + [json.dumps(dict(rec(5, 45.0), run_id="X"))] + [TORN] + [json.dumps(dict(rec(6, 45.0), run_id="X", evidence_quality="PARTIAL", @@ -1030,9 +1052,12 @@ def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, evidence_quality="PARTIAL", skipped_by_limit=4)), json.dumps(dict(rec(6, 48.0), run_id="S"))] rc, j, n = run(three_in_one) - check("B1_13e three copies inside one epoch: the worst of them survives", - (j["history"]["quality"], j["history"]["rejected"].get("duplicate run_id (retry)"), n), - ("PARTIAL", 2, 0)) + # Frozen-decision amendment (§7.6): three DIFFERENT copies under one id are two conflicts. + # RID_15 is the frozen row for the same shape with a population after it. + check("B1_13e three different copies under one id are two conflicts", + (j["history"]["quality"], j["history"]["run_id_conflicts"], + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0), n), + ("EMPTY", 2, 0, 0)) print("\nB1_14/B1_15 - what a cross-family review of the epoch repair found") # a finding that rests on the ledger alone may still promote after a loss — but nothing from diff --git a/tests/test_mutation.py b/tests/test_mutation.py index f97292e..96f9fa3 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -118,12 +118,16 @@ def mutant(pairs): """), ("identitas run: dua agen dgn metrik identik = dua pengamatan", - [("optimize.py", 'rid = r.get("run_id")', 'rid = json.dumps(r.get("shares"), sort_keys=True)')], + [("optimize.py", ' rid = rec.get("run_id")\n if not (isinstance(rid, str) and rid):\n return None', + ' rid = json.dumps(rec.get("shares"), sort_keys=True)')], """ p = w(os.path.join(D, "h.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}), rec(100, {"Bash": 50.0, "Read": 50.0})]) - recs, rej, _, _ = optimize.load_history(p) + recs, rej, _l, ep = optimize.load_history(p) assert len(recs) == 2, "populasi menyusut: dua agen dihitung satu" + # ...dan identitas yang dikarang dari metrik juga tak boleh melahirkan konflik identitas + assert ep["run_id_conflicts"] == 0, ep + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) """), ("fail-closed: bukti PARTIAL tak boleh melahirkan kandidat", @@ -673,12 +677,8 @@ def hist_of(n): ("M_DEDUP_LAUNDERS: retry dgn run_id sama tak boleh menaikkan kualitas yang bertahan", - [("optimize.py", """ kept = seen[rid] - worse = worst_quality([record_quality(kept), record_quality(r)]) - if worse != record_quality(kept): - kept[QUALITY_FLOOR] = worse - continue""", - " continue")], + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return TRUE_RETRY")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="d") for i in range(5)] @@ -687,10 +687,12 @@ def hist_of(n): bad["evidence_quality"] = "PARTIAL" bad["skipped_by_limit"] = 7 p = w(os.path.join(D, "dup.jsonl"), rows + [good, bad]) - recs, rej, _, _ = optimize.load_history(p) - keep, _d = optimize.comparable(recs) - q = optimize.history_quality(keep) - assert q == "PARTIAL", q + recs, rej, _l, ep = optimize.load_history(p) + # salinan yang MENGAKU sesuatu yang lain bukan retry: ia konflik identitas, dan ia memotong. + assert ep["run_id_conflicts"] == 1, ep + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + assert optimize.damage_summary(ep)["boundaries"] == 1, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 0, "populasi pra-konflik masih dipakai" """), @@ -880,7 +882,8 @@ def hist_of(n): """), ("M_SCOPE_DAMAGE_GLOBALIZED: kerusakan milik satu scope tak memotong scope lain", - [("optimize.py", " cuts_scope[scope_of(o)] += 1", " cuts_file += 1")], + [("optimize.py", ' return (e.get("file_global", 0), (e.get("scope_local") or {}).get(scope, 0))', + ' return (e.get("file_global", 0) + sum((e.get("scope_local") or {}).values()), 0)')], """ good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(6)] @@ -979,8 +982,8 @@ def hist_of(n): """), ("M_DEDUP_IGNORES_SCOPE: dua scope bisa punya nomor epoch yang sama", - [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid)', - " rid = (r.get(EPOCH_KEY, (0, 0))[0], rid)")], + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (rec.get(EPOCH_KEY, (0, 0))[0], rid)')], """ def degraded(scope): x = rec(100, {"Bash": 20.0, "Read": 80.0}, scope=scope) @@ -999,6 +1002,9 @@ def degraded(scope): cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "a"] q = optimize.history_quality(optimize.comparable(cur)[0]) assert q == "COMPLETE", (q, rej) + # run_id yang sama di scope LAIN bukan pengamatan yang sama: bukan retry, bukan konflik + assert ep["run_id_conflicts"] == 0, ep + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) """), # --------------------------- the trusted damage boundary (adversarial-review blocker B-UTF8) @@ -1168,7 +1174,8 @@ def degraded(scope): """), ("M_VALID_DEGRADED_POISONS_SCOPE_FOREVER: record sah yang kehilangan bukti membuka epoch", - [("optimize.py", " if degrades_scope(o):", " if False:")], + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if verdict == RUN_ID_CONFLICT:")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(6)] @@ -1203,14 +1210,9 @@ def degraded(scope): ("M_DEGRADED_RECORD_INCLUDED_POST_BOUNDARY: record yang melaporkan kehilangan ada di epoch LAMA", [("optimize.py", - " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])\n" - " recs.append(o)\n" - " if degrades_scope(o):", - " if degrades_scope(o):\n" - " cuts_scope[scope_of(o)] += 1\n" - " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])\n" - " recs.append(o)\n" - " if False:")], + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]\n" + " + (1 if degrades_scope(o) else 0))")], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(6)] @@ -1227,8 +1229,7 @@ def degraded(scope): """), ("M_DEGRADED_RETRY_CUTS_TWICE: satu boundary per run logis, bukan per salinan", - [("optimize.py", " if ident is None or ident not in opened:", - " if True:")], + [("optimize.py", " if verdict == TRUE_RETRY:", " if False:")], """ def deg(seq, bash, run_id): r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", @@ -1251,8 +1252,8 @@ def deg(seq, bash, run_id): """), ("M_DEGRADED_RETRY_NO_RECOVERY: retry tak boleh mencuci kehilangan yang dilaporkan kembarannya", - [("optimize.py", ' rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid)', - " rid = (scope_of(r), r.get(EPOCH_KEY), rid)")], + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (scope_of(rec), rec.get(EPOCH_KEY), rid)')], """ rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") for i in range(5)] @@ -1265,11 +1266,159 @@ def deg(seq, bash, run_id): [json.dumps(r) for r in rows] + [json.dumps(deg), json.dumps(clean_retry)] + [json.dumps(r) for r in later]) recs, rej, _l, ep = optimize.load_history(p) - assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + # amandemen §7.6: kedua salinan BERBEDA, jadi yang belakangan adalah konflik identitas, + # bukan retry. Yang dijaga tetap sama: salinan bersih itu tak boleh jadi bukti epoch pulih. + assert ep["run_id_conflicts"] == 1, dict(rej) + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) cur = optimize.active_records(recs, ep) assert len(cur) == 6, ("salinan bersih dari run yang sama ikut jadi bukti", len(cur)) + assert all(r.get("run_id") != "X" for r in cur), "salinan konflik masuk epoch pulih" + """), + + + # ------------------------------- run-id conflict (author-adversarial review of 71016ee) + ("M_RUNID_CONFLICT_TREATED_AS_RETRY: run_id sama bukan bukti pengamatan sama", + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return TRUE_RETRY")], + """ + def obs(seq, bash, run_id, unreadable=0, scanned=100): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + r["scanned"] = scanned + return json.dumps(r) + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "conflict.jsonl"), + [obs(0, 30.0, "X", unreadable=2)] + mid + + [obs(40, 66.0, "X", unreadable=9, scanned=150)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 1, dict(rej) + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 2}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "bukti menyeberangi konflik identitas" + """), + + ("M_RUNID_CONFLICT_SECOND_LOSS_SUPPRESSED: konflik identitas memotong di posisinya sendiri", + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if degrades_scope(o) and verdict != RUN_ID_CONFLICT:")], + """ + def obs(seq, bash, run_id, unreadable=0): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + return json.dumps(r) + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "second.jsonl"), + [obs(0, 30.0, "X", unreadable=2)] + mid + [obs(40, 66.0, "X", unreadable=9)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 6, "kehilangan kedua tak dipotong" + """), + + ("M_EXACT_RETRY_OPENS_SECOND_BOUNDARY: salinan PERSIS SAMA tetap satu kehilangan", + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return RUN_ID_CONFLICT")], + """ + deg = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "exact.jsonl"), [json.dumps(deg)] + mid + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 1}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "epoch pulih terhapus retry persis" """), + ("M_RUNID_CONFLICT_CROSSES_EPOCH: record yang mengontradiksi identitasnya ada di epoch LAMA", + [("optimize.py", " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]\n" + " + (1 if retry_identity(o) in under else 0))")], + """ + def obs(seq, bash, run_id, unreadable=0): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + return json.dumps(r) + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "crosses.jsonl"), + [obs(0, 30.0, "X"), obs(40, 66.0, "X", unreadable=9)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert all(r.get("run_id") != "X" for r in cur), "record konflik masuk epoch yang ia buka" + assert len(cur) == 6, len(cur) + """), + + ("M_RUNID_CONFLICT_CROSS_SCOPE_COLLIDES: run_id sama di scope lain bukan pengamatan yang sama", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (rec.get(EPOCH_KEY, (0, 0))[0], rid)')], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="b", run_id="X") + p = w(os.path.join(D, "xscope.jsonl"), [json.dumps(a), json.dumps(b)]) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + assert len(recs) == 2, len(recs) + """), + + ("M_RUNID_CONFLICT_GLOBAL_EPOCH_COLLIDES: kehilangan global memutus kontinuitas identitas", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (scope_of(rec), 0, rid)')], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="a", run_id="X") + p = w(os.path.join(D, "xglobal.jsonl"), + [json.dumps(a), chr(123) + '"torn":', json.dumps(b)]) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {}, optimize.damage_summary(ep) + assert len(recs) == 2, len(recs) + """), + + ("M_CLEAN_RUNID_CONFLICT_IGNORED: kontradiksi identitas tanpa kehilangan tetap integritas", + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if degrades_scope(o):")], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="a", run_id="X") + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "cleanconf.jsonl"), [json.dumps(a), json.dumps(b)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 1, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 1}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "bukti menyeberangi kontradiksi identitas" + """), + + ("M_PRIVATE_ANNOTATION_IN_FINGERPRINT: anotasi pembaca bukan bagian dari observasi", + [("optimize.py", " body = {k: v for k, v in rec.items() if k not in READER_PRIVATE}", + " body = dict(rec)")], + """ + deg = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "private.jsonl"), [json.dumps(deg)] + mid + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + assert ep["run_id_conflicts"] == 0, dict(rej) + """), ] diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 1b90ad8..f2cc383 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -80,12 +80,27 @@ def main(): check("baris tak terparse dihitung", rejected["unparseable line"], 1) check("run_id sama di SISI LAIN potongan bukan retry", rejected["duplicate run_id (retry)"], 0) check("hanya epoch terbaru yang aktif", len(optimize.active_records(recs, epochs)), 1) + # Amandemen keputusan-beku (docs/V142_COUNTEREXAMPLES.md §7.6): run_id adalah KLAIM identitas, + # bukan bukti kesamaan. Salinan yang PERSIS sama = retry; salinan yang berbeda (di sini `ts`) + # = konflik identitas yang memotong di posisinya sendiri, dan tak pernah disebut retry. p_same = write(os.path.join(d, "same_epoch.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32), - rec(101, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) - recs_same, rej_same, _l, _e = optimize.load_history(p_same) - check("run_id sama DI DALAM satu epoch tetap dihitung sekali", + rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) + recs_same, rej_same, _l, ep_same = optimize.load_history(p_same) + check("salinan PERSIS SAMA di dalam satu epoch dihitung sekali", (len(recs_same), rej_same["duplicate run_id (retry)"]), (1, 1)) + check("dan salinan persis sama tidak memotong apa pun", + optimize.damage_summary(ep_same)["boundaries"], 0) + p_conf = write(os.path.join(d, "conflict_epoch.jsonl"), + [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32), + rec(101, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) + recs_c, rej_c, _l, ep_c = optimize.load_history(p_conf) + check("salinan BERBEDA di bawah satu run_id adalah konflik, bukan retry", + (len(recs_c), rej_c.get("duplicate run_id (retry)", 0), ep_c["run_id_conflicts"]), + (2, 0, 1)) + check("dan konflik itu memotong di posisinya sendiri", + (optimize.damage_summary(ep_c)["boundaries"], len(optimize.active_records(recs_c, ep_c))), + (1, 0)) # Dua AGEN boleh menghasilkan metrik identik. Tanpa run_id itu dua pengamatan, bukan duplikat: # membuang salah satunya akan mengecilkan populasi yang sedang diukur. p2 = write(os.path.join(d, "twin.jsonl"), diff --git a/tools/optimize.py b/tools/optimize.py index 13869a0..c0dca1f 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -47,7 +47,10 @@ # read off a line that failed validation, so its keys are always # scopes that also appear in `scope.known` · `scope.records_in_scope` # counts the whole scope while `scope.records_in_epoch` counts what - # the current epoch contributes · a CANDIDATE outranks + # the current epoch contributes · `history.run_id_conflicts` + # counts accepted records that repeat a run_id with a DIFFERENT + # persisted observation, an integrity event that cuts like any + # other loss and is never reported as a retry · a CANDIDATE outranks # PARTIAL_EVIDENCE, because each finding is gated on its own # evidence before the run is summarised THRESHOLD_SCHEMA_VERSION = 1 # bump when any threshold below changes, with a reason and a test @@ -290,6 +293,56 @@ def degrades_scope(rec): return _derived_record_quality(rec, schema) == "DEGRADED" +# The reader's own annotations. They are computed while reading, never persisted, and they must +# not take part in deciding what a record IS: otherwise the act of reading a file would make an +# identical retry look like a different observation. +READER_PRIVATE = (EPOCH_KEY, QUALITY_FLOOR) + + +def observation_digest(rec): + """A deterministic digest of the PERSISTED observation -> str. + + The whole record participates, not a hand-picked subset: the defect this exists to close was + caused by an identity that was too weak, and a fingerprint built from a chosen handful would be + the same mistake in a new spelling. Keys are sorted, so JSON key order and whitespace cannot + make two semantically identical observations differ, and a persisted field this reader does not + know still changes the digest — conservatively non-equivalent, which is fail-closed and + recoverable. + """ + body = {k: v for k, v in rec.items() if k not in READER_PRIVATE} + return hashlib.sha256(json.dumps(body, sort_keys=True, separators=(",", ":"), + default=str).encode("utf-8")).hexdigest() + + +def retry_identity(rec): + """The identity two accepted records must SHARE before they can be the same run -> key | None. + + Scope, because two agents legitimately write at once. The FILE-GLOBAL half of the epoch, + because across an unattributable loss the reader cannot establish continuity at all and both + copies stand. Not the scope-local half: a trusted boundary leaves the file intact, so identity + survives it. A record without a run_id cannot be shown to be anyone's retry and is always its + own observation. + """ + rid = rec.get("run_id") + if not (isinstance(rid, str) and rid): + return None + return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid) + + +# `run_id` is an identity CLAIM, not proof of semantic equality. Two records are the same run only +# when their identity AND their persisted observation match; a same-identity pair whose observations +# differ is an integrity event, not bookkeeping. ONE classification, so the boundary, the +# deduplication, the quality floor and the diagnostics can never disagree about the same pair. +TRUE_RETRY, RUN_ID_CONFLICT, FIRST_SIGHTING = "retry", "conflict", "first" + + +def classify_repeat(prior_digest, digest): + """-> TRUE_RETRY | RUN_ID_CONFLICT | FIRST_SIGHTING""" + if prior_digest is None: + return FIRST_SIGHTING + return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT + + def active_epoch_key(scope, epochs): """The epoch a record of `scope` must carry to be part of the CURRENT analysis.""" e = epochs or {} @@ -493,9 +546,9 @@ def load_history(path): and is kept as it is.""" recs, rejected = [], collections.Counter() cuts_file, cuts_scope = 0, collections.Counter() - opened = set() + conflicts, under = 0, {} if not path or not os.path.exists(path): - return recs, rejected, 0, {"file_global": 0, "scope_local": {}} + return recs, rejected, 0, {"file_global": 0, "scope_local": {}, "run_id_conflicts": 0} lines = 0 try: # BYTES, decoded strictly per physical line. `errors="replace"` destroyed the evidence that @@ -508,7 +561,7 @@ def load_history(path): rejected[f"history unreadable: {safe_err(e)}"] += 1 # A file that would not open is a loss nobody can attribute: every scope starts a new epoch # with no records in it, which is the fail-closed answer. - return recs, rejected, 0, {"file_global": 1, "scope_local": {}} + return recs, rejected, 0, {"file_global": 1, "scope_local": {}, "run_id_conflicts": 0} with fh: for raw in fh: raw = raw.strip() @@ -535,26 +588,45 @@ def load_history(path): # never a timestamp, because the clock is exactly what a damaged history cannot # be trusted about. o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]) + # ONE classification for the pair, read here and nowhere else. The boundary, the + # deduplication, the quality floor and the diagnostics all consume this verdict: + # the defect that produced it was a boundary layer and a dedup layer holding two + # notions of identity, one of them too coarse. + ident = retry_identity(o) + prior = under.get(ident) if ident is not None else None + digest = observation_digest(o) if ident is not None else None + verdict = classify_repeat(prior["digest"] if prior else None, digest) + if verdict == TRUE_RETRY: + # The same run reporting the SAME observation again. Dropped as an observation, + # counted, and its quality still travels to the copy that survives — a retry + # cannot launder a bounded or lossy sweep into a complete one. It opens no + # boundary: a repeated copy of one loss is one loss, and letting a late copy + # cut again let `6 healthy records + one more copy of X` erase a recovered + # epoch, on repeat, forever. + rejected["duplicate run_id (retry)"] += 1 + kept = prior["rec"] + worse = worst_quality([record_quality(kept), record_quality(o)]) + if worse != record_quality(kept): + kept[QUALITY_FLOOR] = worse + continue recs.append(o) - if degrades_scope(o): - # The record belongs to the epoch it closes, never to the one it opens: it is - # stamped first and the counter moves after it. A population that recovers is - # not founded on the observation that reported the loss. - # ONE boundary per logical run, though. A retry reports the SAME loss its twin - # already reported, and this reader's own rule calls a deduplicated retry - # bookkeeping rather than a loss; letting a late copy open a second boundary - # let `6 healthy records + one more copy of X` erase a recovered epoch, on - # repeat, forever — the fail-stuck shape this repair exists to remove, rebuilt - # out of its own recovery mechanism. (cross-family review of the trust repair) - # A record with no run_id cannot be shown to be a retry and stays its own - # observation; across a file-global loss the identity differs, because there - # the reader cannot tell whether a repeated id is the same run at all. - rid = o.get("run_id") - ident = ((scope_of(o), cuts_file, rid) if isinstance(rid, str) and rid - else None) - if ident is None or ident not in opened: - opened.add(ident) - cuts_scope[scope_of(o)] += 1 + if ident is not None: + # The NEWEST observation is what this identity currently says, so a later exact + # copy is that one's retry rather than the original's conflict. + under[ident] = {"rec": o, "digest": digest} + if verdict == RUN_ID_CONFLICT: + # Not bookkeeping. The file states two different things under one identity and + # the reader cannot tell which one the population it is about to compare + # belongs to. Counted as a CONFLICT — reporting it as a retry would be a false + # statement — and cut at this record's own physical position, so evidence from + # before it is never combined with evidence after it. Recoverable like every + # other boundary here: enough clean later evidence promotes normally. + conflicts += 1 + if verdict == RUN_ID_CONFLICT or degrades_scope(o): + # The record belongs to the epoch it CLOSES: stamped first, counter moved + # after it. A population that recovers is not founded on the observation that + # reported the loss, nor on the one that contradicted its own identity. + cuts_scope[scope_of(o)] += 1 continue rejected[why] += 1 # EVERY loss a rejected line represents is file-global. A record that failed validation @@ -565,40 +637,11 @@ def load_history(path): # because epochs recover; a crossing of a real loss is not. if rejection_is_loss(why): cuts_file += 1 - seen, uniq = {}, [] - for r in recs: - rid = r.get("run_id") - if isinstance(rid, str) and rid: - # Deduplication is per EPOCH. Two epochs are two populations: the same id appearing - # after a loss is that population's own observation, and dropping it — or carrying the - # excluded copy's floor into it — would let an old gap poison a healthy epoch, which is - # the defect this repair exists to remove. Inside one epoch nothing changes: the retry - # is dropped, counted, and cannot launder the survivor's quality. - # The stamp is a pair of COUNTS, so a record of scope "a" and one of scope "b" can - # carry the same numbers while belonging to different epochs. Identity therefore needs - # the scope: without it, one scope's retry lowered another scope's record through - # QUALITY_FLOOR. (cross-family review of the B1 repair) - # ...and only the FILE-GLOBAL half of the stamp. Across a file-global loss the reader - # cannot tell whether a repeated id is the same run, so both copies stand. Across a - # TRUSTED scope-local boundary the file is intact and the reader knows exactly what - # happened: the repeat is the same logical run retrying, and letting its cleaner copy - # into the recovered epoch would launder the loss its own twin reported. - rid = (scope_of(r), r.get(EPOCH_KEY, (0, 0))[0], rid) - if rid in seen: - rejected["duplicate run_id (retry)"] += 1 - # The retry is dropped as an OBSERVATION, never as provenance. One run_id that - # says COMPLETE once and PARTIAL once cannot be resolved in favour of the better - # claim: the sweep that reported less is part of what this id actually saw, and - # keeping the better one would let a retry launder a bounded sweep into a - # complete one. The floor travels with the record the law already reads. - kept = seen[rid] - worse = worst_quality([record_quality(kept), record_quality(r)]) - if worse != record_quality(kept): - kept[QUALITY_FLOOR] = worse - continue - seen[rid] = r - uniq.append(r) - return uniq, rejected, lines, {"file_global": cuts_file, "scope_local": dict(cuts_scope)} + # Deduplication already happened, in the read pass, in physical order, from the SAME verdict + # the boundary used. A second pass with its own notion of identity is exactly how the two + # layers came to disagree about one pair. + return recs, rejected, lines, {"file_global": cuts_file, "scope_local": dict(cuts_scope), + "run_id_conflicts": conflicts} def by_scope(recs): @@ -1139,6 +1182,10 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop f"({dmg.get('file_global', 0)} unattributable, " f"{sum((dmg.get('scope_local') or {}).values())} scope-local) — evidence from " f"before the newest one is not combined with evidence after it") + if hist.get("run_id_conflicts"): + L.append(f" identity : {hist['run_id_conflicts']} record(s) repeat a run id with a " + f"different observation — an identity contradiction, counted as a conflict and " + f"cut where it appears, never reported as a retry") L.append("") if live and live["sessions"]: C = sum(live["carry"].values()) or 1 @@ -1239,7 +1286,8 @@ def main(argv=None): damage = damage_summary(epochs) hist = {"total": len(recs), "in_scope": len(scopes.get(scope, [])), "in_epoch": len(scoped), "comparable": keep, "dropped": dropped, "rejected": rejected, "lines": lines, - "time_order": time_order(keep), "damage": damage} + "time_order": time_order(keep), "damage": damage, + "run_id_conflicts": (epochs or {}).get("run_id_conflicts", 0)} ledger = load_ledger(a.ledger) live = cold = None @@ -1310,6 +1358,12 @@ def main(argv=None): # ...and what the file holds regardless: a historical gap stays true while # the evidence after it is independently complete "damage": damage, + # An observable identity contradiction: two accepted records under one + # run_id whose persisted observations differ. A bounded COUNT, never a map + # keyed by anything a corrupt file chose — and separate from `rejected`, + # because a conflicting record is ACCEPTED and kept, not a line that failed + # to become one. + "run_id_conflicts": (epochs or {}).get("run_id_conflicts", 0), "rejected": dict(rejected), "time_order": hist["time_order"]}, "ledger": ledger, "live": ({"sessions": live["sessions"], "turns": live["turns"], "scanned": live["scanned"],