diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index b7a4a5f..cdbf18f 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -49,6 +49,10 @@ jobs: run: python3 tests/test_evidence_phase2.py - name: offline optimizer (privacy canary, corruption, zero policy mutation) run: python3 tests/test_optimize.py + - name: legacy evidence integrity - the frozen v1.4.2 counterexample matrix + # Ambiguous, incomplete, invalid, non-comparable or host-shifted evidence must not produce + # a candidate specification unless the bounded-evidence exception was explicitly asked for. + run: python3 tests/test_evidence_integrity.py - name: the 1.4.0 release ships shadow evaluation and nothing that writes # Sebuah rilis paling mudah "menyalakan" sesuatu tanpa sengaja saat versinya dinaikkan. # Suite ini yang membuat kalimat "promotion/persistence tidak aktif" bisa diperiksa. diff --git a/README.md b/README.md index 93cccf8..8d8a040 100644 --- a/README.md +++ b/README.md @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2. docs/reference-audits/ — nine projects read at pinned commits docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile) docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements -tests/ 1201 assertions in sixteen suites, mutation-tested; CI on Python 3.9 and 3.12 +tests/ 1601 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12 plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host ``` diff --git a/docs/V142_COUNTEREXAMPLES.md b/docs/V142_COUNTEREXAMPLES.md new file mode 100644 index 0000000..d562683 --- /dev/null +++ b/docs/V142_COUNTEREXAMPLES.md @@ -0,0 +1,627 @@ +# v1.4.2 — frozen counterexample matrix + +Written **before** the repair, against `main` at `77e3677`. Every expectation below is what the +optimizer *must* do; the second block records what it actually did when the matrix was frozen. An +expectation may only change afterwards if the original expectation is independently proven wrong, +and the change must be recorded in §3. + +Scope: the **legacy** optimizer path (`tools/optimize.py`, history schemas 0/1/2, live sweeps from +`tools/carry.py`). Schema-4 typed evidence stays unsupported by this optimizer and the v1.4 typed +promotion stays shadow-only; neither is touched here. + +## 1. The matrix + +| case | fixture | status | strict exit | candidate files | comparable | evidence quality | +|---|---|---|---|---|---|---| +| **R142_01** | 6 schema-2 records, `evidence_quality=PARTIAL`, `skipped_by_limit=5`, no live sweep, no flag | `PARTIAL_EVIDENCE` | 40 | 0 | 6 | `PARTIAL` | +| **R142_01P** | same fixture **with** `--accept-partial` | `CANDIDATE` | 10 | 1 | 6 | `PARTIAL` | +| **R142_02** | 6 eligible `COMPLETE` records + 1 newer `INVALID` record (`sessions=0`, `scanned>0`), incomparable corpus size | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **R142_03** | 6 `COMPLETE` records whose runtime set changes mid-history, share moving ≥ 5 pp | `HOST_BEHAVIOR_SHIFT` | 30 | 0 | 6 | `COMPLETE` | +| **R142_04** | 6 records claiming `COMPLETE` with `sessions=turns=carry_bytes=0` | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | `INVALID` | +| **R142_05** | live sweep over a transcript holding one torn JSON record, with **and** without `--accept-partial` | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | `DEGRADED` (`carry` reports `PARTIAL`, never `COMPLETE`) | +| **R142_06** | `--max-files 1` where discovery order is the reverse of mtime order | carry sweep and skill-listing scan select the **same single source** (the newest) | — | — | — | — | +| **U1** | 6 schema-1 records (a generation with no `evidence_quality` field), with and without `--accept-partial` | `PARTIAL_EVIDENCE` | 40 | 0 | 6 | `UNKNOWN` | +| **P1** | 6 `COMPLETE` records, acquisition counters all zero | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **P3** | P1's population + one `INVALID` record in **another scope** | `CANDIDATE` | 10 | 1 | 6 | `COMPLETE` | +| **P4** | one current-generation (schema-4 envelope) record | `INSUFFICIENT_DATA` | 20 | 0 | 0 | unsupported, refused by name | +| **S_NOACTION** | 6 `COMPLETE` records, share flat (no finding over threshold) | `NO_ACTION` | 0 | 0 | 6 | `COMPLETE` | +| **S_LOCK** | P1's fixture with the emit lock already held | `ALREADY_RUNNING` | 41 | 0 | 6 | `COMPLETE` | + +Quality vocabulary for the legacy law (worst wins, never a majority): + +```text +COMPLETE < PARTIAL < DEGRADED < EMPTY < UNKNOWN < INVALID +``` + +* `PARTIAL` = a bound the caller chose (`--max-files`). Only this one is rescued by + `--accept-partial`. +* `DEGRADED` = evidence that was selected and then lost (unreadable, oversize, malformed, + identity changed under the read, conflicting records). A loss is never a chosen bound, so + `--accept-partial` does not accept it. +* `UNKNOWN` = the record cannot attest its own quality: a schema older than the field, or a + schema-2 record without the acquisition counters its writer always wrote. +* `INVALID` / `EMPTY` = the producer's own terms for a sweep that found nothing usable, plus any + record whose own numbers contradict its claim. + +## 2. Observed on `main` 77e3677 when the matrix was frozen + +Run against these exact fixtures, before a line of the repair existed: + +```text +R142_01 CANDIDATE exit 10 files 1 comparable 6 bounded history promotes +R142_01P CANDIDATE exit 10 files 1 comparable 6 the flag changed nothing: it was never read +R142_02 INSUFFICIENT_DATA exit 20 files 0 comparable 0 the INVALID record anchored, six eligible stranded +R142_03 HOST_BEHAVIOR_SHIFT exit 30 files 1 comparable 6 the refusing status still wrote the file +R142_04 CANDIDATE exit 10 files 1 comparable 6 sessions=turns=carry_bytes=0 promoted +R142_05 carry quality=COMPLETE, malformed=1 a lost record is not a loss +R142_05 NO_ACTION exit 0 files 0 and the run reports nothing wrong +R142_06 listing sample read the OLDEST transcript two definitions of one bound +U1 CANDIDATE exit 10 files 1 comparable 6 a field that never existed read as COMPLETE +P1 CANDIDATE exit 10 files 1 comparable 6 (already correct) +P3 INSUFFICIENT_DATA exit 20 files 0 comparable 0 an invalid record in ANOTHER scope chose the scope +P4 INSUFFICIENT_DATA exit 20 files 0 comparable 0 (already correct) +S_NOACTION NO_ACTION exit 0 files 0 comparable 6 (already correct) +S_LOCK ALREADY_RUNNING exit 41 files 0 comparable 6 (already correct) +``` + +## 3. Deviations from the frozen expectations + +None. + +## 4. Cases added AFTER the freeze + +Not changed expectations — new rows, each from an attack on the repair itself rather than on the +original defect. They are listed separately so the frozen matrix stays readable as what it was. + +| case | fixture | expectation | +|---|---|---| +| **R142_07** | two records sharing one `run_id`, the first `COMPLETE`, the retry `PARTIAL` with `skipped_by_limit=7` | the retry is dropped as an observation and counted, and the quality it reported travels with the record that survives: `PARTIAL_EVIDENCE` / exit 40 / 0 files, `CANDIDATE` with `--accept-partial` | +| **R142_07b** | `schema_version` of `"2"` (string) or `2.0` (float) | `UNKNOWN` — a version this reader cannot name is not a newer generation to trust | +| **R142_08** | schema-2 record with zeroed counters and no `sessions` / `turns` / `carry_bytes` at all | `UNKNOWN` — zero counters say nothing went wrong, not that a sweep happened (cross-family review, round 1) | +| **R142_09** | `--max-files 2` where one of three sources cannot be dated | the datable sources still order by mtime; one unreadable mtime no longer sends the whole selection back to a discovery-order slice (cross-family review, round 1) | + +Confirmation round, attacking the repair again: + +| case | fixture | expectation | +|---|---|---| +| **R142_15** | a `PARTIAL` claim with every counter at zero | `UNKNOWN` — a bound shows up in `skipped_by_limit` and a loss in a loss counter; a claim no counter can explain is not a bound `--accept-partial` may adopt | +| **R142_15b** | a record carrying `malformed`, `malformed_lines`, `identity_changed`, `conflicted_sources` or `records_rejected` | `DEGRADED` — a loss must be honoured wherever the reader can see it, not only in the two counters the first cut looked at | +| **R142_16** | history + one record from an unsupported schema 3, and one whose shares sum to 10 | `PARTIAL_EVIDENCE` / exit 40 / 0 files — a rejected line is damage unless it is a refusal by design, a deduplicated retry, or a line that was never a carry record (control: a foreign line still promotes) | +| **R142_17** | ledger of 100 `checked` lines plus one `{"event": "garbage"}` | the unaccountable line counts as rejected and the guard only observes | + +## 5. B1 — the blocker an independent acceptance review found, and the epoch matrix + +The first head of this branch closed the six original defects (§1–§4) and introduced one of its own. +An **independent acceptance review of PR #13 BLOCKED it**: container damage was permanent. One torn +line — the crash fragment `docs/MULTI_AGENT.md` §Concurrency calls an expected event, the one the +reader is designed to "reject exactly that line and count it" — set the whole file to `DEGRADED` +forever. Nothing in the product expires, rotates or repairs a history (`docs/MULTI_AGENT.md`: "The +optimizer does not schedule, expire or rotate"), and `--accept-partial` cannot adopt `DEGRADED` by +design, so a single crash permanently disabled promotion for **every scope** sharing +`~/logs/carry_history.jsonl`. + +Measured on that head before the repair (every row: zero candidate files, exit 40, forever): + +```text +6 good + torn PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +6 good + torn + 2 good PARTIAL_EVIDENCE exit 40 comparable 8 hq DEGRADED +6 good + torn + 6 good PARTIAL_EVIDENCE exit 40 comparable 12 hq DEGRADED +6 good + torn + 60 good PARTIAL_EVIDENCE exit 40 comparable 66 hq DEGRADED +scope-a good + torn + scope-b good PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +torn first + 6 good PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +zero-carry record (shares {}) PARTIAL_EVIDENCE exit 40 comparable 6 hq DEGRADED +6 good + torn + 60 good, --accept-partial exit 40 (no flag can adopt a loss) +``` + +### The replacement: a history EPOCH, cut at the physical position of the loss + +An unattributable loss cuts the promotion history **at that line's position in the file**. Evidence +before the cut and evidence after it are never combined for a promotion; the loss stays reported; +the newest epoch is, by construction, free of damage. No time window, no expiry, no ratio, no new +override flag — and `--accept-partial` still adopts only a caller's chosen bound, never a loss. + +| case | fixture (physical order) | status | exit | files | comparable | active epoch | history.quality | boundaries | +|---|---|---|---|---|---|---|---|---| +| **B1_01** | 6 good, torn | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | `EMPTY` | 1 | +| **B1_02** | 6 good, torn, 2 good | `INSUFFICIENT_DATA` | 20 | 0 | 2 | 2 | `COMPLETE` | 1 | +| **B1_03** | 6 good, torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | +| **B1_04** | 6 good, torn, 60 good | `CANDIDATE` | 10 | 1 | 60 | 60 | `COMPLETE` | 1 | +| **B1_05** | scope-a 6 good, torn, scope-b 6 good | `CANDIDATE` (scope b) · `INSUFFICIENT_DATA` with `--scope-id a` | 10 · 20 | 1 · 0 | 6 · 0 | 6 · 0 | `COMPLETE` · `EMPTY` | 1 | +| **B1_06** | 6 good, torn, 6 good, torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 2 | +| **B1_07** | torn, 6 good | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | +| **B1_08** | 6 good, truncated final line | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | `EMPTY` | 1 | +| **B1_09** | 6 good, foreign JSON line | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 0 | +| **B1_10** | 6 good, schema-4 envelope line | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 0 | +| **B1_11** | 6 good, zero-carry record (`shares {}`, `carry_bytes 0`) | `CANDIDATE` | 10 | 1 | 6 | 7 | `COMPLETE` | 0 | +| **B1_12** | 6 records the law calls INVALID | `PARTIAL_EVIDENCE` | 40 | 0 | 0 | 6 | `EMPTY` | 0 | +| **B1_13** | 5 good, `run_id=X` good, torn, `run_id=X` good ×6 | `CANDIDATE` | 10 | 1 | 6 | 6 | `COMPLETE` | 1 | + +Rules the matrix encodes: + +* **Unattributable loss → file-global boundary.** An unparseable line, an oversized line or an + unreadable file cannot name a scope, so the cut applies to every scope. +* **Parseable rejection carrying a readable `scope_id` → scope-local boundary.** Only that scope's + continuity is cut; a sibling agent is not punished for a neighbour's corrupted record. Attribution + trusts the record's own `scope_id` (a string of 1–64 characters, no control bytes) and falls back + to file-global whenever it cannot be read. +* **Not damage, so not a boundary:** a well-formed foreign JSON line, current-generation (schema-4) + evidence refused by design, and a deduplicated retry. +* **The damage is still reported** — `history.rejected` counts it and `history.damage` says how many + boundaries the file holds — while `history.quality` describes only the evidence eligible for the + CURRENT analysis. A historical gap stays true while the post-gap evidence is independently + complete. +* **`history.quality` with nothing comparable is `EMPTY`, never `COMPLETE`** (the acceptance + review's MEDIUM M1). +* **A zero-carry sweep is readable evidence, not corruption** (MEDIUM M2). `carry.py` says it in its + own report — "A session whose every item lands on its final turn carries nothing" — the 1.3 writer + emits `shares: {}` for it and the current writer guards `if C else {}`. Measured on this branch: + `carry.accumulate()` on such a transcript returns `sessions=1 turns=6 carry_total=0`. The record + is `EMPTY` for the quality law: nothing to compare, and nothing wrong with the file. + +### 5.1 Deviations from this frozen matrix + +None. + +### 5.2 Expectations the epoch model supersedes + +Three rows written for the first head's permanent-damage model are now wrong, and are replaced +rather than quietly re-run. In each, the damaged line sits at the END of the file, so there is no +post-loss epoch: the outcome is still **no promotion and no candidate file**, and what changes is +only the word for it and whether later evidence can ever clear it. + +```text +R142_11 torn line in the history PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +R142_15 unsupported schema, bad shares PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +R142_18 carry record with no shares KEY PARTIAL_EVIDENCE / DEGRADED -> INSUFFICIENT_DATA / EMPTY +``` + +Each keeps its protection and gains a companion case: the same rejected line placed BEFORE a healthy +population, where the population after the loss must stand on its own. The records before the loss +are never counted with the records after it — `R142_11` pins that as `comparable == 2` for +`6 good + torn + 2 good`. + +Dedup across a boundary is also new, and is stated rather than inherited: deduplication is +**per epoch**. The same `run_id` on the far side of a loss is that population's own observation and +is kept; inside one epoch the retry is still dropped, still counted, and still cannot launder the +survivor's quality (`tests/test_optimize.py` pins both). + +### 5.3 What a cross-family lane found in the epoch repair itself + +Two more places where the repair still consulted the population it had just cut, both reproduced +before they were believed and both now pinned by a case and a mutant: + +| case | fixture | expectation | +|---|---|---| +| **B1_14** | six `stale` records, a torn line, and a ledger of 100 writes | the ledger finding may still promote — its evidence is the ledger — but the scope, and therefore the candidate id, comes from the current epoch or from nothing: `default`, never `stale` | +| **B1_15** | scope-local loss in `a`, six clean `a` records, scope-local loss in `b`, one `b` record sharing a `run_id` with `a` | the epoch stamp is a pair of counts, so two scopes can hold the same numbers; deduplication identity carries the scope, and `a` stays `COMPLETE` instead of inheriting `b`'s `PARTIAL` through the retry floor | + +The long-run simulator was discarding the epoch it was handed, so it could combine records across a +loss; it now analyses `active_records()` like the CLI. + +## 6. Trusted damage boundaries (frozen before the second repair) + +Frozen against the branch head `e6c1f4f`, **before** any of it was implemented. The adversarial +review of the epoch model found that attribution was taken from the very line that had just failed +validation: a `scope_id` holding invalid UTF-8 survived `errors="replace"` as `U+FFFD`, read as a +"readable" label, and cut a scope that does not exist — while the real population kept crossing the +loss. §2 of that review reproduced it as `CANDIDATE`, one candidate file, twelve comparable records +spanning both sides of the gap. + +The rule this section freezes is about provenance, not about Unicode: + +> **A record that failed validation is not a trustworthy authority for its own scope attribution.** + +Two consequences, and one deliberate trade: + +* **Every rejection that is a loss is a FILE-GLOBAL boundary.** No rejected line may name a scope, + whatever its `scope_id` looks like — `rev`, `agent-b`, a path, a 64-character label, or bytes that + never decoded. This over-blocks: one corrupt line cuts scopes that were never damaged. That is the + chosen half of the trade (§23 of the task), because epochs recover and a fail-open crossing of a + real loss does not. +* **The history is read as BYTES and decoded strictly, per physical line.** `errors="replace"` + destroyed the evidence that decoding had failed; a line that cannot decode is now a named + rejection (`line is not valid UTF-8`) and a file-global boundary, and the reader continues at the + next line rather than abandoning the file. +* **A scope-local boundary now has exactly one trusted source** (§6.2): a record that PASSED + validation and whose own canonical loss counters prove that evidence was lost. + +### 6.1 TUTF — invalid UTF-8 (all FILE_GLOBAL, no ghost scope) + +Fixture unless stated: `6 clean scope=rev` + the bad line + `6 clean scope=rev`. The bad line is +SameWrite-shaped, holds invalid UTF-8 in the named field, and independently fails `valid_record()` +(its shares sum to 20, not ~100). + +| case | bad line | boundary | `damage.scope_local` | active epoch | status | +|---|---|---|---|---|---| +| **TUTF_01** | invalid UTF-8 inside `scope_id` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_02** | invalid UTF-8 inside `shares` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_03** | invalid UTF-8 inside `run_id` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_04** | invalid UTF-8 inside `workload_class` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_05** | one undecodable line before every record (`bad + 6 clean`) | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TUTF_06** | one undecodable line at EOF (`6 clean + bad`) | FILE_GLOBAL ×1 | `{}` | 0 | `INSUFFICIENT_DATA` | +| **TUTF_07** | two undecodable lines (`6 + bad + 6 + bad + 6`) | FILE_GLOBAL ×2 | `{}` | 6 | `CANDIDATE` | +| **TUTF_08** | one undecodable line between two scopes (`6×a + bad + 6×b`) | FILE_GLOBAL ×1 | `{}` | 6 (`b`); 0 with `--scope-id a` | `CANDIDATE`; `INSUFFICIENT_DATA` | + +In every row: no candidate may name a `run_id` from before the boundary, `scope.known` may not +contain a label that came from the rejected line, and no raw undecodable byte may appear anywhere in +the JSON output, the human output or a candidate file. + +### 6.2 TSCOPE — a rejected record does not authenticate its own `scope_id` + +Fixture: `6 clean scope=rev` + one rejected record carrying a syntactically perfect +`scope_id="ghost"` + `6 clean scope=rev`. Every row below is a loss. + +| case | rejected record | boundary | `damage.scope_local` | active epoch | status | +|---|---|---|---|---|---| +| **TSCOPE_01** | shares sum to 20, not ~100 | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_02** | a non-numeric share value | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_03** | `sessions` negative (implausible) | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_04** | `evidence_quality` outside the vocabulary | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_05** | `schema_version: 3`, a legacy schema this reader cannot name | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_06** | a carry record with no `shares` key at all | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | +| **TSCOPE_07** | `record_type` is not `carry_run` | FILE_GLOBAL ×1 | `{}` | 6 | `CANDIDATE` | + +Still **not** a loss, so still no boundary at all: a well-formed foreign JSON line, a bare object +with no `shares` and no sign of being ours, and current-generation (schema-4) evidence refused by +design. A migration must not read as file damage. + +Canaries, each used as the rejected record's `scope_id`, each of which must leave +`damage.scope_local == {}` and must not appear in `scope.known`: `/etc/passwd.d/synthetic`, +`agent-b`, `rev`, a 63-character label, a 64-character label, a label holding control bytes, and a +label that never decoded. **Cardinality:** 2000 rejected lines carrying 2000 distinct `scope_id` +values produce `file_global == 2000`, `scope_local == {}` — attacker-controlled strings cannot add a +single public map key. + +### 6.3 The hypothesis this repair had to test first: a VALID record that is DEGRADED + +Reproduced on `e6c1f4f` before anything was designed for it, with a matched control: + +```text +6 clean + 1 valid schema-2 record with unreadable=3 + N clean, scope agent-a + N=2 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files + N=6 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files + N=60 PARTIAL_EVIDENCE / history.quality DEGRADED / 0 candidate files +control (identical populations, no degraded record) + N=2,6,60 CANDIDATE / history.quality COMPLETE / 1 candidate file +``` + +`VALID_DEGRADED_FAIL_STUCK=YES`. The record passes `valid_record()`, so no rejection and no +boundary was ever considered; it stays in `comparable()` forever, and `history_quality()` is the +worst of that set. No amount of later healthy evidence clears it — the same fail-stuck shape the +epoch model was built to remove, reached through the one door the epoch model did not watch. + +So §9 of the task applies, and narrowly. A **fully validated** record whose OWN canonical loss +counters prove degraded evidence creates a **scope-local** recovery boundary after itself. This is +trusted where a rejected line is not: the record passed structural validation, its `scope_id` is the +same field every accepted record publishes through `scope.known`, and the damage fact comes from +`RECORD_LOSS_COUNTERS`, not from guessing what a corrupt line meant. + +| case | fixture (scope `a` unless stated) | boundary | active epoch | status | `history.quality` | +|---|---|---|---|---|---| +| **VD_01** | 6 clean + DEGRADED + 2 clean | SCOPE_LOCAL `{a: 1}` | 2 | `INSUFFICIENT_DATA` | `COMPLETE` | +| **VD_02** | 6 clean + DEGRADED + 6 clean | SCOPE_LOCAL `{a: 1}` | 6 | `CANDIDATE` | `COMPLETE` | +| **VD_03** | 6 clean + DEGRADED + 60 clean | SCOPE_LOCAL `{a: 1}` | 60 | `CANDIDATE` | `COMPLETE` | +| **VD_04** | VD_02, reading the candidate | — | — | the candidate names none of the pre-boundary `run_id`s, and not the degraded record's own | | +| **VD_05** | 6 clean + DEGRADED at EOF | SCOPE_LOCAL `{a: 1}` | 0 | `INSUFFICIENT_DATA` | `EMPTY` | +| **VD_06** | `6×a`, `6×b`, DEGRADED `a`, `6×a`, `6×b` | SCOPE_LOCAL `{a: 1}` | `a`: 6 · `b`: 12 | `CANDIDATE` both | `COMPLETE` | +| **VD_07** | 6 clean + valid `PARTIAL` by `skipped_by_limit=3` + 6 clean | **none** | 13 | `PARTIAL_EVIDENCE`; `CANDIDATE` with `--accept-partial` | `PARTIAL` | +| **VD_08** | 6 clean + a schema-1 record (`UNKNOWN`, cannot attest) + 6 clean | **none** | 13 | `PARTIAL_EVIDENCE` | `UNKNOWN` | +| **VD_09** | 6 clean + a valid `INVALID` record + a valid `EMPTY` (zero-carry) record + 6 clean | **none** | 14 | `CANDIDATE` | `COMPLETE` | + +The degraded record belongs to the OLD epoch (§14): it sits before its own boundary, in physical +order, never in the population that recovers. Its degradation stays reported in `history.damage`. + +`PARTIAL`, `UNKNOWN`, `INVALID` and `EMPTY` are deliberately excluded. Only an actual loss counter +cuts: a bound the caller asked for is an intentional population, a legacy schema that cannot attest +completeness is not proof that a line was lost, and an ineligible record is not a damaged one. + +### 6.4 Retry and a trusted boundary (§15) + +Deduplication identity becomes `(scope, FILE-GLOBAL epoch, run_id)` — the scope-local component is +deliberately **not** part of it. Across a file-global loss the reader cannot tell whether a repeated +`run_id` is the same run, so both copies stand (B1_13). Across a *trusted* scope-local boundary the +file is intact and the reader knows exactly what happened, so a repeated `run_id` is the same +logical run retrying, and the later copy is bookkeeping. + +| case | physical order (scope `a`) | expectation | +|---|---|---| +| **VD_10** | `clean X`, `degraded retry X`, 6 clean | boundary `{a: 1}` after the degraded copy; both copies pre-boundary; `duplicate run_id (retry)` = 1; active epoch 6, `CANDIDATE`, `COMPLETE` | +| **VD_11** | `degraded X`, `clean retry X`, 6 clean | boundary `{a: 1}` after the degraded copy; the clean retry is dropped as the same run's bookkeeping and never becomes evidence in the recovered epoch; `duplicate run_id (retry)` = 1; active epoch 6, `CANDIDATE`, `COMPLETE` | + +Neither order launders the loss into the recovered epoch, and neither poisons it forever. + +### 6.5 File-global recovery is unchanged (§20, §21) + +| case | fixture | expectation | +|---|---|---| +| **MSG_01** | `6×a`, `6×b`, torn line, `6×a`, `2×b` | `a`: epoch 6, `CANDIDATE`; `b`: epoch 2, `INSUFFICIENT_DATA`; nothing from before the cut helps either | +| **MSG_02** | the same with the sufficiencies reversed (`2×a`, `6×b` after the cut) | `a`: `INSUFFICIENT_DATA`; `b`: `CANDIDATE` | + +File-global means every scope is cut **at that position**, never that a scope is disabled forever: +`B1_01`–`B1_04` still pin `+0 / +2 / +6 / +60`. + +### 6.6 Frozen-decision amendments to §5 (§27) + +Two rows of the B1 matrix were written against the old attribution rule and are wrong under this +one. Both are replaced rather than deleted, and the property each was protecting keeps a case. + +```text +B1_13e three copies of one run_id in one epoch, the middle one carrying unreadable=1 + was: history.quality DEGRADED, 2 duplicates, 0 files + now: the middle copy is a VALID record proving a loss, so it cuts scope-locally. + The property "the worst copy survives inside one epoch" moves to a fixture whose + copies are PARTIAL/COMPLETE (no loss counter); the degraded-copy behaviour is + VD_10/VD_11 above. + +B1_15 two scope-local boundaries taken from two REJECTED lines + was: damage.scope_local == {a: 1, b: 1} from rejected records + now: rejected lines are file-global, so the same fixture yields file_global == 2. + The property it protected — two scopes can hold the same epoch NUMBERS, so the + deduplication identity must carry the scope — is re-pinned with the same shape + built from TRUSTED sources: a valid DEGRADED record in each scope. +``` + +No other B1 or R142 expectation changes. `R142_01`–`R142_06` and `B1_01`–`B1_14` are re-run +unchanged. + +### 6.7 The residual limit this repair does not close (§24) + +`LEGACY_STRUCTURALLY_VALID_CORRUPTION_LIMITATION=YES`. A legacy flat record carries no integrity +tag. A corruption that turns one valid record into a *different* valid record — `scope_id` flipped +from `agent-a` to `agent-b`, a share vector rewritten to another vector that still sums to ~100 — is +indistinguishable from a record the producer meant to write. Nothing in this repair detects it, and +nothing can: the information needed to tell them apart is absent from the format. What the repair +does guarantee is narrower and checkable: a line that *fails* validation never supplies attribution, +and a line that cannot be decoded is never mistaken for one that can. + +### 6.8 Deviations from this frozen section + +None. + +### 6.9 Added after the freeze (neither a deviation nor a replacement) + +Two cases were added while the repair was built. Neither changes a frozen expectation; each pins a +property the frozen rows implied but did not state. + +```text +TUTF_09 a line that would be a perfectly good record BUT FOR its bytes. + Every TUTF_01..08 fixture also fails validation on its own, so a lenient decode and a + strict one could in principle agree on the outcome by accident. Here they cannot: with + errors="replace" the line is ACCEPTED, publishes U+FFFD through scope.known and creates no + boundary; strictly it is a loss like any other. Expectation: 12 records read (not 13), + file-global boundary ×1, scope.known == ["rev"]. + +TPRIV a rejected line carrying synthetic secret- and path-shaped strings. + Found while verifying §35: valid_record() echoed the rejected line's own + `schema_version` VALUE into `history.rejected`, which reaches --json, the human report and + every log that keeps them. Truncating it is not a bound — thirty characters of a + credential is still the credential — so only a NUMBER is echoed now, and any other value + is named by type ("unsupported schema_version of type str"). A real schema number is still + named in full. +``` + +Third addition, found while verifying §6 of the task ("a migration must not read as file damage") +rather than reported by anyone: + +```text +TSCOPE control: another tool's record_type in a shared history. + `{"record_type": "hermes_run", "ts": ..., "note": ...}` was classified + "unknown record_type" and counted as a LOSS, so a foreign entry cut the file. The + distinction `no shares` already draws one check further down now applies here too: a line + that ALSO carries our fields (schema_version / run_id / carry_bytes) with an unknown + record_type is a corrupted record of ours and stays a loss (TSCOPE_07); a line that carries + none of them is another tool's entry and is counted without a boundary. +``` + +### 6.10 What a cross-family review found in the trust repair itself + +| case | fixture (scope `a`) | expectation | +|---|---|---| +| **VD_12** | `DEGRADED X`, 6 clean, `DEGRADED X` again at EOF | one boundary, not two: active epoch 6, `CANDIDATE`, `scope_local == {a: 1}`, one counted retry | +| **VD_12b** | the same pattern repeated (`DEG X`, 6 clean, `DEG X`, 6 clean, `DEG X`) | still one boundary; active epoch 12, `CANDIDATE` | +| **VD_13** | `DEGRADED X`, 6 clean, `DEGRADED Y` (a different run) | two boundaries — a real second loss still cuts | +| **VD_13b** | two `DEGRADED` records carrying no `run_id` at all | two boundaries: a record that cannot be shown to be a retry is its own observation | +| **VD_13c** | `DEGRADED X` in scope `a` and `DEGRADED X` in scope `b` | `{a: 1, b: 1}` — the same id in another scope is that scope's own loss | + +The boundary was opened before deduplication, so a copy that was then discarded as +`duplicate run_id (retry)` still moved the counter. Measured before the fix, the reviewer's own +fixture gave `INSUFFICIENT_DATA`, `records_in_epoch=0`, `scope_local={"a": 2}` with six healthy +records sitting between the two copies — and repeating `6 healthy + one more copy of X` held the +scope down indefinitely: **the fail-stuck shape this repair exists to remove, rebuilt out of its own +recovery mechanism.** A boundary is now opened at most once per `(scope, file-global epoch, run_id)`. +A record with no `run_id` cannot be shown to be a retry and stays its own observation; across a +file-global loss the identity differs, because there the reader cannot tell whether a repeated id is +the same run at all. + +### 6.11 The second finding of that review: `scope_id` was made an authority without a contract + +Round 2 returned `NEEDS-FIX` with a HIGH that goes to the root of §6.2's own rationale. The trust +model says a scope-local boundary is safe because "the record passed structural validation and its +`scope_id` is the same field every accepted record publishes". The first half was true; the second +was an assumption. `valid_record()` never checked `scope_id` at all, and `scope_of()` is +`str(r.get("scope_id") or "default")` — so a record that passes validation with +`scope_id = ["rev"]` becomes the scope `"['rev']"`. + +Reproduced before it was believed, on `189db57`: + +```text +6 clean rev · one valid DEGRADED record with scope_id = ["rev"] · 6 clean rev + status CANDIDATE, exit 10, one candidate file + records_in_epoch=12, comparable=12 + damage {"file_global": 0, "scope_local": {"['rev']": 1}} + candidate names rid-rev-0..5 AND rid-rev-20..25 — both sides of the loss +``` + +That is the blocked B-UTF8 defect rebuilt through the one door this repair opened: attribution taken +from a field nobody had checked, a cut landing on a population that does not exist, and the real +population crossing the loss. + +**The fix is the producer's own contract, not a new invention.** `tools/carry.py` writes exactly one +shape: `"scope_id": str(scope_id or "default")[:64]`. So a value that is not a string, or a string +longer than 64 characters, was not written by it — the record is corrupt and is refused with a +STATIC reason (its content must never be echoed), which makes it a file-global loss like any other +unattributable one. + +| case | `scope_id` | expectation | +|---|---|---| +| **TSCOPE_08** | `["rev"]` · `5` · `{"s": 1}` · a 200-character string | file-global ×1, `scope_local == {}`, active epoch 6, `scope.known == ["rev"]`, candidate names only the post-loss run ids | +| **TSCOPE_08 controls** | absent · `"rev"` · `""` · exactly 64 characters · a label holding a control byte | still a record — the producer can write all of these | + +The control byte stays accepted deliberately: the producer's cap truncates length but does not strip +control characters, such a label is already published through `scope.known` on every head of this +branch, and calling it corruption would invent damage where the file is intact. That is the residual +this repair states rather than hides. + +## 7. Run-id conflict: identity is a claim, not proof (frozen before the repair) + +The author-adversarial review of `71016ee` found that two **materially different** valid observations +carrying the same `run_id` were collapsed into one retry, so the second loss never opened a +boundary. Reproduced on that head before anything was designed for it: + +```text +DEGRADED X1(run_id=SHARED, ts 0, share 30, unreadable 2, sessions 40, turns 1000, scanned 80) +6 healthy +DEGRADED X2(run_id=SHARED, ts 40, share 66, unreadable 9, sessions 91, turns 2400, scanned 150) +6 healthy + + CANDIDATE / exit 10 / 1 candidate file + records_in_epoch 12, comparable 12, scope_local boundaries 1 (for TWO reported losses) + "duplicate run_id (retry)" = 1 + candidate evidence = 6 records from before X2 and 6 from after it +``` + +Nine persisted fields differ between X1 and X2 (`ts`, `shares`, `sessions`, `turns`, `carry_bytes`, +`scanned`, `unreadable`, `runtimes`, `models`). Both pass `valid_record()`; both are independently +`DEGRADED`. Controls on the same head: a different `run_id` gives two boundaries and an active epoch +of six; an identical duplicate, a key-order-only difference and a whitespace-only difference all give +one boundary and an epoch of twelve; a file-global loss between the copies gives two scope-local +boundaries plus one file-global. + +### 7.1 The rule this section freezes + +> **`run_id` is an identity CLAIM, not proof of semantic equality.** +> Two records may be treated as one retry only when their identity AND their persisted observation +> are equivalent. + +`SAME_RUN_ID_ONLY_IS_RETRY=NO`. Three cases, and one classification decides all of them — boundary +opening, deduplication, the quality floor and the diagnostics read the same verdict: + +* **Case A — TRUE_RETRY.** Same scope, same file-global epoch, same `run_id`, **and the same + canonical persisted observation.** Deduplicated exactly as before: the later copy is dropped, it + is counted, its quality still travels to the survivor, and it does **not** open a second boundary. +* **Case B — RUN_ID_CONFLICT.** Same identity, **different** canonical persisted observation. This + is an observable integrity event, not bookkeeping. The later record is **kept** (it is a valid + observation), it is **not** reported as a retry, and it opens a recoverable **scope-local** + boundary at its own physical position. It belongs to the epoch it closes. +* **Case C — across a file-global boundary.** Unchanged: the reader cannot establish continuity + across an unattributable loss, so the later occurrence is a fresh identity. No retry, no conflict, + no imported quality floor. + +Equivalence is decided on the **whole persisted record**, not a hand-picked subset — the defect being +repaired exists precisely because the previous identity was too weak. The comparison is a digest of +the parsed object with keys sorted, so JSON key order and whitespace cannot make two semantically +identical observations differ, and the reader's own private annotations (`_history_epoch`, +`_evidence_quality_floor`) are excluded, because reading a file must not change what a record is. A +difference in any other persisted field — including one this reader does not know — conservatively +makes the observations non-equivalent: that is fail-closed, and epochs recover. + +### 7.2 The frozen matrix + +Pieces: `A` = 6 healthy COMPLETE records, `B` = 6 more, all scope `s1`, one file-global epoch unless +stated. `scope_local` is the total count of scope-local boundaries. + +| case | fixture | status | exit | files | epoch | comparable | scope_local | file_global | retries | conflicts | history.quality | +|---|---|---|---|---|---|---|---|---|---|---|---| +| **RID_01** | `DEG X1 · A · DEG X1 (identical) · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_02** | `DEG X1 · A · DEG X2 (different)` | `INSUFFICIENT_DATA` | 20 | 0 | 0 | 0 | 2 | 0 | 0 | 1 | `EMPTY` | +| **RID_03** | `DEG X1 · A · DEG X2 · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_04** | `CLEAN X · A · DEG X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_05** | `DEG X · A · CLEAN X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_06** | `COMPLETE X · A · COMPLETE X' (different) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_07** | `PARTIAL X · A · PARTIAL X' (different) · B`, with and without `--accept-partial` | `CANDIDATE` | 10 | 1 | 6 | 6 | 1 | 0 | 0 | 1 | `COMPLETE` | +| **RID_08** | `DEG X scope a · DEG X scope b · 6×a · 6×b`, analysing `a` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 0 | `COMPLETE` | +| **RID_09** | `DEG X1 · A · torn line · DEG X1 (identical) · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 1 | 0 | 0 | `COMPLETE` | +| **RID_10** | `DEG X1 · A · DEG X1 with the keys in reverse order · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_11** | `DEG X1 · A · DEG X1 with different separators · B` | `CANDIDATE` | 10 | 1 | 12 | 12 | 1 | 0 | 1 | 0 | `COMPLETE` | +| **RID_12** | `DEG X1 · A · DEG X1 with another `workload_class` · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_13** | `DEG X1 · A · DEG X1 with other `shares` · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_14** | `DEG X1 · A · DEG X1 with another loss counter · B` | `CANDIDATE` | 10 | 1 | 6 | 6 | 2 | 0 | 0 | 1 | `COMPLETE` | +| **RID_15** | `DEG X1 · 3 healthy · DEG X2 · 3 healthy · DEG X3 · B` (all three different) | `CANDIDATE` | 10 | 1 | 6 | 6 | 3 | 0 | 0 | 2 | `COMPLETE` | + +`RID_03`, `RID_04`, `RID_05`, `RID_06`, `RID_07`, `RID_12`–`RID_15` additionally require the +candidate's `evidence_run_ids` to name **only** records after the last boundary. + +### 7.3 Complete-versus-complete is a boundary, and why + +`RID_06` is the row that needed an argument rather than an inference. Neither record reports an +acquisition loss, so nothing was lost — but the file states two different things under one identity, +and the reader has no way to tell which one the population it is about to compare actually belongs +to. A trend drawn across that point is drawn over a file whose identity discipline has already +failed. The choice is therefore a **recoverable scope-local boundary**: it costs the pre-conflict +records, it costs nothing permanently, and the alternative — silently keeping one of the two and +calling it a retry — is the exact statement the repair exists to stop making. The diagnostic stays +honest either way: the conflict is counted as a conflict, never as a retry. + +### 7.4 The limit this repair cannot close + +`IDENTICAL_REUSED_RUNID_LIMITATION=YES`. Two physically distinct losses in one scope and one +file-global epoch that carry the same `run_id` **and identical persisted content** are +indistinguishable from one run retried identically. The flat legacy record has no other identity to +read. Line position is deliberately NOT used as identity: doing so would make every true retry open +a fresh boundary and recreate the fail-stuck behaviour the previous round removed. + +### 7.5 Deviations from this frozen section + +None. + +### 7.6 Frozen-decision amendments the identity law forces (§21 order: recorded before editing) + +Five existing cases were written under the rule this repair overturns — they use **one `run_id` for +two different observations** and assert that the pair is a retry. Under §7.1 such a pair is a +conflict, so each is replaced rather than quietly re-run, and each keeps the property it was +protecting. + +```text +R142_07 "a retry cannot upgrade what its own run_id saw" + was: run_id "dup" appears twice, the second claiming PARTIAL with skipped_by_limit=7, + and the survivor is lowered to PARTIAL through QUALITY_FLOOR + -> PARTIAL_EVIDENCE / 40 / 0 files + now: the two copies differ, so they are a CONFLICT: the later one cuts and nothing + before it can promote -> INSUFFICIENT_DATA / 20 / 0 files + The property is unchanged and the guarantee is STRICTER: no repeated id can upgrade what + it saw. It is now enforced by a boundary rather than by a floor. The case gains its + companion: two IDENTICAL copies are a true retry, counted, no boundary, quality untouched. + +B1_13b "a retry inside one epoch still lowers the survivor" -> same cause, same replacement. +B1_13e "three copies inside one epoch: the worst of them survives" -> the three copies differ, + so they are two conflicts; RID_15 is the frozen row for that shape. +VD_10 "clean first, the retry reports the loss" frozen in §6.4 on the assumption that one +VD_11 "the loss first, the retry reports clean" run_id means one run. That assumption is + exactly what this repair removes. Both pairs differ materially, so each later copy is a + conflict that cuts at its own position; the later population still recovers, and neither + copy can launder anything. VD_11 now shows two scope-local boundaries instead of one. +``` + +**A consequence worth stating rather than hiding:** under the new law a TRUE_RETRY has, by +construction, the same persisted observation and therefore the same `record_quality`, so +`QUALITY_FLOOR` can no longer change anything. It is kept as a guard, not as a live path, and the +anti-laundering property it used to carry is now carried by the conflict boundary. No test claims to +exercise a floor that cannot fire. + +### 7.7 What the machine output gains, and where it is NOT + +`history.run_id_conflicts` is a bounded integer beside `history.rejected`, and `output_schema_version` +stays **2** — v2 is unreleased and this is it evolving, not a second contract. + +It is deliberately **not** in `history.rejected`: that map counts lines that failed to become +records, and a conflicting record is accepted and kept. Reporting it there would be the same false +statement as calling it a retry. It is also not a map keyed by anything the file chose — six hundred +distinct conflicting ids produce the integer `600` and zero new keys, so a corrupt history cannot +grow the diagnostics. The human report names the count and what it means, never a value from either +record. + +`history.damage` is unchanged and its invariant still holds: +`boundaries == file_global + sum(scope_local)`. A conflict cut is a scope-local boundary like any +other, so a conflict that lands on a record which was ALSO degraded produces one boundary, not two — +a boundary is a position, not a tally of reasons — while `run_id_conflicts` keeps counting the +reasons separately. diff --git a/experiments/aivos/bash_residue.py b/experiments/aivos/bash_residue.py index 21ecc39..45badc6 100644 --- a/experiments/aivos/bash_residue.py +++ b/experiments/aivos/bash_residue.py @@ -44,8 +44,7 @@ def main(): a = ap.parse_args() paths, _ = profiles.resolve(a.scan or []) - if a.max_files and len(paths) > a.max_files: - paths = sorted(paths, key=os.path.getmtime, reverse=True)[:a.max_files] + paths = carry.bounded_paths(paths, a.max_files) # one definition of "the newest N" sizes = collections.defaultdict(list) total = collections.Counter() diff --git a/experiments/aivos/longrun.py b/experiments/aivos/longrun.py index 2d0313d..2579ace 100644 --- a/experiments/aivos/longrun.py +++ b/experiments/aivos/longrun.py @@ -30,8 +30,12 @@ def rec(ts, shares, scope, turns=1000, workload="", runtimes=None): + # The acquisition counters a real 1.2/1.3 sweep always wrote. Since 1.4.2 a record claiming + # COMPLETE without them cannot carry a promotion, so a simulation that omitted them would be + # simulating a population no producer ever writes. return {"schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(), - "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7, + "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7, "scanned": 100, + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "scope_id": scope, "workload_class": workload, "evidence_quality": "COMPLETE", "runtimes": runtimes or {"2.1.270": 40}, "models": {"m1": 40}, "shares": shares, "bpt": {k: 1.0 for k in shares}} @@ -154,17 +158,22 @@ def longrun(days, cycles, out, plateau=20): day_new = 0 for _cycle in range(cycles): - recs, _rej, _lines = optimize.load_history(hist) - scopes = optimize.by_scope(recs) + recs, _rej, _lines, ep = optimize.load_history(hist) + # the same epoch the CLI analyses: evidence from before a loss is not combined with + # evidence after it, here either (cross-family review of the B1 repair) + current = optimize.active_records(recs, ep) + scopes = optimize.by_scope(current) for role in ROLES: keep, dropped = optimize.comparable(scopes.get(role, [])) h = {"comparable": keep, "total": len(recs), "in_scope": len(scopes.get(role, [])), - "rejected": {}, "dropped": dropped, "time_order": optimize.time_order(keep)} + "in_epoch": len(scopes.get(role, [])), "rejected": dict(_rej), + "dropped": dropped, "time_order": optimize.time_order(keep), + "damage": optimize.damage_summary(ep)} lv = live(int(30 + min(day, plateau) * (30.0 / plateau))) f = optimize.analyse(lv, h, None, None, scope=role) st = optimize.overall_status(f, h, lv, optimize.population(keep), False) statuses[st] += 1 - w, e, _fail = optimize.emit_candidates(f, cand) + w, e, _fail = optimize.emit_candidates(f, cand, st) written += len(w) existing += len(e) day_new += len(w) diff --git a/experiments/aivos/readiness.py b/experiments/aivos/readiness.py index fbbf244..5db515a 100644 --- a/experiments/aivos/readiness.py +++ b/experiments/aivos/readiness.py @@ -130,7 +130,11 @@ def main(): "schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(), "ts": 1_750_000_000 + i * 604800, "sessions": 60, "turns": 3000, "carry_bytes": 10 ** 8, "scope_id": "governed", "workload_class": "audit", - "evidence_quality": "COMPLETE", "runtimes": {"2.1.271": 60}, "models": {"m": 60}, + "evidence_quality": "COMPLETE", "scanned": 120, + # the acquisition counters the producer writes; since 1.4.2 a COMPLETE claim + # without them cannot promote (tests/test_evidence_integrity.py) + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, + "runtimes": {"2.1.271": 60}, "models": {"m": 60}, "shares": {"Bash": 40.0 + i * 9, "Read": 60.0 - i * 9}, "bpt": {"Bash": 1.0, "Read": 1.0}}) + "\n") @@ -153,6 +157,25 @@ def main(): check("JSON leaks no secret", CANARY in rj.stdout, False) check("JSON leaks no path", ("/home/" in rj.stdout) or (ws in rj.stdout), False) + # The same population with the counters stripped claims a completeness it cannot show: it must + # refuse, and it must write nothing. Without this, the fixture edit above could hide the gate. + bare_hist = os.path.join(state, "bare_history.jsonl") + bare_cand = os.path.join(state, "bare_candidates") + with open(hist, encoding="utf-8") as fh, open(bare_hist, "w", encoding="utf-8") as out_fh: + for line in fh: + o = json.loads(line) + for k in ("unreadable", "oversize", "skipped_by_limit"): + o.pop(k, None) + out_fh.write(json.dumps(o) + "\n") + rb = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "optimize.py"), + "--history", bare_hist, "--ledger", os.path.join(state, "none.jsonl"), + "--scan", "--scope-id", "governed", "--json", "--strict-exit", + "--emit-candidate", bare_cand], + capture_output=True, text=True, timeout=900) + check("a record that cannot attest its own sweep does not promote", rb.returncode, 40) + check("and nothing is written for it", + [f for r_, _d, fs in os.walk(bare_cand) for f in fs], []) + specs = [os.path.join(r_, f) for r_, _d, fs in os.walk(cand) for f in fs if f.endswith(".md")] check("a specification was written, outside the governed repository", bool(specs), True) check("specifications live outside the workspace", diff --git a/tests/test_evidence_integrity.py b/tests/test_evidence_integrity.py new file mode 100644 index 0000000..095fc97 --- /dev/null +++ b/tests/test_evidence_integrity.py @@ -0,0 +1,1158 @@ +#!/usr/bin/env python3 +"""v1.4.2 — one case per row of docs/V142_COUNTEREXAMPLES.md. + +The legacy optimizer promoted findings from evidence it had never checked: a history built from +bounded sweeps, a record claiming a completeness its own numbers contradict, a population anchored +on the one record that was thrown away, a sweep that lost records to torn JSON, and a status that +refused promotion while still writing the candidate file to disk. + +Every case below is a counterexample first and a regression second: it was RED on 77e3677, and the +matrix that says what it must do was frozen before the repair existed. + +Standalone: run this file.""" +import json +import os +import subprocess +import sys +import tempfile + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +OPT = os.path.join(ROOT, "tools", "optimize.py") +sys.path.insert(0, os.path.join(ROOT, "tools")) +import carry # noqa: E402 +import optimize # noqa: E402 + +P = F = 0 +TS0 = 1_750_000_000 + + +def check(label, got, want): + global P, F + if got == want: + P += 1 + print(f" PASS {label}") + else: + F += 1 + print(f" FAIL {label}: got {got!r}, want {want!r}") + + +def rec(i, share, quality="COMPLETE", schema=2, counters=True, sessions=40, turns=1000, + carry_bytes=10 ** 7, scanned=100, scope="default", workload="code", runtime="2.1.270", + unreadable=0, oversize=0, skipped=0): + """One legacy history record, shaped like the one carry.history() wrote in 1.2/1.3. + + `counters=False` is the record a hand-edited file or a back-filled migration produces: it + claims a quality it never acquired the facts for.""" + r = {"schema_version": schema, "record_type": "carry_run", "run_id": f"r{scope}{i}", + "scope_id": scope, "workload_class": workload, "ts": TS0 + i * 604800, + "sessions": sessions, "turns": turns, "carry_bytes": carry_bytes, "scanned": scanned, + "runtimes": {runtime: 40}, "models": {"m1": 40}, + "shares": {"Bash": round(share, 4), "Read": round(100.0 - share, 4)}, + "bpt": {"Bash": 1.0, "Read": 1.0}} + if schema >= 2: + r["evidence_quality"] = quality + if counters: + r.update(unreadable=unreadable, oversize=oversize, skipped_by_limit=skipped) + return r + + +def write(path, rows): + """Rows may be bytes: a history holding a line that is not valid UTF-8 is exactly the input + the decode rule has to be tested against, and it cannot be written through a text handle.""" + with open(path, "wb") as fh: + for r in rows: + if not isinstance(r, bytes): + r = (r if isinstance(r, str) else json.dumps(r)).encode("utf-8") + fh.write(r + b"\n") + return path + + +def run(rows, extra=(), scan=(), lock=False): + """Run the optimizer as the scheduler runs it -> (exit code, parsed --json, files on disk).""" + d = tempfile.mkdtemp(prefix="sw-142-") + hist = write(os.path.join(d, "history.jsonl"), rows) + out = os.path.join(d, "cand") + if lock: + os.makedirs(out, exist_ok=True) + open(os.path.join(out, ".optimize.lock"), "w").write("1") + argv = [sys.executable, OPT, "--history", hist, "--ledger", os.path.join(d, "none.jsonl"), + "--emit-candidate", out, "--json", "--strict-exit", "--scan"] + list(scan) + list(extra) + p = subprocess.run(argv, capture_output=True, text=True, timeout=300) + try: + j = json.loads(p.stdout) + except Exception: + raise SystemExit(f"optimizer produced no JSON (rc={p.returncode}):\n{p.stdout}\n{p.stderr}") + files = [f for _b, _d, fs in os.walk(out) for f in fs if f != ".optimize.lock"] + return p.returncode, j, len(files) + + +def case(label, rows, status, code, files, comparable=None, extra=(), scan=(), lock=False, + history_quality=None): + rc, j, n = run(rows, extra=extra, scan=scan, lock=lock) + check(f"{label}: status", j["status"], status) + check(f"{label}: strict exit", rc, code) + check(f"{label}: candidate files on disk", n, files) + if comparable is not None: + check(f"{label}: comparable", j["history"]["comparable"], comparable) + if history_quality is not None: + check(f"{label}: history quality", j["history"].get("quality"), history_quality) + return j + + +def transcript(path, turns=80, listing=False, torn=False): + """A minimal Claude-Code-shaped transcript: enough turns to be a session, optionally one + skill listing attachment, optionally one torn JSON record.""" + rows = [] + if listing: + rows.append(json.dumps({"type": "user", "attachment": { + "type": "skill_listing", + "content": "- alpha: does a thing that is described at some length\n" + "- beta: does another thing, also described at some length\n"}})) + for t in range(turns): + rows.append(json.dumps({"type": "assistant", "message": { + "id": f"msg_{t}", "usage": {"input_tokens": 10, "output_tokens": 7}, + "content": [{"type": "tool_use", "name": "Bash", "input": {"command": "ls -la"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 400}]}})) + if torn: + rows[len(rows) // 2] = '{"type": "assistant", "message": {"usage": {"out' + with open(path, "w", encoding="utf-8") as fh: + fh.write("\n".join(rows) + "\n") + return path + + +EMPTY_HIST = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + + + +def trusted_boundaries(): + """docs/V142_COUNTEREXAMPLES.md §6 — a line that failed validation is not an authority. + + The epoch model's attribution came from the very line that had just been rejected: invalid + UTF-8 in `scope_id` survived `errors="replace"` as U+FFFD, read as a "readable" label, and cut + a scope that does not exist while the real population kept crossing the loss. Every case below + was frozen before this repair existed and is RED on e6c1f4f. + """ + d = tempfile.mkdtemp(prefix="sw-142-tb-") + + def ser(n, first=0, scope="rev", per_week=3.0, base=30.0): + return [json.dumps(rec(first + i, base + per_week * i, scope=scope)) for i in range(n)] + + def later(n=6, scope="rev"): + """The population AFTER the boundary: its own climb, so it can promote on its own.""" + return ser(n, 20, scope=scope, base=48.0) + + def bad_utf8(field, scope="ghost"): + """A SameWrite-shaped line holding invalid UTF-8, which also fails validation on its own.""" + r = dict(rec(99, 50.0, scope=scope), shares={"Bash": 10.0, "Read": 10.0}) + if field == "shares": + r["shares"] = {"Bash": 10.0, "@@M@@": 10.0} + else: + r[field] = "@@M@@" + return json.dumps(r).encode().replace(b"@@M@@", b"\xff\xfe\x80") + + def rejected_rec(scope="ghost", **over): + r = dict(rec(99, 50.0, scope=scope), shares={"Bash": 10.0, "Read": 10.0}) + r.update(over) + return json.dumps(r) + + def dmg(j): + x = j["history"].get("damage") or {} + return x.get("file_global"), x.get("scope_local") + + def shape(rc, j, n): + return ((j["status"], rc, n, j["scope"]["records_in_epoch"], j["history"]["comparable"]) + + dmg(j)) + + def run_paths(rows, extra=()): + dd = tempfile.mkdtemp(dir=d) + hist = write(os.path.join(dd, "history.jsonl"), rows) + out = os.path.join(dd, "cand") + p = subprocess.run([sys.executable, OPT, "--history", hist, "--ledger", + os.path.join(dd, "none.jsonl"), "--emit-candidate", out, "--json", + "--strict-exit", "--scan"] + list(extra), + capture_output=True, text=True, timeout=300) + files = [os.path.join(b, f) for b, _sub, fs in os.walk(out) for f in fs + if f != ".optimize.lock"] + return p.returncode, json.loads(p.stdout), files + + # ---------------------------------------------------------------- TUTF (§6.1) + print("\nTUTF - a line that cannot be decoded cannot name a scope") + for name, field in (("TUTF_01", "scope_id"), ("TUTF_02", "shares"), + ("TUTF_03", "run_id"), ("TUTF_04", "workload_class")): + rc, j, n = run(ser(6) + [bad_utf8(field)] + later()) + check(f"{name} invalid UTF-8 in {field} is an unattributable loss", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + check(f"{name} the decode failure is counted by name", + j["history"]["rejected"].get("line is not valid UTF-8"), 1) + check(f"{name} nothing from the rejected line reaches the report", + ("�" in json.dumps(j, ensure_ascii=False), "ghost" in j["scope"]["known"]), + (False, False)) + + rc, j, n = run([bad_utf8("scope_id")] + ser(6)) + check("TUTF_05 an undecodable line before every record", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + rc, j, n = run(ser(6) + [bad_utf8("scope_id")]) + check("TUTF_06 an undecodable line at EOF leaves no epoch", + shape(rc, j, n), ("INSUFFICIENT_DATA", 20, 0, 0, 0, 1, {})) + rc, j, n = run(ser(6) + [bad_utf8("scope_id")] + ser(6, 20, base=48.0) + + [bad_utf8("run_id")] + ser(6, 40, base=66.0)) + check("TUTF_07 two undecodable lines are two boundaries", + shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 2, {})) + between = ser(6, 0, "a") + [bad_utf8("scope_id")] + later(6, "b") + rc, j, n = run(between) + check("TUTF_08 an undecodable line between two scopes cuts both", + shape(rc, j, n) + (j["scope"]["analysed"],), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, "b")) + rc, j, n = run(between, extra=["--scope-id", "a"]) + check("TUTF_08 ...and the scope before it has no epoch left", + shape(rc, j, n), ("INSUFFICIENT_DATA", 20, 0, 0, 0, 1, {})) + + rc, j, files = run_paths(ser(6) + [bad_utf8("scope_id")] + later()) + body = open(files[0], encoding="utf-8").read() if files else "" + ids = sorted(x.strip() for line in body.splitlines() if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check("TUTF_01 the candidate rests only on the recovered epoch", + ids, sorted("rrev%d" % i for i in range(20, 26))) + + # A line that would be a perfectly good record BUT FOR its bytes: with a lenient decode it is + # accepted and publishes U+FFFD as a scope; strictly, it is a loss like any other. (post-freeze + # addition, docs §6.9 — it strengthens TUTF_01 rather than changing any frozen expectation.) + whole = dict(rec(99, 50.0, scope="@@M@@")) + intact_but_undecodable = json.dumps(whole).encode().replace(b"@@M@@", b"\xff\xfe\x80") + rc, j, n = run(ser(6) + [intact_but_undecodable] + later()) + check("TUTF_09 an otherwise-valid record with undecodable bytes is a loss, not a record", + shape(rc, j, n) + (j["history"]["records"], j["scope"]["known"]), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, 12, ["rev"])) + + # TSCOPE_08: `scope_id` is an ATTRIBUTION AUTHORITY, so it is checked before a record is + # accepted. The producer writes exactly one shape — `str(scope_id or "default")[:64]` — and a + # value of another type let `["rev"]` become the scope `"['rev']"`: the cut landed on a + # population nobody has while the real `rev` records kept crossing the loss. (docs §6.11) + print("\nTSCOPE_08 - a scope label the producer could not have written") + for label, sid in (("a list", ["rev"]), ("an integer", 5), ("an object", {"s": 1}), + ("a label longer than the producer's own cap", "x" * 200)): + forged = dict(rec(6, 48.0, scope="rev", unreadable=2), run_id="LOSS-X") + forged["scope_id"] = sid + rc, j, n = run(ser(6) + [json.dumps(forged)] + later()) + check(f"TSCOPE_08 {label} is a loss nobody can attribute", + shape(rc, j, n) + (j["scope"]["known"],), + ("CANDIDATE", 10, 1, 6, 6, 1, {}, ["rev"])) + for label, sid in (("absent", None), ("a plain label", "rev"), ("the empty string", ""), + ("exactly 64 characters", "y" * 64), + ("a control byte the producer can write", "a\u0001b")): + ok = dict(rec(6, 48.0, scope="rev"), run_id="OK-X") + if sid is None: + ok.pop("scope_id") + else: + ok["scope_id"] = sid + check(f"TSCOPE_08 control: {label} is still a record", + optimize.valid_record(ok), (True, "")) + + # ---------------------------------------------------------------- privacy (§35) + print("\nTPRIV - a rejected line's own content is never echoed into public output") + secret = "sk-synthetic-NOTAREALKEY-0123456789" + leaky = dict(rec(99, 50.0, scope="/home/synthetic/.ssh/id_ed25519"), + shares={"Bash": 10.0, "Read": 10.0}, workload_class=secret, + run_id=secret + "-run", schema_version=secret) + rc, j, files = run_paths(ser(6) + [json.dumps(leaky)] + later()) + blob = json.dumps(j, ensure_ascii=False) + spec = "".join(open(f, encoding="utf-8").read() for f in files) + check("TPRIV nothing from the rejected line reaches --json or a candidate file", + (secret[:16] in blob, "/home/synthetic" in blob, + secret[:16] in spec, "/home/synthetic" in spec, dmg(j)), + (False, False, False, False, (1, {}))) + check("TPRIV the reason names the value's TYPE, never the value", + sorted(j["history"]["rejected"]), ["unsupported schema_version of type str"]) + check("TPRIV a real schema number is still named", + optimize.valid_record({"schema_version": 3, "shares": {"Bash": 100.0}})[1], + "unsupported schema_version 3") + + # ---------------------------------------------------------------- TSCOPE (§6.2) + print("\nTSCOPE - a rejected record does not authenticate its own scope_id") + for label, bad in ( + ("TSCOPE_01 shares that do not sum to a population", rejected_rec()), + ("TSCOPE_02 a non-numeric share", + rejected_rec(shares={"Bash": "lots", "Read": 50.0})), + ("TSCOPE_03 an impossible session count", rejected_rec(sessions=-1)), + ("TSCOPE_04 a quality word outside the vocabulary", + rejected_rec(evidence_quality="SPLENDID")), + ("TSCOPE_05 a legacy schema this reader cannot name", rejected_rec(schema_version=3)), + ("TSCOPE_06 a carry record with no shares key", + json.dumps({k: v for k, v in rec(99, 50.0, scope="ghost").items() + if k != "shares"})), + ("TSCOPE_07 a record_type this reader does not know", + rejected_rec(record_type="carry_note"))): + rc, j, n = run(ser(6) + [bad] + later()) + check(label, shape(rc, j, n), ("CANDIDATE", 10, 1, 6, 6, 1, {})) + check(label + " — and no ghost scope is published", "ghost" in j["scope"]["known"], False) + + for label, line in (("a foreign JSON line", json.dumps({"note": "another tool's entry"})), + ("another tool's record_type in a shared history", + json.dumps({"record_type": "hermes_run", "ts": TS0, "note": "not ours"})), + ("a bare object that never claimed to be ours", + json.dumps({"note": "x", "shares": None})), + ("current-generation evidence refused by design", + json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}))): + rc, j, n = run(ser(6) + [line] + later()) + check("TSCOPE control: " + label + " is not damage", + shape(rc, j, n), ("CANDIDATE", 10, 1, 12, 12, 0, {})) + + print("\nTSCOPE canaries - no rejected label may become a public map key") + for canary in ("/etc/passwd.d/synthetic", "agent-b", "rev", "x" * 63, "y" * 64, + "ctl\x01label", "shares: 100"): + rc, j, n = run(ser(6, scope="real") + [rejected_rec(scope=canary)] + + later(6, "real")) + check("TSCOPE canary %r stays out of the public report" % canary[:18], + (dmg(j), canary in j["scope"]["known"]), ((1, {}), False)) + + flood = [rejected_rec(scope="s%04d" % i) for i in range(2000)] + rc, j, n = run(flood + later()) + check("TSCOPE cardinality: 2000 rejected labels add no public key", + (dmg(j), j["scope"]["known"], len(json.dumps(j["history"]["damage"])) < 200), + ((2000, {}), ["rev"], True)) + + # ---------------------------------------------------------------- VD (§6.3) + print("\nVD - a VALID record whose own counters prove a loss cuts its own scope") + deg = json.dumps(rec(6, 48.0, scope="a", unreadable=3)) + check("the fail-stuck hypothesis names a real quality", + optimize.record_quality(json.loads(deg)), "DEGRADED") + for label, tail, epoch, status, code, files, quality in ( + ("VD_01 an epoch too small to carry a trend", ser(2, 20, "a", base=48.0), 2, + "INSUFFICIENT_DATA", 20, 0, "COMPLETE"), + ("VD_02 a sufficient epoch promotes on its own", ser(6, 20, "a", base=48.0), 6, + "CANDIDATE", 10, 1, "COMPLETE"), + ("VD_03 and it still promotes sixty records later", + ser(60, 20, "a", per_week=1.0, base=20.0), 60, "CANDIDATE", 10, 1, "COMPLETE")): + rc, j, n = run(ser(6, 0, "a") + [deg] + tail) + check(label, (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"].get("quality")) + dmg(j), + (status, code, files, epoch, quality, 0, {"a": 1})) + + rc, j, files = run_paths(ser(6, 0, "a") + [deg] + ser(6, 20, "a", base=48.0)) + body = open(files[0], encoding="utf-8").read() if files else "" + ids = sorted(x.strip() for line in body.splitlines() if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check("VD_04 the degraded record is not evidence in the epoch it opened", + ids, sorted("ra%d" % i for i in range(20, 26))) + + rc, j, n = run(ser(6, 0, "a") + [deg]) + check("VD_05 a degradation at EOF leaves no epoch to promote from", + (j["status"], rc, n, j["scope"]["records_in_epoch"], j["history"].get("quality")) + + dmg(j), ("INSUFFICIENT_DATA", 20, 0, 0, "EMPTY", 0, {"a": 1})) + + both = (ser(6, 0, "a") + ser(6, 0, "b") + [deg] + ser(6, 20, "a", base=48.0) + + ser(6, 20, "b", base=48.0)) + rc, j, n = run(both, extra=["--scope-id", "a"]) + check("VD_06 the degraded scope starts again after its own loss", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 6, 0, {"a": 1})) + rc, j, n = run(both, extra=["--scope-id", "b"]) + check("VD_06 ...and the neighbour keeps every record it ever had", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 12, 0, {"a": 1})) + + bounded = json.dumps(rec(6, 48.0, scope="a", quality="PARTIAL", skipped=3)) + rows = ser(6, 0, "a") + [bounded] + ser(6, 20, "a", base=48.0) + rc, j, n = run(rows) + check("VD_07 an intentional bound is not damage", + (j["status"], j["scope"]["records_in_epoch"], j["history"].get("quality")) + dmg(j), + ("PARTIAL_EVIDENCE", 13, "PARTIAL", 0, {})) + rc, j, n = run(rows, extra=["--accept-partial"]) + check("VD_07 ...and the flag still adopts it", (j["status"], n), ("CANDIDATE", 1)) + + rc, j, n = run(ser(6, 0, "a") + [json.dumps(rec(6, 48.0, scope="a", schema=1))] + + ser(6, 20, "a", base=48.0)) + check("VD_08 a schema that cannot attest completeness is not a loss", + (j["status"], j["scope"]["records_in_epoch"], j["history"].get("quality")) + dmg(j), + ("PARTIAL_EVIDENCE", 13, "UNKNOWN", 0, {})) + + rc, j, n = run(ser(6, 0, "a") + + [json.dumps(rec(6, 48.0, scope="a", sessions=0, scanned=40)), + json.dumps(dict(rec(7, 48.0, scope="a"), shares={}, bpt={}, carry_bytes=0))] + + ser(6, 20, "a", base=48.0)) + check("VD_09 an ineligible record is not a damaged one", + (j["status"], j["scope"]["records_in_epoch"], j["history"]["comparable"]) + dmg(j), + ("CANDIDATE", 14, 12, 0, {})) + + # ---------------------------------------------------------------- retry (§6.4) + print("\nVD_10/VD_11 - a retry may neither launder a loss nor poison the epoch after it") + clean_x = json.dumps(dict(rec(5, 45.0, scope="a"), run_id="X")) + deg_x = json.dumps(dict(rec(6, 48.0, scope="a", unreadable=2), run_id="X")) + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §7.6): both pairs differ materially, + # so under the identity law each later copy is a CONFLICT, not a retry. The population after it + # still recovers and neither copy can launder anything — the guarantee these rows exist for. + for label, rows, loc in (("VD_10 clean first, then something else under the same id", + ser(5, 0, "a") + [clean_x, deg_x] + ser(6, 20, "a", base=48.0), + {"a": 1}), + ("VD_11 the loss first, then something else under the same id", + ser(5, 0, "a") + [deg_x, clean_x] + ser(6, 20, "a", base=48.0), + {"a": 2})): + rc, j, n = run(rows) + check(label, (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"].get("quality"), + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0), + j["history"]["run_id_conflicts"]) + dmg(j), + ("CANDIDATE", 10, 1, 6, "COMPLETE", 0, 1, 0, loc)) + + # VD_12/VD_13: what a cross-family review of THIS repair found. A retry reports the same loss + # its twin already reported, and a deduplicated retry is bookkeeping by this reader's own rule. + # Letting a late copy open a SECOND boundary let "six healthy records, then one more copy of X" + # erase a recovered epoch — on repeat, forever: the fail-stuck shape this repair exists to + # remove, rebuilt out of its own recovery mechanism. (docs §6.10) + print("\nVD_12/VD_13 - one boundary per logical run, and still one per real loss") + healthy = ser(6, 20, "a", base=48.0) + rc, j, n = run([deg] + healthy + [deg]) + check("VD_12 a late copy of the SAME degraded run does not cut again", + (j["status"], rc, n, j["scope"]["records_in_epoch"], + j["history"]["rejected"].get("duplicate run_id (retry)")) + dmg(j), + ("CANDIDATE", 10, 1, 6, 1, 0, {"a": 1})) + rc, j, n = run([deg] + healthy + [deg] + ser(6, 40, "a", base=66.0) + [deg]) + check("VD_12 ...and repeating the pattern cannot hold the scope down forever", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 12, 0, {"a": 1})) + other = json.dumps(dict(json.loads(deg), run_id="ra-other", ts=TS0 + 30 * 604800)) + rc, j, n = run([deg] + healthy + [other]) + check("VD_13 a genuinely different second loss still cuts", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + ("INSUFFICIENT_DATA", 20, 0, 0, 0, {"a": 2})) + anon = json.dumps({k: v for k, v in json.loads(deg).items() if k != "run_id"}) + rc, j, n = run([anon, anon] + healthy) + check("VD_13 a degraded record with no run_id cannot be shown to be a retry", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), ("CANDIDATE", 6, 0, {"a": 2})) + twin_b = json.dumps(dict(json.loads(deg), scope_id="b")) + rc, j, n = run([deg, twin_b] + healthy, extra=["--scope-id", "a"]) + check("VD_13 the same run_id in another scope is that scope's own loss", + (j["status"], j["scope"]["records_in_epoch"]) + dmg(j), + ("CANDIDATE", 6, 0, {"a": 1, "b": 1})) + + # ---------------------------------------------------------------- file-global (§6.5) + print("\nMSG - a file-global cut applies to every scope at that position, and only there") + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 5' + msg1 = (ser(6, 0, "a") + ser(6, 0, "b") + [TORN] + ser(6, 20, "a", base=48.0) + + ser(2, 20, "b", base=48.0)) + for scope, status, code, files, epoch in (("a", "CANDIDATE", 10, 1, 6), + ("b", "INSUFFICIENT_DATA", 20, 0, 2)): + rc, j, n = run(msg1, extra=["--scope-id", scope]) + check(f"MSG_01 scope {scope} after an unattributable loss", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + (status, code, files, epoch, 1, {})) + msg2 = (ser(6, 0, "a") + ser(6, 0, "b") + [TORN] + ser(2, 20, "a", base=48.0) + + ser(6, 20, "b", base=48.0)) + for scope, status, code, files, epoch in (("a", "INSUFFICIENT_DATA", 20, 0, 2), + ("b", "CANDIDATE", 10, 1, 6)): + rc, j, n = run(msg2, extra=["--scope-id", scope]) + check(f"MSG_02 scope {scope} when the sufficiencies are reversed", + (j["status"], rc, n, j["scope"]["records_in_epoch"]) + dmg(j), + (status, code, files, epoch, 1, {})) + + +def run_id_conflicts(): + """docs/V142_COUNTEREXAMPLES.md §7 — run_id is an identity CLAIM, not proof of equality. + + Two materially different valid observations sharing one run_id were collapsed into a single + retry on 71016ee, so the second loss never opened a boundary and a candidate was written whose + evidence spanned it. Every row below was frozen before the repair existed and is RED on that + head.""" + d = tempfile.mkdtemp(prefix="sw-142-rid-") + + def A(n, first, base, scope="s1", **kw): + return [json.dumps(rec(first + i, base + 3.0 * i, scope=scope, **kw)) for i in range(n)] + + def X(run_id="SHARED", **kw): + """One persisted observation under a chosen identity.""" + return dict(rec(0, 30.0, scope="s1", **kw), run_id=run_id) + + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 4' + MID, TAIL = A(6, 10, 36.0), A(6, 50, 72.0) + X1 = X(unreadable=2) + X2 = dict(X(unreadable=9), ts=TS0 + 40 * 604800, sessions=91, turns=2400, + carry_bytes=3 * 10 ** 7, scanned=150, + shares={"Bash": 66.0, "Read": 34.0}) + X3 = dict(X2, ts=TS0 + 41 * 604800, unreadable=4, sessions=55, + shares={"Bash": 70.0, "Read": 30.0}) + + def rid(label, rows, status, code, files, epoch, comparable, scope_local, file_global, + retries, conflicts, quality, extra=(), post_only=None): + rc, j, n = run(rows, extra=extra) + h, sc = j["history"], j["scope"] + dm = h["damage"] + got = (j["status"], rc, n, sc["records_in_epoch"], h["comparable"], + sum(dm["scope_local"].values()), dm["file_global"], + (h["rejected"] or {}).get("duplicate run_id (retry)", 0), + h.get("run_id_conflicts", 0), h["quality"]) + check(label, got, (status, code, files, epoch, comparable, scope_local, file_global, + retries, conflicts, quality)) + + rid("RID_01 an exact duplicate is one retry and one boundary", + [json.dumps(X1)] + MID + [json.dumps(dict(X1))] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + rid("RID_02 a materially different copy under one id is a conflict", + [json.dumps(X1)] + MID + [json.dumps(X2)], + "INSUFFICIENT_DATA", 20, 0, 0, 0, 2, 0, 0, 1, "EMPTY") + rid("RID_03 ...and the population after it recovers on its own", + [json.dumps(X1)] + MID + [json.dumps(X2)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_04 a clean record and a degraded one under one id conflict", + [json.dumps(X(unreadable=0))] + MID + [json.dumps(X2)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE") + rid("RID_05 ...and so do a degraded one and a clean one", + [json.dumps(X1)] + MID + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_06 two different COMPLETE records under one id still contradict", + [json.dumps(X(unreadable=0))] + MID + + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE") + partial_a = json.dumps(X(quality="PARTIAL", skipped=4)) + partial_b = json.dumps(dict(X2, unreadable=0, evidence_quality="PARTIAL", skipped_by_limit=9)) + for flag in ((), ("--accept-partial",)): + rid(f"RID_07 a bounded pair under one id conflicts too {list(flag)}", + [partial_a] + MID + [partial_b] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 1, 0, 0, 1, "COMPLETE", extra=flag) + rid("RID_08 the same id in another scope is another observation", + [json.dumps(dict(X1, scope_id="a")), json.dumps(dict(X1, scope_id="b"))] + + A(6, 20, 48.0, scope="a") + A(6, 20, 48.0, scope="b"), + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 0, "COMPLETE", extra=["--scope-id", "a"]) + rid("RID_09 a file-global loss makes the later copy a fresh identity", + [json.dumps(X1)] + MID + [TORN, json.dumps(dict(X1))] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 1, 0, 0, "COMPLETE") + rid("RID_10 key order alone is not a different observation", + [json.dumps(X1)] + MID + + [json.dumps({k: X1[k] for k in reversed(list(X1))})] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + rid("RID_11 whitespace alone is not a different observation", + [json.dumps(X1)] + MID + [json.dumps(X1, separators=(" , ", " : "))] + TAIL, + "CANDIDATE", 10, 1, 12, 12, 1, 0, 1, 0, "COMPLETE") + for label, other in (("RID_12 another workload class", dict(X1, workload_class="other")), + ("RID_13 other shares", dict(X1, shares={"Bash": 31.0, "Read": 69.0})), + ("RID_14 another loss counter", dict(X1, unreadable=5))): + rid(label + " is a conflict", [json.dumps(X1)] + MID + [json.dumps(other)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 2, 0, 0, 1, "COMPLETE") + rid("RID_15 three different copies under one id are three boundaries", + [json.dumps(X1)] + A(3, 10, 36.0) + [json.dumps(X2)] + A(3, 20, 48.0) + + [json.dumps(X3)] + TAIL, + "CANDIDATE", 10, 1, 6, 6, 3, 0, 0, 2, "COMPLETE") + + # the candidate may rest only on the population after the last boundary + for label, rows in (("RID_03", [json.dumps(X1)] + MID + [json.dumps(X2)] + TAIL), + ("RID_06", [json.dumps(X(unreadable=0))] + MID + + [json.dumps(dict(X2, unreadable=0, oversize=0))] + TAIL), + ("RID_15", [json.dumps(X1)] + A(3, 10, 36.0) + [json.dumps(X2)] + + A(3, 20, 48.0) + [json.dumps(X3)] + TAIL)): + dd = tempfile.mkdtemp(dir=d) + hist = write(os.path.join(dd, "history.jsonl"), rows) + out = os.path.join(dd, "cand") + subprocess.run([sys.executable, OPT, "--history", hist, "--ledger", + os.path.join(dd, "none.jsonl"), "--emit-candidate", out, "--json", + "--strict-exit", "--scan"], capture_output=True, text=True, timeout=300) + body = "".join(open(os.path.join(b, f), encoding="utf-8").read() + for b, _s, fs in os.walk(out) for f in fs if f != ".optimize.lock") + names = sorted(x.strip() for line in body.splitlines() + if line.startswith("evidence_run_ids:") + for x in line.split(":", 1)[1].split(",")) + check(f"{label} the candidate names only the recovered population", + names, sorted("rs1%d" % i for i in range(50, 56))) + + # §18: a reader annotation may never change what a record IS + a = dict(X1) + b = dict(X1) + a[optimize.EPOCH_KEY] = (3, 7) + b[optimize.QUALITY_FLOOR] = "DEGRADED" + check("a private annotation does not change the observation", + (optimize.observation_digest(a), optimize.observation_digest(b)), + (optimize.observation_digest(dict(X1)), optimize.observation_digest(dict(X1)))) + check("but a persisted field does", + optimize.observation_digest(dict(X1, ts=X1["ts"] + 1)) + != optimize.observation_digest(X1), True) + check("an extra persisted field this reader does not know makes them differ", + optimize.observation_digest(dict(X1, future_field=1)) + != optimize.observation_digest(X1), True) + + +def main(): + d = tempfile.mkdtemp(prefix="sw-142-fx-") + good_rows = [rec(i, 30.0 + 3.0 * i) for i in range(6)] + + # ---------------------------------------------------------------- the law itself + print("\nlegacy evidence-quality law") + check("worst wins, not the majority", + optimize.worst_quality(["COMPLETE", "COMPLETE", "PARTIAL", "COMPLETE"]), "PARTIAL") + check("no evidence is not bad evidence", optimize.worst_quality([]), "COMPLETE") + check("a schema older than the field cannot claim the field", + optimize.record_quality(rec(0, 40.0, schema=1, counters=False)), "UNKNOWN") + check("schema 2 without the counters its writer always wrote", + optimize.record_quality(rec(0, 40.0, counters=False)), "UNKNOWN") + check("counters present and zero: COMPLETE is attested", + optimize.record_quality(rec(0, 40.0)), "COMPLETE") + check("a chosen bound is PARTIAL", + optimize.record_quality(rec(0, 40.0, skipped=5)), "PARTIAL") + check("a loss is DEGRADED, not a chosen bound", + optimize.record_quality(rec(0, 40.0, unreadable=3)), "DEGRADED") + check("the claim cannot be better than the counters", + optimize.record_quality(rec(0, 40.0, quality="COMPLETE", oversize=1)), "DEGRADED") + check("the counters cannot be better than the claim", + optimize.record_quality(rec(0, 40.0, quality="PARTIAL", skipped=5)), "PARTIAL") + check("...but a PARTIAL claim no counter can explain is not a bound to adopt", + optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), "UNKNOWN") + check("and the flag does not adopt it either", + optimize.may_promote(optimize.record_quality(rec(0, 40.0, quality="PARTIAL")), True), + False) + for counter in ("malformed", "malformed_lines", "identity_changed", "conflicted_sources", + "records_rejected"): + check(f"a record carrying {counter} is not COMPLETE", + optimize.record_quality(dict(rec(0, 40.0), **{counter: 1})), "DEGRADED") + check("looked and found nothing usable: INVALID", + optimize.record_quality(rec(0, 40.0, sessions=0, scanned=100)), "INVALID") + check("had nothing to look at: EMPTY", + optimize.record_quality(rec(0, 40.0, sessions=0, scanned=0)), "EMPTY") + check("sessions without turns is not a sweep that happened", + optimize.record_quality(rec(0, 40.0, turns=0)), "INVALID") + check("sessions without carry is not a sweep that happened", + optimize.record_quality(rec(0, 40.0, carry_bytes=0)), "INVALID") + check("a count that is not a count", + optimize.record_quality(rec(0, 40.0, unreadable=-1)), "INVALID") + check("a boolean is not a count", + optimize.record_quality(rec(0, 40.0, oversize=True)), "INVALID") + check("--accept-partial accepts the bound it is named for", + optimize.may_promote("PARTIAL", True), True) + check("--accept-partial does not accept a loss", + optimize.may_promote("DEGRADED", True), False) + check("--accept-partial does not accept what cannot be verified", + optimize.may_promote("UNKNOWN", True), False) + check("--accept-partial does not accept an invalid record", + optimize.may_promote("INVALID", True), False) + check("COMPLETE needs no flag", optimize.may_promote("COMPLETE", False), True) + check("PARTIAL without the flag stays refused", optimize.may_promote("PARTIAL", False), False) + + # ---------------------------------------------------------------- R142_01 / R142_01P + print("\nR142_01 - a history built from bounded sweeps is not a population") + partial = [rec(i, 30.0 + 3.0 * i, quality="PARTIAL", skipped=5) for i in range(6)] + case("R142_01", partial, "PARTIAL_EVIDENCE", 40, 0, comparable=6, history_quality="PARTIAL") + case("R142_01P (--accept-partial)", partial, "CANDIDATE", 10, 1, comparable=6, + extra=["--accept-partial"], history_quality="PARTIAL") + + # ---------------------------------------------------------------- R142_02 + print("\nR142_02 - the anchor comes from evidence that survives its own filter") + stranding = [rec(i, 30.0 + 3.0 * i) for i in range(6)] + stranding.append(rec(9, 55.0, quality="INVALID", sessions=0, turns=100000, carry_bytes=0)) + case("R142_02", stranding, "CANDIDATE", 10, 1, comparable=6, history_quality="COMPLETE") + + # ---------------------------------------------------------------- R142_03 + print("\nR142_03 - a status that refuses promotion writes nothing") + shifted = ([rec(i, 30.0 + 4.0 * i, runtime="2.1.270") for i in range(3)] + + [rec(i, 30.0 + 4.0 * i, runtime="2.1.290") for i in range(3, 6)]) + case("R142_03", shifted, "HOST_BEHAVIOR_SHIFT", 30, 0, comparable=6) + + # ---------------------------------------------------------------- R142_04 + print("\nR142_04 - a record cannot claim a completeness its own numbers contradict") + impossible = [rec(i, 30.0 + 3.0 * i, sessions=0, turns=0, carry_bytes=0) for i in range(6)] + case("R142_04", impossible, "PARTIAL_EVIDENCE", 40, 0, comparable=0) + + # ---------------------------------------------------------------- R142_05 + print("\nR142_05 - a sweep that lost records to torn JSON is not COMPLETE") + clean = transcript(os.path.join(d, "clean.jsonl")) + torn = transcript(os.path.join(d, "torn.jsonl"), torn=True) + a_clean, a_torn = carry.accumulate([clean], min_turns=1), carry.accumulate([torn], min_turns=1) + check("control: a clean sweep is still COMPLETE", a_clean["quality"], "COMPLETE") + check("the torn record is counted", a_torn["malformed"], 1) + check("and the sweep is no longer COMPLETE", a_torn["quality"] != "COMPLETE", True) + check("the optimizer reads it as a loss, not a bound", + optimize.sweep_quality(a_torn), "DEGRADED") + check("control: the clean sweep reads COMPLETE", optimize.sweep_quality(a_clean), "COMPLETE") + check("a loss is not rescued by --accept-partial", + optimize.may_promote(optimize.sweep_quality(a_torn), True), False) + case("R142_05 (live, torn)", [], "PARTIAL_EVIDENCE", 40, 0, scan=[torn, "--min-turns", "1"]) + case("R142_05 (live, torn, --accept-partial)", [], "PARTIAL_EVIDENCE", 40, 0, + scan=[torn, "--min-turns", "1"], extra=["--accept-partial"]) + + # ---------------------------------------------------------------- R142_06 + print("\nR142_06 - one definition of a bounded sample") + paths = [] + for n, name in enumerate(("a.jsonl", "b.jsonl", "c.jsonl")): + p = transcript(os.path.join(d, name), turns=60, listing=(name == "a.jsonl")) + os.utime(p, (TS0 + n * 1000, TS0 + n * 1000)) # a oldest, c newest + paths.append(p) + check("the bound is the newest N", + [os.path.basename(p) for p in carry.bounded_paths(paths, 1)], ["c.jsonl"]) + check("and it does not depend on discovery order", + carry.bounded_paths(list(reversed(paths)), 1), carry.bounded_paths(paths, 1)) + check("an unbounded sweep keeps every source", carry.bounded_paths(paths, 0), paths) + _rc, j, _n = run([], scan=paths + ["--max-files", "1", "--min-turns", "1"]) + check("the listing scan reads the sweep's sample, not a discovery-order slice", + j["listing"], None) + _rc, j2, _n = run([], scan=paths + ["--min-turns", "1"]) + check("control: unbounded, the listing in the oldest transcript IS read", + bool(j2["listing"]), True) + + # ---------------------------------------------------------------- the bound itself + print("\nthe bound a caller asked for is not a loss (and is still not COMPLETE)") + bounded = carry.accumulate(paths, min_turns=1, max_files=1) + check("a bounded sweep records what it skipped", bounded["skipped_by_limit"], 2) + check("the producer labels it PARTIAL", bounded["quality"], "PARTIAL") + check("the optimizer reads a bound, not a loss", optimize.sweep_quality(bounded), "PARTIAL") + check("refused without the flag", + optimize.may_promote(optimize.sweep_quality(bounded), False), False) + check("adopted with it", optimize.may_promote(optimize.sweep_quality(bounded), True), True) + + # ---------------------------------------------------------------- U1 + print("\nU1 - a generation that never had the field cannot have defaulted to COMPLETE") + old = [rec(i, 30.0 + 3.0 * i, schema=1, counters=False) for i in range(6)] + case("U1", old, "PARTIAL_EVIDENCE", 40, 0, comparable=6, history_quality="UNKNOWN") + case("U1 (--accept-partial does not rescue it)", old, "PARTIAL_EVIDENCE", 40, 0, + extra=["--accept-partial"]) + + # ---------------------------------------------------------------- a retry cannot launder + print("\nR142_07 - a retry cannot upgrade what its own run_id saw") + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §7.6): the two copies below differ, + # so under the identity law they are a CONFLICT rather than a retry. The property is the same + # and the guarantee is stricter — no repeated id can upgrade what it saw — but it is now + # enforced by a boundary instead of by a quality floor. + same = [rec(i, 30.0 + 3.0 * i) for i in range(5)] + twice = dict(rec(5, 45.0), run_id="dup") + same += [dict(twice), dict(twice)] + hist_path = write(os.path.join(d, "same.jsonl"), same) + recs, rejected, _lines, ep = optimize.load_history(hist_path) + check("an identical copy is dropped as an observation", len(recs), 6) + check("and counted as a retry", rejected["duplicate run_id (retry)"], 1) + check("and it opens no boundary", optimize.damage_summary(ep)["boundaries"], 0) + dup = [rec(i, 30.0 + 3.0 * i) for i in range(5)] + dup.append(dict(twice)) + worse = rec(5, 45.0, quality="PARTIAL", skipped=7) + worse["run_id"] = "dup" + dup.append(worse) + hist_path = write(os.path.join(d, "dup.jsonl"), dup) + recs, rejected, _lines, ep = optimize.load_history(hist_path) + check("a copy that says something else is kept, not merged", len(recs), 7) + check("and it is NEVER called a retry", + rejected.get("duplicate run_id (retry)", 0), 0) + check("it is counted as an identity conflict", ep["run_id_conflicts"], 1) + check("and it cuts where it appears", optimize.damage_summary(ep)["boundaries"], 1) + check("so nothing before it can promote", + len(optimize.active_records(recs, ep)), 0) + case("R142_07", dup, "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") + case("R142_07 (--accept-partial is not an identity override)", dup, + "INSUFFICIENT_DATA", 20, 0, extra=["--accept-partial"]) + check("a schema_version this reader cannot name is not a newer one", + optimize.record_quality(dict(rec(0, 40.0), schema_version="2")), "UNKNOWN") + check("nor is a float one", + optimize.record_quality(dict(rec(0, 40.0), schema_version=2.0)), "UNKNOWN") + + # ------------------------------------------------- round-1 cross-family review findings + print("\nR142_08 - what a record must SHOW before its zeroes mean anything") + no_corpus = {"schema_version": 2, "record_type": "carry_run", "ts": TS0, + "evidence_quality": "COMPLETE", "unreadable": 0, "oversize": 0, + "skipped_by_limit": 0, "shares": {"Bash": 60.0, "Read": 40.0}, + "bpt": {"Bash": 1.0, "Read": 1.0}} + check("zero counters do not say a sweep happened", + optimize.record_quality(no_corpus), "UNKNOWN") + for missing in ("sessions", "turns", "carry_bytes"): + partial_rec = {k: v for k, v in rec(0, 40.0).items() if k != missing} + check(f"a record without {missing} cannot attest completeness", + optimize.record_quality(partial_rec), "UNKNOWN") + + print("\nR142_09 - one undateable source does not collapse the bound") + undated = [] + for n, name in enumerate(("x.jsonl", "y.jsonl", "z.jsonl")): + q = transcript(os.path.join(d, name), turns=5) + os.utime(q, (TS0 + n * 1000, TS0 + n * 1000)) # z newest + undated.append(q) + os.remove(undated[2]) # ...and now undateable + picked = [os.path.basename(q) for q in carry.bounded_paths(undated, 2)] + check("the datable sources still order by mtime", picked, ["y.jsonl", "x.jsonl"]) + check("and discovery order still does not matter", + carry.bounded_paths(list(reversed(undated)), 2), carry.bounded_paths(undated, 2)) + + print("\nR142_10 - counters that cannot describe one sweep") + check("more sessions than transcripts scanned", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=1)), "INVALID") + check("sessions out of a sweep that scanned nothing", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=0)), "INVALID") + check("...and the bound flag does not rescue that either", + optimize.may_promote(optimize.record_quality( + rec(0, 40.0, quality="PARTIAL", sessions=40, scanned=0, skipped=5)), True), False) + check("control: sessions within what was scanned", + optimize.record_quality(rec(0, 40.0, sessions=40, scanned=40)), "COMPLETE") + case("R142_10", [rec(i, 30.0 + 3.0 * i, sessions=40, scanned=1) for i in range(6)], + "PARTIAL_EVIDENCE", 40, 0, comparable=0) + + print("\nR142_11 - a torn line in the history file is lost evidence, not a footnote") + # Expectation updated by the B1 repair (docs/V142_COUNTEREXAMPLES.md §5.1): the loss still + # refuses the records it followed — nothing is promoted and nothing is written — but it is a + # BOUNDARY, not a verdict on the file, so the status is "no epoch to analyse" rather than a + # permanent PARTIAL_EVIDENCE that no later evidence could ever clear. + torn_hist = [json.dumps(r) for r in good_rows] + ['{"torn":'] + j = case("R142_11", torn_hist, "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") + check("the torn line is still counted", j["history"]["rejected"].get("unparseable line"), 1) + check("and the file's loss is named", (j["history"].get("damage") or {}).get("boundaries"), 1) + case("R142_11 (--accept-partial does not adopt a loss)", torn_hist, + "INSUFFICIENT_DATA", 20, 0, extra=["--accept-partial"]) + check("the records BEFORE the loss cannot support a promotion after it", + run([json.dumps(r) for r in good_rows] + ['{"torn":'] + + [json.dumps(rec(i, 30.0 + 3.0 * i)) for i in range(20, 22)])[1]["history"]["comparable"], + 2) + envelope_line = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}) + foreign = json.dumps({"note": "another tool's line in a shared file"}) + case("R142_11 control: a foreign line that parses is counted, not called damage", + [json.dumps(r) for r in good_rows] + [foreign], "CANDIDATE", 10, 1, + history_quality="COMPLETE") + case("R142_11 control: a refusal by design is not damage", + [json.dumps(r) for r in good_rows] + [envelope_line], "CANDIDATE", 10, 1, + history_quality="COMPLETE") + + print("\nR142_12 - a ledger that lost a line is not a field sample") + ledger_dir = tempfile.mkdtemp(dir=d) + clean_ledger = os.path.join(ledger_dir, "clean.jsonl") + with open(clean_ledger, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n") + torn_ledger = os.path.join(ledger_dir, "torn.jsonl") + with open(torn_ledger, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n" + '{"event":' + "\n") + clean = optimize.load_ledger(clean_ledger) + lost = optimize.load_ledger(torn_ledger) + check("control: a clean sample retires the guard", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), clean, None)], ["CANDIDATE"]) + check("a sample that lost a line only observes", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), lost, None)], ["OBSERVED"]) + check("a ledger that exists and cannot be read is not 'no ledger'", + (optimize.load_ledger(ledger_dir) or {}).get("rejected"), 1) + + print("\nR142_13 - a status that promised a candidate and could not write one") + fail_dir = tempfile.mkdtemp(dir=d) + findings = optimize.analyse(None, {"comparable": good_rows, "total": 6, "in_scope": 6, + "rejected": {}, "dropped": [], "time_order": "ok"}, + None, None) + blocked = [f["candidate_id"] for f in findings if f["state"] == "CANDIDATE"][0] + open(os.path.join(fail_dir, blocked), "w").write("a file where a directory must go") + hist_file = write(os.path.join(fail_dir, "h.jsonl"), good_rows) + argv = [sys.executable, OPT, "--history", hist_file, "--ledger", + os.path.join(fail_dir, "none.jsonl"), "--scan", "--emit-candidate", fail_dir, + "--json", "--strict-exit"] + proc = subprocess.run(argv, capture_output=True, text=True, timeout=300) + jf = json.loads(proc.stdout) + check("nothing landed, so the status does not claim it did", jf["status"], "INTERNAL_ERROR") + check("and the exit code follows the status", proc.returncode, 50) + check("the failure is named", len(jf["candidates_failed"]), 1) + + print("\nR142_14 - the scope fallback is a label on an empty run, not a way back in") + only_bad = [{"schema_version": 2, "record_type": "carry_run", "ts": TS0, "sessions": 0, + "scanned": 1, "scope_id": "attack", "evidence_quality": "INVALID", + "shares": {"Bash": 100.0}}] + j = case("R142_14 (nothing eligible anywhere)", only_bad, "PARTIAL_EVIDENCE", 40, 0, + comparable=0) + check("the scope it names is the one real record's scope", j["scope"]["analysed"], "attack") + prod = [dict(r, scope_id="prod", run_id="prod%d" % i) for i, r in enumerate(good_rows)] + j2 = case("R142_14 (an eligible population is never stranded by it)", prod + only_bad, + "CANDIDATE", 10, 1, comparable=6) + check("and the eligible population chooses the scope", j2["scope"]["analysed"], "prod") + + print("\nR142_15 - a rejected line is damage unless it is one of the named exceptions") + for line, label, status, quality in ( + ('{"schema_version":3,"record_type":"carry_run","shares":{"Bash":60.0,"Read":40.0}}', + "a record from a schema this reader does not know", "INSUFFICIENT_DATA", "EMPTY"), + ('{"schema_version":2,"record_type":"carry_run","shares":{"Bash":10.0}}', + "a carry record whose shares do not sum to a population", "INSUFFICIENT_DATA", + "EMPTY"), + ('{"note": "another tool\'s line in a shared file"}', + "a line that was never a carry record", "CANDIDATE", "COMPLETE")): + code = {"INSUFFICIENT_DATA": 20, "PARTIAL_EVIDENCE": 40, "CANDIDATE": 10}[status] + case(f"R142_15: {label}", [json.dumps(r) for r in good_rows] + [line], + status, code, 1 if status == "CANDIDATE" else 0, history_quality=quality) + if status != "CANDIDATE": + # the same rejected line BEFORE a healthy population: the loss cuts, it does not kill + case(f"R142_15: {label} — and a population after it still stands", + [line] + [json.dumps(r) for r in good_rows], "CANDIDATE", 10, 1, + comparable=6, history_quality="COMPLETE") + + print("\nR142_16 - a ledger line that is neither a write nor a denial is a line lost") + led_dir = tempfile.mkdtemp(dir=d) + unknown_event = os.path.join(led_dir, "unknown.jsonl") + with open(unknown_event, "w", encoding="utf-8") as fh: + fh.write("\n".join('{"event": "checked"}' for _ in range(100)) + "\n") + fh.write('{"event": "garbage"}\n') + led = optimize.load_ledger(unknown_event) + check("the unaccountable line is counted", (led["writes"], led["rejected"]), (100, 1)) + check("and the guard only observes", + [f["state"] for f in optimize.analyse(None, dict(EMPTY_HIST), led, None)], ["OBSERVED"]) + + print("\nR142_18 - round-4 findings: the live sweep, the shares reason, the temp file") + live_impossible = {"sessions": 40, "turns": 0, "scanned": 40, "unreadable": 0, "oversize": 0, + "skipped_by_limit": 0, "malformed": 0, "identity_changed": 0, + "conflicted_sources": 0, "quality": "COMPLETE"} + check("a live sweep with sessions and no turns is impossible too", + optimize.sweep_quality(live_impossible), "INVALID") + check("control: the same sweep with turns is COMPLETE", + optimize.sweep_quality(dict(live_impossible, turns=2000)), "COMPLETE") + claims_ours = json.dumps({"schema_version": 2, "record_type": "carry_run", "ts": TS0, + "sessions": 40, "turns": 1000, "carry_bytes": 10 ** 7}) + case("R142_18: a carry record with no shares key at all is a loss", + [json.dumps(r) for r in good_rows] + [claims_ours], + "INSUFFICIENT_DATA", 20, 0, history_quality="EMPTY") + case("R142_18: ...and a population after that loss still stands", + [claims_ours] + [json.dumps(r) for r in good_rows], "CANDIDATE", 10, 1, + comparable=6, history_quality="COMPLETE") + tmp_dir = tempfile.mkdtemp(dir=d) + blocked_id = [f["candidate_id"] for f in optimize.analyse( + None, {"comparable": good_rows, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"}, None, None) if f["state"] == "CANDIDATE"][0] + open(os.path.join(tmp_dir, blocked_id), "w").write("a file where a directory must go") + hist2 = write(os.path.join(tmp_dir, "h.jsonl"), good_rows) + subprocess.run([sys.executable, OPT, "--history", hist2, "--ledger", + os.path.join(tmp_dir, "none.jsonl"), "--scan", "--emit-candidate", tmp_dir, + "--json", "--strict-exit"], capture_output=True, text=True, timeout=300) + check("a failed write leaves no half-written file behind", + [f for _b, _dd, fs in os.walk(tmp_dir) for f in fs if ".tmp-" in f], []) + bound_dir = tempfile.mkdtemp(dir=d) + trio = [] + for n, name in enumerate(("p.jsonl", "q.jsonl", "r.jsonl")): + q = transcript(os.path.join(bound_dir, name), turns=30) + os.utime(q, (TS0 + n * 1000, TS0 + n * 1000)) + trio.append(q) + once = carry.bounded_paths(trio, 2) + swept = carry.accumulate(trio, min_turns=1, max_files=2, selected=once) + check("a caller can hand the sweep the selection it already made", + sorted(os.path.basename(x) for x in swept["parsed"]), + sorted(os.path.basename(x) for x in once)) + check("and the bound is still reported against the whole population", + swept["skipped_by_limit"], 1) + + # ============================================================ B1: the history epoch + # An unattributable loss cuts the promotion history at that physical position. Evidence before + # the cut never joins evidence after it; the loss stays reported; the newest epoch is, by + # construction, free of damage. docs/V142_COUNTEREXAMPLES.md §5 froze every row below. + print("\nB1 - a damaged history recovers, and the loss is still reported") + TORN = '{"schema_version": 2, "record_type": "carry_run", "ts": 1750000000, "sessions": 5' + FOREIGN = json.dumps({"note": "another tool's line in a shared file"}) + ENVELOPE = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, "payload": {}, + "certificate": {}}) + + def series(n, first=0, scope="default", per_week=3.0, base=30.0): + return [json.dumps(rec(first + i, base + per_week * i, scope=scope)) for i in range(n)] + + def b1(label, rows, status, code, files, comparable, epoch, quality, boundaries, extra=()): + rc, j, n = run(rows, extra=extra) + got = (j["status"], rc, n, j["history"]["comparable"], + j["scope"].get("records_in_epoch"), j["history"].get("quality"), + (j["history"].get("damage") or {}).get("boundaries")) + check(label, got, (status, code, files, comparable, epoch, quality, boundaries)) + + b1("B1_01 loss at the tail leaves no epoch to promote from", + series(6) + [TORN], "INSUFFICIENT_DATA", 20, 0, 0, 0, "EMPTY", 1) + b1("B1_02 an epoch too small to carry a trend", + series(6) + [TORN] + series(2, 20), "INSUFFICIENT_DATA", 20, 0, 2, 2, "COMPLETE", 1) + b1("B1_03 a sufficient post-loss epoch promotes on its own", + series(6) + [TORN] + series(6, 20), "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_04 and it still promotes sixty records later", + series(6) + [TORN] + series(60, 20, per_week=1.0, base=20.0), + "CANDIDATE", 10, 1, 60, 60, "COMPLETE", 1) + b1("B1_05 an unattributable loss cuts every scope (analysing the recovered one)", + series(6, scope="agent-a") + [TORN] + series(6, 20, scope="agent-b"), + "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_05b ...and the scope with nothing after the loss says so", + series(6, scope="agent-a") + [TORN] + series(6, 20, scope="agent-b"), + "INSUFFICIENT_DATA", 20, 0, 0, 0, "EMPTY", 1, extra=["--scope-id", "agent-a"]) + b1("B1_06 two boundaries: only the newest epoch is analysed", + series(6) + [TORN] + series(6, 20) + [TORN] + series(6, 40), + "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 2) + b1("B1_07 a loss before any record does not stop the file", + [TORN] + series(6), "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + b1("B1_09 a foreign line is not a loss", + series(6) + [FOREIGN], "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 0) + b1("B1_10 current-generation evidence refused by design is not a loss", + series(6) + [ENVELOPE], "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 0) + b1("B1_12 records the law calls INVALID are refused, not a loss", + [json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)], + "PARTIAL_EVIDENCE", 40, 0, 0, 6, "EMPTY", 0) + + # B1_08: a truncated final line, written the way a crash leaves one + trunc_dir = tempfile.mkdtemp(dir=d) + trunc = os.path.join(trunc_dir, "history.jsonl") + with open(trunc, "w", encoding="utf-8") as fh: + fh.write("\n".join(series(6)) + "\n") + fh.write('{"schema_version": 2, "record_type": "carry_run", "ts": 17500000') + out_dir = os.path.join(trunc_dir, "cand") + proc = subprocess.run([sys.executable, OPT, "--history", trunc, "--ledger", + os.path.join(trunc_dir, "none.jsonl"), "--scan", "--emit-candidate", + out_dir, "--json", "--strict-exit"], capture_output=True, text=True, + timeout=300) + jt = json.loads(proc.stdout) + # the emit directory itself is created by the LOCK, before any status is known; what the + # matrix froze is the candidate-FILE count + spec_files = [f for _b, _dd, fs in os.walk(out_dir) for f in fs if f != ".optimize.lock"] + check("B1_08 a truncated final line is the same loss", + (jt["status"], proc.returncode, len(spec_files), + (jt["history"].get("damage") or {}).get("boundaries")), + ("INSUFFICIENT_DATA", 20, 0, 1)) + + print("\nB1_11 / M2 - a zero-carry sweep is readable evidence, not corruption") + zero_carry = dict(rec(9, 0.0), shares={}, bpt={}, carry_bytes=0) + ok, why = optimize.valid_record(zero_carry) + check("the reader accepts the shape the producer writes", (ok, why), (True, "")) + check("and the quality law calls it EMPTY, not INVALID", + optimize.record_quality(zero_carry), "EMPTY") + b1("B1_11 one zero-carry record does not stop a healthy population", + series(6) + [json.dumps(zero_carry)], "CANDIDATE", 10, 1, 6, 7, "COMPLETE", 0) + b1("B1_11b a history of nothing but zero-carry records is not damage", + [json.dumps(dict(rec(i, 0.0), shares={}, bpt={}, carry_bytes=0)) for i in range(6)], + "INSUFFICIENT_DATA", 20, 0, 0, 6, "EMPTY", 0) + check("carry with no shares is still a contradiction", + optimize.record_quality(dict(rec(9, 40.0), carry_bytes=0)), "INVALID") + check("shares with no carry is still a contradiction", + optimize.record_quality(dict(rec(9, 40.0), shares={}, bpt={})), "INVALID") + # the producer's own words, executed: a session whose items all land on its last turn + zt = os.path.join(d, "zero_carry.jsonl") + with open(zt, "w", encoding="utf-8") as fh: + for i in range(5): + fh.write(json.dumps({"type": "assistant", "message": { + "id": f"t{i}", "usage": {"output_tokens": 3}, "content": []}}) + "\n") + fh.write(json.dumps({"type": "assistant", "message": { + "id": "last", "usage": {"output_tokens": 3}, + "content": [{"type": "tool_use", "name": "Bash", "input": {"command": "ls"}}]}}) + "\n") + zsweep = carry.accumulate([zt], min_turns=1) + check("the producer really can measure zero carry", + (zsweep["sessions"] > 0, sum(zsweep["carry"].values())), (True, 0)) + + print("\nB1_13 - one run_id on both sides of a boundary") + dup_before = series(5) + [json.dumps(dict(json.loads(series(1, 5)[0]), run_id="carried"))] + dup_after = [json.dumps(dict(json.loads(r), run_id="carried" if i == 0 else None)) + for i, r in enumerate(series(6, 20))] + dup_after = [json.dumps({k: v for k, v in json.loads(r).items() if v is not None}) + for r in dup_after] + b1("B1_13 the post-loss epoch keeps its own copy", + dup_before + [TORN] + dup_after, "CANDIDATE", 10, 1, 6, 6, "COMPLETE", 1) + same_epoch = series(5) + [json.dumps(dict(json.loads(series(1, 5)[0]), run_id="twin")), + json.dumps(dict(json.loads(series(1, 6)[0]), run_id="twin", + evidence_quality="PARTIAL", skipped_by_limit=9))] + rc, j, n = run(same_epoch) + # Frozen-decision amendment (§7.6): the twins differ, so they are an identity conflict. + check("B1_13b a copy that says something else cuts instead of lowering", + (j["history"]["quality"], j["status"], n, j["history"]["run_id_conflicts"], + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0)), + ("EMPTY", "INSUFFICIENT_DATA", 0, 1, 0)) + # both line orders, because which copy comes first is exactly what a crash decides + clean_then_worse = (series(5) + [json.dumps(dict(rec(5, 45.0), run_id="X"))] + [TORN] + + [json.dumps(dict(rec(6, 45.0), run_id="X", evidence_quality="PARTIAL", + skipped_by_limit=9))] + + series(5, 21)) + rc, j, n = run(clean_then_worse) + check("B1_13c the post-loss copy is judged on its OWN evidence", + j["history"]["quality"], "PARTIAL") + worse_then_clean = (series(5) + + [json.dumps(dict(rec(5, 45.0), run_id="X", evidence_quality="PARTIAL", + skipped_by_limit=9))] + [TORN] + + [json.dumps(dict(rec(6, 45.0), run_id="X"))] + series(5, 21)) + rc, j, n = run(worse_then_clean) + check("B1_13d and an excluded pre-loss copy does not poison it", + (j["history"]["quality"], j["status"] == "PARTIAL_EVIDENCE"), ("COMPLETE", False)) + # Frozen-decision amendment (docs/V142_COUNTEREXAMPLES.md §6.6): the middle copy used to carry + # unreadable=1, which under the trust model is a VALID record proving a loss — it now opens an + # epoch rather than sitting inside one. The property this case protects is unchanged and the + # fixture states it with copies that lose nothing; the degraded-copy behaviour is VD_10/VD_11. + three_in_one = series(4) + [json.dumps(dict(rec(4, 42.0), run_id="S")), + json.dumps(dict(rec(5, 45.0), run_id="S", + evidence_quality="PARTIAL", skipped_by_limit=4)), + json.dumps(dict(rec(6, 48.0), run_id="S"))] + rc, j, n = run(three_in_one) + # Frozen-decision amendment (§7.6): three DIFFERENT copies under one id are two conflicts. + # RID_15 is the frozen row for the same shape with a population after it. + check("B1_13e three different copies under one id are two conflicts", + (j["history"]["quality"], j["history"]["run_id_conflicts"], + (j["history"]["rejected"] or {}).get("duplicate run_id (retry)", 0), n), + ("EMPTY", 2, 0, 0)) + + print("\nB1_14/B1_15 - what a cross-family review of the epoch repair found") + # a finding that rests on the ledger alone may still promote after a loss — but nothing from + # before the loss may name it, because the scope travels into the candidate id + led_lines = ['{"event": "checked"}'] * 100 + stale = [json.dumps(dict(rec(i, 30.0 + 3.0 * i), scope_id="stale")) for i in range(6)] + ld = os.path.join(tempfile.mkdtemp(dir=d), "ledger.jsonl") + with open(ld, "w", encoding="utf-8") as fh: + fh.write("\n".join(led_lines) + "\n") + hd = tempfile.mkdtemp(dir=d) + hp = write(os.path.join(hd, "history.jsonl"), stale + [TORN]) + out14 = os.path.join(hd, "cand") + pr = subprocess.run([sys.executable, OPT, "--history", hp, "--ledger", ld, "--scan", + "--emit-candidate", out14, "--json", "--strict-exit"], + capture_output=True, text=True, timeout=300) + j14 = json.loads(pr.stdout) + check("B1_14 a stale population cannot name a candidate raised after the loss", + (j14["scope"]["analysed"], j14["scope"]["records_in_epoch"], + any("stale" in c for c in j14["candidate_ids"])), ("default", 0, False)) + check("B1_14 and the ledger finding itself still stands", + (j14["status"], [c.split("-")[0] for c in j14["candidate_ids"]]), ("CANDIDATE", ["noop"])) + + # two scopes can hold the same epoch NUMBERS; identity needs the scope as well + # Frozen-decision amendment (§6.6): the two boundaries used to be taken from REJECTED lines, + # which no longer name a scope at all. The shape this case needs — one scope-local cut in each + # of two scopes, so both carry the same epoch numbers — is built from the only trusted source + # there is: a validated record whose own loss counter proves it lost evidence. + broken_a = json.dumps(dict(rec(0, 20.0), scope_id="a", run_id="dega", unreadable=1)) + broken_b = json.dumps(dict(rec(0, 20.0), scope_id="b", run_id="degb", unreadable=1)) + rows_a = [json.dumps(dict(rec(i, 30.0 + 3.0 * i), scope_id="a", + run_id="shared" if i == 0 else f"a{i}")) for i in range(6)] + row_b = json.dumps(dict(rec(9, 50.0, quality="PARTIAL", skipped=1), scope_id="b", + run_id="shared")) + rc15, j15, n15 = run([broken_a] + rows_a + [broken_b, row_b], extra=["--scope-id", "a"]) + check("B1_15 a retry in another scope is not this scope's retry", + (j15["history"]["quality"], j15["history"]["comparable"], + j15["history"]["rejected"].get("duplicate run_id (retry)")), ("COMPLETE", 6, None)) + check("B1_15 and the loss in each scope cut only that scope", + (j15["history"]["damage"]["boundaries"], + sorted((j15["history"]["damage"]["scope_local"] or {}).items())), + (2, [("a", 1), ("b", 1)])) + + trusted_boundaries() + run_id_conflicts() + + print("\nM1 - history.quality describes the evidence eligible RIGHT NOW") + rc, j, n = run([json.dumps(rec(i, 30.0 + 3.0 * i, sessions=0, scanned=40)) for i in range(6)]) + check("nothing comparable is never reported as COMPLETE", + (j["history"]["comparable"], j["history"]["quality"]), (0, "EMPTY")) + + print("\nthe damage is reported even while the current epoch is clean") + rc, j, n = run(series(6) + [TORN] + series(6, 20)) + check("the rejection is still counted", j["history"]["rejected"].get("unparseable line"), 1) + check("and the file's historical damage is named", + ((j["history"].get("damage") or {}).get("boundaries"), + (j["history"].get("damage") or {}).get("file_global")), (1, 1)) + check("while the analysed epoch is clean and promotes", + (j["history"]["quality"], j["status"]), ("COMPLETE", "CANDIDATE")) + + # ---------------------------------------------------------------- positive controls + print("\npositive controls - a gate that refuses everything is not a gate") + good = good_rows + case("P1 clean COMPLETE population", good, "CANDIDATE", 10, 1, comparable=6, + history_quality="COMPLETE") + case("P3 an invalid record in ANOTHER scope does not poison this one", + good + [rec(9, 55.0, quality="INVALID", sessions=0, scope="other")], + "CANDIDATE", 10, 1, comparable=6) + envelope = json.dumps({"envelope": {"schema_version": 4, "run_id": "x"}, + "payload": {}, "certificate": {}}) + rc, j, n = run([envelope]) + check("P4 current-generation evidence stays unsupported", j["status"], "INSUFFICIENT_DATA") + check("P4 strict exit", rc, 20) + check("P4 refused by name, not consumed", + any("current-generation" in k for k in j["history"]["rejected"]), True) + check("P4 nothing written", n, 0) + check("P4 the supported schema set is unchanged", j["history_schema_supported"], [0, 1, 2]) + flat = [rec(i, 50.0) for i in range(6)] + case("S_NOACTION nothing moved", flat, "NO_ACTION", 0, 0, comparable=6) + case("S_LOCK another run holds the lock", good, "ALREADY_RUNNING", 41, 0, lock=True) + + # ---------------------------------------------------------------- the emission gate itself + print("\nthe gate is in the emitter, not only in its caller") + f = optimize.finding("x", "CANDIDATE", "head", "ev", scope="s", bucket=1) + for status in ("PARTIAL_EVIDENCE", "HOST_BEHAVIOR_SHIFT", "INSUFFICIENT_DATA", + "INTERNAL_ERROR", "ALREADY_RUNNING", "NO_ACTION"): + out = os.path.join(d, "gate_" + status) + w, e, fail = optimize.emit_candidates([f], out, status) + check(f"{status} writes nothing", (w, e, fail, os.path.exists(out)), ([], [], [], False)) + out = os.path.join(d, "gate_CANDIDATE") + w, _e, _f = optimize.emit_candidates([f], out, "CANDIDATE") + check("CANDIDATE still writes", (len(w), os.path.exists(out)), (1, True)) + + print(f"\n{P} PASS, {F} FAIL") + return 1 if F else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_evidence_phase2.py b/tests/test_evidence_phase2.py index edeed36..48afa82 100644 --- a/tests/test_evidence_phase2.py +++ b/tests/test_evidence_phase2.py @@ -334,7 +334,7 @@ def legacy_row(schema, scope="s", ts=None): # ---------------------------------------------------------------- OPTIMIZER ISOLATION p = os.path.join(d, "iso.jsonl") carry.history(p, facts(14), 100, scope_id="s") - records, rejected, _lines = optimize.load_history(p) + records, rejected, _lines, _ep = optimize.load_history(p) check("optimizer 1.3 tak menerima satu pun record generasi ini", len(records), 0) check("penolakannya terhitung, bukan senyap", sum(rejected.values()), 1) check("penolakannya menyebut skema", diff --git a/tests/test_multiagent.py b/tests/test_multiagent.py index 851b49e..b8e6aab 100644 --- a/tests/test_multiagent.py +++ b/tests/test_multiagent.py @@ -32,9 +32,14 @@ def check(label, got, want): def rec(ts, shares, scope="default", turns=1000, sessions=40, **kw): + # The acquisition counters are part of the record, not decoration: carry.history() has written + # them since schema 2, and since 1.4.2 a record that claims COMPLETE without them cannot carry + # a promotion (see tests/test_evidence_integrity.py). A fixture that omitted them was asserting + # a completeness no real sweep asserts that way. r = {"schema_version": 2, "record_type": "carry_run", "ts": ts, "sessions": sessions, "turns": turns, "carry_bytes": 10 ** 7, "scanned": 100, "scope_id": scope, "workload_class": "", "evidence_quality": "COMPLETE", + "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "run_id": kw.pop("run_id", None) or carry.new_run_id(), "shares": shares, "bpt": {k: 1.0 for k in shares}} r.update(kw) @@ -112,7 +117,7 @@ def main(): for i in range(6)] mixed += [rec(100 + i * 86400, {"Bash": 80.0, "Read": 20.0}, scope="reviewer") for i in range(6)] p = write(os.path.join(d, "mixed.jsonl"), mixed) - recs, _, _ = optimize.load_history(p) + recs, _, _, _ = optimize.load_history(p) check("dua scope terbaca utuh", len(recs), 12) groups = optimize.by_scope(recs) check("record dikelompokkan per scope", sorted(groups), ["builder", "reviewer"]) @@ -207,7 +212,7 @@ def main(): # WHY: phase 2 ports acquisition, not promotion. An optimizer that guessed at a # current-generation record would be reading fields whose meaning it does not know — # the exact "legacy gains current trust" failure, in the other direction. - recs, rej, _ = optimize.load_history(hp) + recs, rej, _, _ = optimize.load_history(hp) check(f"{n} penulis paralel: optimizer 1.3 menolak skema baru, fail closed", (len(recs), sum(rej.values())), (0, n)) @@ -295,12 +300,27 @@ def main(): far["candidate_id"] != f1["candidate_id"], True) check("candidate_id membawa scope-nya", f1["candidate_id"].split("-")[-2], "x") + # Lock the rule the fixture above now satisfies: the SAME record without its counters claims a + # completeness it cannot show, and a population of those cannot promote anything. + bare = [{k: v for k, v in r.items() if k not in ("unreadable", "oversize", "skipped_by_limit")} + for r in [rec(100 + i * 86400 * 7, {"Bash": 30.0 + i * 8, "Read": 70.0 - i * 8}, + scope="bare") for i in range(6)]] + keep_bare, _dropped_bare = optimize.comparable(bare) + check("record tanpa counters: tetap dibandingkan, tapi kualitasnya UNKNOWN", + (len(keep_bare), optimize.history_quality(keep_bare)), (6, "UNKNOWN")) + h_bare = {"comparable": keep_bare, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"} + check("dan UNKNOWN tak melahirkan kandidat, dgn atau tanpa --accept-partial", + {x["state"] for x in optimize.analyse(None, h_bare, None, None, scope="bare")} + | {x["state"] for x in optimize.analyse(None, h_bare, None, None, scope="bare", + accept_partial=True)}, {"OBSERVED"}) + out2 = os.path.join(d, "cand_dedup") - w, e, _fail = optimize.emit_candidates([f1], out2) + w, e, _fail = optimize.emit_candidates([f1], out2, "CANDIDATE") check("emisi pertama menulis", (len(w), len(e)), (1, 0)) - w, e, _fail = optimize.emit_candidates([f2], out2) + w, e, _fail = optimize.emit_candidates([f2], out2, "CANDIDATE") check("emisi kedua dgn bukti setara: EXISTING, nol penulisan ulang", (len(w), len(e)), (0, 1)) - w, e, _fail = optimize.emit_candidates([far], out2) + w, e, _fail = optimize.emit_candidates([far], out2, "CANDIDATE") check("bukti yang benar-benar bergerak: kandidat baru ditulis", (len(w), len(e)), (1, 0)) # ------------------------------------------------------------ 8b. identitas tren stabil @@ -483,11 +503,11 @@ def deny(*a, **k): h1 = [rec(100 + i * 86400, {"Bash": 40.0, "Read": 60.0}, scope="fleet") for i in range(3)] h2 = [rec(100 + i * 86400, {"Bash": 41.0, "Read": 59.0}, scope="fleet") for i in range(3)] merged = write(os.path.join(d, "fleet.jsonl"), h1 + h2) # dua host, satu scope, digabung - recs, rej, _ = optimize.load_history(merged) + recs, rej, _, _ = optimize.load_history(merged) check("record dua host dalam satu scope bisa digabung tanpa tabrakan", len(recs), 6) check("nol tolakan saat penggabungan", sum(rej.values()), 0) dup = write(os.path.join(d, "fleet_dup.jsonl"), h1 + h2 + h1) # rsync menyalin dua kali - recs, rej, _ = optimize.load_history(dup) + recs, rej, _, _ = optimize.load_history(dup) check("penggabungan ulang idempoten (run_id sama = satu observasi)", len(recs), 6) check("salinan ganda dihitung sebagai retry", rej["duplicate run_id (retry)"], 3) @@ -509,7 +529,7 @@ def deny(*a, **k): big = os.path.join(d, "budget.jsonl") write(big, [rec(100 + i * 3600, {"Bash": 50.0, "Read": 50.0}, scope="b") for i in range(20000)]) t0 = time.time() - recs, _, _ = optimize.load_history(big) + recs, _, _, _ = optimize.load_history(big) dt = time.time() - t0 check("20k record terbaca", len(recs), 20000) check(f"20k record dalam waktu terbatas (terukur {dt:.1f} s, plafon 120 s)", dt < 120, True) diff --git a/tests/test_mutation.py b/tests/test_mutation.py index 7d89c17..96f9fa3 100644 --- a/tests/test_mutation.py +++ b/tests/test_mutation.py @@ -41,14 +41,19 @@ def acc(sessions=40, turns=1000, quality="COMPLETE", carry_map=None): "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "quality": quality, "carry": collections.Counter(carry_map or {{"Bash": 90, "Read": 10}})}} def rec(ts, shares, scope="default", turns=1000, run_id=None): + # acquisition counters included: since 1.4.2 a record claiming COMPLETE without the + # counters its writer always wrote is UNKNOWN, and UNKNOWN cannot promote return {{"schema_version": 2, "record_type": "carry_run", "ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10**7, "scope_id": scope, "workload_class": "", "evidence_quality": "COMPLETE", "run_id": run_id or carry.new_run_id(), + "scanned": 100, "unreadable": 0, "oversize": 0, "skipped_by_limit": 0, "shares": shares, "bpt": {{k: 1.0 for k in shares}}}} def w(path, rows): - with open(path, "w", encoding="utf-8") as fh: + with open(path, "wb") as fh: for r in rows: - fh.write((r if isinstance(r, str) else json.dumps(r)) + chr(10)) + if not isinstance(r, bytes): + r = (r if isinstance(r, str) else json.dumps(r)).encode() + fh.write(r + chr(10).encode()) return path import evidence_acquire, evidence_history from evidence.absence import Absence @@ -113,16 +118,20 @@ def mutant(pairs): """), ("identitas run: dua agen dgn metrik identik = dua pengamatan", - [("optimize.py", 'rid = r.get("run_id")', 'rid = json.dumps(r.get("shares"), sort_keys=True)')], + [("optimize.py", ' rid = rec.get("run_id")\n if not (isinstance(rid, str) and rid):\n return None', + ' rid = json.dumps(rec.get("shares"), sort_keys=True)')], """ p = w(os.path.join(D, "h.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}), rec(100, {"Bash": 50.0, "Read": 50.0})]) - recs, rej, _ = optimize.load_history(p) + recs, rej, _l, ep = optimize.load_history(p) assert len(recs) == 2, "populasi menyusut: dua agen dihitung satu" + # ...dan identitas yang dikarang dari metrik juga tak boleh melahirkan konflik identitas + assert ep["run_id_conflicts"] == 0, ep + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) """), ("fail-closed: bukti PARTIAL tak boleh melahirkan kandidat", - [("optimize.py", 'usable = accept_partial or quality == "COMPLETE"', "usable = True")], + [("optimize.py", "usable = may_promote(quality, accept_partial)", "usable = True")], """ f = optimize.analyse(acc(quality="PARTIAL"), EMPTY_HIST, None, None, scope="s") assert [x["state"] for x in f] == ["OBSERVED"], "kandidat lahir dari sapuan setengah jadi" @@ -148,8 +157,8 @@ def mutant(pairs): """ f = optimize.analyse(acc(), EMPTY_HIST, None, None, scope="x")[0] out = os.path.join(D, "c") - optimize.emit_candidates([f], out) - w2, e2, _ = optimize.emit_candidates([f], out) + optimize.emit_candidates([f], out, "CANDIDATE") + w2, e2, _ = optimize.emit_candidates([f], out, "CANDIDATE") assert (len(w2), len(e2)) == (0, 1), "penjadwal menulis ulang usulan yang sama tiap siklus" """), @@ -347,7 +356,7 @@ def mutant(pairs): """ p = os.path.join(D, "h.jsonl") carry.history(p, facts(1), 100, scope_id="s") - recs, rej, _ = optimize.load_history(p) + recs, rej, _, _ = optimize.load_history(p) assert (len(recs), sum(rej.values())) == (0, 1), (len(recs), dict(rej)) """), @@ -375,8 +384,8 @@ def mutant(pairs): """), ("identitas tren stabil: jendela membesar bukan usulan baru", - [("optimize.py", " bucket=1 if per_month > 0 else -1,", - " bucket=bucket_of(abs(per_month)),")], + [("optimize.py", " direction = 1 if per_month > 0 else -1", + " direction = bucket_of(abs(per_month))")], """ # Deret harus NAIK lalu MENDATAR. Deret linier sempurna punya kemiringan yang sama di # jendela mana pun, jadi ia tak bisa membedakan identitas-dari-arah dari @@ -538,6 +547,878 @@ def hist_of(n): c = evidence_history.read_container(p) assert c.lines_rejected == 1, c.lines_rejected """), + + # ------------------------------------ v1.4.2: the legacy optimizer's evidence-integrity gate + ("M_PARTIAL_PROMOTES: riwayat PARTIAL tak boleh dipromosikan tanpa izin eksplisit", + [("optimize.py", + 'return quality == "COMPLETE" or (quality == "PARTIAL" and bool(accept_partial))', + "return True")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + for r in rows: + r["evidence_quality"] = "PARTIAL" + r["skipped_by_limit"] = 5 + keep, dropped = optimize.comparable(rows) + h = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": dropped, + "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="p") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + st = optimize.overall_status(f, h, None, optimize.population(keep), False) + assert st == "PARTIAL_EVIDENCE", st + """), + + ("M_INVALID_ANCHOR: jangkar komparabilitas hanya dari rekaman yang lolos filternya sendiri", + [("optimize.py", + 'usable = [r for r in recs if record_quality(r) not in ("INVALID", "EMPTY")]', + "usable = list(recs)")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + bad = rec(100 + 9 * 604800, {"Bash": 55.0, "Read": 45.0}, scope="a", turns=100000) + bad["evidence_quality"] = "INVALID" + bad["sessions"] = 0 + keep, _d = optimize.comparable(rows + [bad]) + assert len(keep) == 6, len(keep) + """), + + ("M_HOST_SHIFT_WRITES: status yang menolak promosi tidak menulis berkas", + [("optimize.py", ' if status != "CANDIDATE":\n return written, existing, failed\n', + "")], + """ + f = optimize.finding("x", "CANDIDATE", "h", "e", scope="s", bucket=1) + out = os.path.join(D, "gate") + for st in ("HOST_BEHAVIOR_SHIFT", "PARTIAL_EVIDENCE", "INSUFFICIENT_DATA", "ALREADY_RUNNING"): + w, e, fail = optimize.emit_candidates([f], out, st) + assert (w, e, fail) == ([], [], []), (st, w, e, fail) + assert not os.path.exists(out), st + w, e, fail = optimize.emit_candidates([f], out, "CANDIDATE") + assert len(w) == 1, (w, e, fail) + """), + + ("M_MALFORMED_COMPLETE: transcript dengan baris JSON robek bukan sapuan COMPLETE", + [("carry.py", + 'LOSS_FIELDS = ("unreadable", "oversize", "malformed", "identity_changed", "conflicted_sources")', + 'LOSS_FIELDS = ("unreadable", "oversize", "identity_changed", "conflicted_sources")')], + """ + d = tempfile.mkdtemp() + p = os.path.join(d, "t.jsonl") + rows = [] + for t in range(30): + rows.append(json.dumps({"type": "assistant", "message": {"id": "m%d" % t, + "usage": {"output_tokens": 5}, + "content": [{"type": "tool_use", "name": "Bash", + "input": {"command": "ls"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 200}]}})) + rows[11] = chr(123) + '"type": "assistant", "message": {"usage": {"out' + open(p, "w").write(chr(10).join(rows) + chr(10)) + a = carry.accumulate([p], min_turns=1) + assert a["malformed"] == 1, a["malformed"] + assert a["quality"] != "COMPLETE", a["quality"] + assert optimize.sweep_quality(a) == "DEGRADED", optimize.sweep_quality(a) + """), + + ("M_BOUND_SAMPLE_DIVERGES: sapuan dan pindaian listing memakai sampel terbatas yang SAMA", + [("optimize.py", " listing, uses, sess = skills_tool.scan(selected)", + " listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths)")], + """ + import subprocess + d = tempfile.mkdtemp() + paths = [] + for n, name in enumerate(("a.jsonl", "b.jsonl", "c.jsonl")): + q = os.path.join(d, name) + rows = [] + if name == "a.jsonl": + rows.append(json.dumps({"type": "user", "attachment": {"type": "skill_listing", + "content": chr(10).join(["- alpha: does a thing described at length", + "- beta: does another thing, also at length", + ""])}})) + for t in range(40): + rows.append(json.dumps({"type": "assistant", "message": {"id": "m%d" % t, + "usage": {"output_tokens": 5}, + "content": [{"type": "tool_use", "name": "Bash", + "input": {"command": "ls"}}]}})) + rows.append(json.dumps({"type": "user", "message": { + "content": [{"type": "tool_result", "content": "o" * 200}]}})) + open(q, "w").write(chr(10).join(rows) + chr(10)) + os.utime(q, (1750000000 + n * 1000, 1750000000 + n * 1000)) + paths.append(q) + empty = os.path.join(d, "none.jsonl") + open(empty, "w").write("") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", empty, "--ledger", empty, "--json", + "--scan"] + paths + ["--max-files", "1", "--min-turns", "1"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["listing"] is None, j["listing"] + r2 = subprocess.run([sys.executable, opt, "--history", empty, "--ledger", empty, "--json", + "--scan"] + paths + ["--min-turns", "1"], + capture_output=True, text=True, timeout=300) + assert json.loads(r2.stdout)["listing"], "kontrol: sapuan penuh HARUS membaca listing itu" + """), + + ("M_TREND_QUALITY_BYPASS: tren dari riwayat tak layak dilaporkan, bukan dipromosikan", + [("optimize.py", " if not hist_usable:", " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="q") + for i in range(6)] + for r in rows: + r["evidence_quality"] = "PARTIAL" + r["skipped_by_limit"] = 5 + keep, dropped = optimize.comparable(rows) + h = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": dropped, + "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="q") + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + g = optimize.analyse(None, h, None, None, scope="q", accept_partial=True) + assert [x["state"] for x in g] == ["CANDIDATE"], [x["state"] for x in g] + """), + + + ("M_DEDUP_LAUNDERS: retry dgn run_id sama tak boleh menaikkan kualitas yang bertahan", + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return TRUE_RETRY")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="d") + for i in range(5)] + good = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="d", run_id="dup") + bad = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="d", run_id="dup") + bad["evidence_quality"] = "PARTIAL" + bad["skipped_by_limit"] = 7 + p = w(os.path.join(D, "dup.jsonl"), rows + [good, bad]) + recs, rej, _l, ep = optimize.load_history(p) + # salinan yang MENGAKU sesuatu yang lain bukan retry: ia konflik identitas, dan ia memotong. + assert ep["run_id_conflicts"] == 1, ep + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + assert optimize.damage_summary(ep)["boundaries"] == 1, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 0, "populasi pra-konflik masih dipakai" + """), + + + ("M_IMPOSSIBLE_COUNTERS: sesi lebih banyak daripada transcript yang dipindai = mustahil", + [("optimize.py", ' if "scanned" in nums and nums["scanned"] < nums.get("sessions", 0):', + " if False:")], + """ + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r["scanned"] = 1 + q = optimize.record_quality(r) + assert q == "INVALID", q + """), + + ("M_CONTAINER_DAMAGE_IGNORED: baris robek di berkas history memotong epoch", + [("optimize.py", " return not any(x in str(reason) for x in NOT_A_LOSS)", + " return False")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="c") + for i in range(6)] + p = w(os.path.join(D, "torn_hist.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 0, ("epoch tak terpotong", len(cur)) + keep, dropped = optimize.comparable(cur) + h = {"comparable": keep, "total": len(recs), "in_scope": len(recs), "rejected": rej, + "dropped": dropped, "time_order": "ok"} + f = optimize.analyse(None, h, None, None, scope="c", accept_partial=True) + assert all(x["state"] != "CANDIDATE" for x in f), [x["state"] for x in f] + """), + + ("M_LEDGER_TORN_PROMOTES: ledger yang kehilangan baris bukan sampel lapangan", + [("optimize.py", """ ledger_usable = (bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES + and not ledger.get("rejected"))""", + " ledger_usable = bool(ledger) and ledger.get(\"writes\", 0) >= LEDGER_MIN_WRITES")], + """ + p = os.path.join(D, "led.jsonl") + with open(p, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + fh.write('{"event":' + chr(10)) + led = optimize.load_ledger(p) + assert led["writes"] == 100 and led["rejected"] == 1, led + EMPTY = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + f = optimize.analyse(None, EMPTY, led, None) + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + """), + + ("M_EMIT_FAILURE_SILENT: kandidat yang gagal ditulis tak boleh dilaporkan CANDIDATE", + [("optimize.py", """ if failed and not written:""", " if False:")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="e") + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), rows) + empty = os.path.join(d, "none.jsonl") + open(empty, "w").write("") + out = os.path.join(d, "cand") + os.makedirs(out) + keep, _dr = optimize.comparable(rows) + hh = {"comparable": keep, "total": 6, "in_scope": 6, "rejected": {}, "dropped": [], + "time_order": "ok"} + cid = [x["candidate_id"] for x in optimize.analyse(None, hh, None, None, scope="e") + if x["state"] == "CANDIDATE"][0] + open(os.path.join(out, cid), "w").write("a file where a directory must go") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", + "--emit-candidate", out, "--json", "--strict-exit"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["status"] == "INTERNAL_ERROR", (j["status"], j["candidates_failed"]) + assert r.returncode == 50, r.returncode + """), + + + ("M_BARE_PARTIAL_CLAIM: klaim PARTIAL tanpa counter yang menjelaskannya bukan bound", + [("optimize.py", """ if claimed == "PARTIAL" and derived == "COMPLETE":""", " if False:")], + """ + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r["evidence_quality"] = "PARTIAL" + q = optimize.record_quality(r) + assert q == "UNKNOWN", q + assert optimize.may_promote(q, True) is False, q + """), + + ("M_RECORD_LOSS_IGNORED: counter kehilangan pada record ikut menentukan kualitas", + [("optimize.py", + 'RECORD_LOSS_COUNTERS = ("unreadable", "oversize", "malformed", "malformed_lines",\n "identity_changed", "conflicted_sources", "records_rejected")', + 'RECORD_LOSS_COUNTERS = ("unreadable", "oversize")')], + """ + for counter in ("malformed", "identity_changed", "conflicted_sources"): + r = rec(100, {"Bash": 60.0, "Read": 40.0}) + r[counter] = 1 + q = optimize.record_quality(r) + assert q == "DEGRADED", (counter, q) + """), + + ("M_STRUCTURAL_REJECT_NOT_DAMAGE: penolakan struktural memotong epoch, bukan sekadar dicatat", + [("optimize.py", + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares",\n "not our record_type")', + 'NOT_A_LOSS = ("unsupported schema_version", "shares sum to")')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="g") + for i in range(6)] + bad = chr(123) + '"schema_version":3,"record_type":"carry_run","shares":' \ + + chr(123) + '"Bash":60.0,"Read":40.0' + chr(125) + chr(125) + p = w(os.path.join(D, "struct.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 0, ("penolakan struktural tak memotong epoch", len(cur)) + """), + + ("M_LEDGER_UNKNOWN_EVENT: baris ledger yang tak terhitung adalah baris yang hilang", + [("optimize.py", """ else: + # Neither a write nor a prevented write: a line this reader cannot account for. + # Counting it as nothing at all let a ledger full of unknown events look like a + # clean sample. (cross-family review, confirmation round) + rejected += 1""", " else:\n pass")], + """ + p = os.path.join(D, "unknown_led.jsonl") + with open(p, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + fh.write('{"event": "garbage"}' + chr(10)) + led = optimize.load_ledger(p) + assert led["rejected"] == 1, led + EMPTY = {"comparable": [], "total": 0, "in_scope": 0, "rejected": {}, "dropped": [], + "time_order": "ok"} + f = optimize.analyse(None, EMPTY, led, None) + assert [x["state"] for x in f] == ["OBSERVED"], [x["state"] for x in f] + """), + + + # ------------------------------------------- B1: the history epoch (acceptance-review blocker) + ("M_DAMAGE_POISONS_FOREVER: kerusakan lama tak boleh memblokir epoch yang bersih", + [("optimize.py", " hist_q = history_quality(hist.get(\"comparable\", []))", + " hist_q = worst_quality([history_quality(hist.get(\"comparable\", [])),\n" + " \"DEGRADED\" if (hist.get(\"damage\") or {}).get(\"boundaries\") " + "else \"COMPLETE\"])")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="p") + for i in range(6)] + p = w(os.path.join(D, "recover.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + keep, dropped = optimize.comparable(cur) + h = {"comparable": keep, "total": len(recs), "in_scope": len(cur), "rejected": rej, + "dropped": dropped, "time_order": "ok", + "damage": optimize.damage_summary(ep)} + f = optimize.analyse(None, h, None, None, scope="p") + assert any(x["state"] == "CANDIDATE" for x in f), [x["state"] for x in f] + """), + + ("M_DAMAGE_IGNORED_COMPLETELY: kehilangan yang tak teratribusi tetap memotong", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if False:\n cuts_file += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="q") + for i in range(6)] + p = w(os.path.join(D, "tail.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 0, "epoch tak terpotong" + """), + + ("M_EPOCH_MERGES_ACROSS_GAP: dua epoch tak boleh digabung", + [("optimize.py", " scoped = [r for r in current if scope_of(r) == scope]", + " scoped = [r for r in recs if scope_of(r) == scope]")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + empty = os.path.join(d, "none.jsonl") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["history"]["comparable"] == 6, j["history"]["comparable"] + assert j["scope"]["records_in_epoch"] == 6, j["scope"]["records_in_epoch"] + """), + + ("M_SCOPE_DAMAGE_GLOBALIZED: kerusakan milik satu scope tak memotong scope lain", + [("optimize.py", ' return (e.get("file_global", 0), (e.get("scope_local") or {}).get(scope, 0))', + ' return (e.get("file_global", 0) + sum((e.get("scope_local") or {}).values()), 0)')], + """ + good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + deg_b = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="b") + deg_b["unreadable"] = 2 + p = w(os.path.join(D, "scoped.jsonl"), [json.dumps(r) for r in good_a] + [json.dumps(deg_b)]) + recs, rej, _l, ep = optimize.load_history(p) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "a"] + assert len(cur) == 6, ("scope a ikut terpotong", len(cur)) + """), + + ("M_UNATTRIBUTABLE_DAMAGE_SCOPED: baris robek memotong SEMUA scope", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " cuts_scope[scope_of(o) if isinstance(o, dict) else 'default'] += 1")], + """ + good_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + p = w(os.path.join(D, "global.jsonl"), + [json.dumps(r) for r in good_a] + [chr(123) + '"torn":']) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 0, "scope a lolos dari potongan global" + """), + + ("M_EMPTY_COMPARABLE_REPORTS_COMPLETE: nol rekaman layak bukan COMPLETE", + [("optimize.py", ' return "EMPTY"\n return worst_quality([record_quality(r) for r in records])', + ' return "COMPLETE"\n return worst_quality([record_quality(r) for r in records])')], + """ + assert optimize.history_quality([]) == "EMPTY", optimize.history_quality([]) + """), + + ("M_ZERO_CARRY_BECOMES_DAMAGE: sapuan tanpa carry itu bukti terbaca, bukan korupsi", + [("optimize.py", ' if not claims_ours:\n return False, "no shares"\n sh = {}', + ' return False, "carry record without shares"')], + """ + zero = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}) + zero["shares"] = {} + zero["bpt"] = {} + zero["carry_bytes"] = 0 + ok, why = optimize.valid_record(zero) + assert ok, ("record zero-carry ditolak", why) + assert optimize.record_quality(zero) == "EMPTY", optimize.record_quality(zero) + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}) for i in range(6)] + p = w(os.path.join(D, "zero.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(zero)]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + """), + + ("M_PRE_DAMAGE_RECORD_ANCHORS: jangkar datang dari epoch yang sedang dianalisis", + [("optimize.py", " anchor = (eligible_anchor(current)", + " anchor = (eligible_anchor(recs) or eligible_anchor(current)")], + """ + import subprocess + d = tempfile.mkdtemp() + # jam mundur: rekaman SEBELUM potongan punya ts paling baru, tapi epoch adalah POSISI FISIK + old_scope = [rec(9_000_000_000 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, + scope="stale") for i in range(6)] + new_scope = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, + scope="fresh") for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), + [json.dumps(r) for r in old_scope] + [chr(123) + '"torn":'] + + [json.dumps(r) for r in new_scope]) + empty = os.path.join(d, "none.jsonl") + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", empty, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["scope"]["analysed"] == "fresh", j["scope"]["analysed"] + assert j["status"] == "CANDIDATE", (j["status"], j["history"]["comparable"]) + """), + + + ("M_STALE_SCOPE_ANCHOR: populasi pra-loss tak boleh menamai kandidat pasca-loss", + [("optimize.py", + " anchor = (eligible_anchor(current)\n" + " or (max(current, key=lambda r: r.get(\"ts\") or 0) if current else None))\n" + " scope = scope_of(anchor) if anchor is not None else \"default\"", + " anchor = (eligible_anchor(current) or eligible_anchor(recs)\n" + " or max(recs, key=lambda r: r.get(\"ts\") or 0))\n" + " scope = scope_of(anchor)")], + """ + import subprocess + d = tempfile.mkdtemp() + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="stale") + for i in range(6)] + h = w(os.path.join(d, "h.jsonl"), [json.dumps(r) for r in rows] + [chr(123) + '"torn":']) + led = os.path.join(d, "led.jsonl") + with open(led, "w") as fh: + fh.write(chr(10).join('{"event": "checked"}' for _ in range(100)) + chr(10)) + opt = os.path.join(os.path.dirname(carry.__file__), "optimize.py") + r = subprocess.run([sys.executable, opt, "--history", h, "--ledger", led, "--scan", "--json"], + capture_output=True, text=True, timeout=300) + j = json.loads(r.stdout) + assert j["scope"]["analysed"] == "default", j["scope"]["analysed"] + assert not any("stale" in c for c in j["candidate_ids"]), j["candidate_ids"] + """), + + ("M_DEDUP_IGNORES_SCOPE: dua scope bisa punya nomor epoch yang sama", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (rec.get(EPOCH_KEY, (0, 0))[0], rid)')], + """ + def degraded(scope): + x = rec(100, {"Bash": 20.0, "Read": 80.0}, scope=scope) + x["unreadable"] = 1 + return json.dumps(x) + broken_a = degraded("a") + broken_b = degraded("b") + rows_a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a", + run_id="shared" if i == 0 else "a%d" % i) for i in range(6)] + row_b = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="b", run_id="shared") + row_b["evidence_quality"] = "PARTIAL" + row_b["skipped_by_limit"] = 1 + p = w(os.path.join(D, "twoscope.jsonl"), + [broken_a] + [json.dumps(r) for r in rows_a] + [broken_b, json.dumps(row_b)]) + recs, rej, _l, ep = optimize.load_history(p) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "a"] + q = optimize.history_quality(optimize.comparable(cur)[0]) + assert q == "COMPLETE", (q, rej) + # run_id yang sama di scope LAIN bukan pengamatan yang sama: bukan retry, bukan konflik + assert ep["run_id_conflicts"] == 0, ep + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + """), + + # --------------------------- the trusted damage boundary (adversarial-review blocker B-UTF8) + ("M_UTF8_REPLACEMENT_ATTRIBUTED: byte yang tak ter-decode bukan record, dan bukan label", + [("optimize.py", ' text = raw.decode("utf-8")', + ' text = raw.decode("utf-8", "replace")')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + # sah dalam segala hal KECUALI byte-nya: dengan decode longgar ia jadi record diterima + ghost = rec(100 + 9 * 604800, {"Bash": 50.0, "Read": 50.0}, scope="@@M@@") + bad = json.dumps(ghost).encode().replace("@@M@@".encode(), bytes([255, 254, 128])) + p = w(os.path.join(D, "utf8.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("line is not valid UTF-8") == 1, dict(rej) + assert len(recs) == 6, ("baris tak ter-decode diterima sebagai record", len(recs)) + assert optimize.damage_summary(ep)["file_global"] == 1, optimize.damage_summary(ep) + assert all(chr(65533) not in optimize.scope_of(r) for r in recs), "U+FFFD masuk sebagai scope" + """), + + ("M_FOREIGN_RECORD_TYPE_IS_DAMAGE: entri alat lain di berkas bersama bukan kehilangan kita", + [("optimize.py", + ' return False, ("unknown record_type" if claims_ours else "not our record_type")', + ' return False, "unknown record_type"')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + foreign = json.dumps({"record_type": "hermes_run", "ts": 100, "note": "bukan record kita"}) + p = w(os.path.join(D, "foreign_type.jsonl"), [json.dumps(r) for r in rows] + [foreign]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "populasi rev ikut terpotong" + # kontrol: baris yang MENGAKU record kita dengan record_type asing TETAP kehilangan + ours = json.dumps({"record_type": "carry_note", "schema_version": 2, "run_id": "x", + "shares": {"Bash": 100.0}}) + p2 = w(os.path.join(D, "ours_type.jsonl"), [json.dumps(r) for r in rows] + [ours]) + recs2, rej2, _l2, ep2 = optimize.load_history(p2) + assert optimize.damage_summary(ep2)["file_global"] == 1, optimize.damage_summary(ep2) + """), + + ("M_SCOPE_LABEL_UNCHECKED: scope_id adalah otoritas atribusi, jadi diperiksa sebelum diterima", + [("optimize.py", + " if sid is not None and (not isinstance(sid, str) or len(sid) > 64):", + " if False:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + forged = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, run_id="LOSS-X") + forged["unreadable"] = 2 + forged["scope_id"] = ["rev"] + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="rev") for i in range(6)] + p = w(os.path.join(D, "forged_scope.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(forged)] + + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["file_global"] == 1, d + assert d["scope_local"] == {}, d + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("populasi rev menyeberangi loss", len(cur)) + assert sorted({optimize.scope_of(r) for r in recs}) == ["rev"], "scope hantu terbit" + # kontrol: label yang MEMANG ditulis produser tetap diterima + fine = rec(100, {"Bash": 50.0, "Read": 50.0}, scope="y" * 64) + assert optimize.valid_record(fine) == (True, ""), optimize.valid_record(fine) + """), + + ("M_REJECTED_VALUE_ECHOED: isi baris yang ditolak tak boleh masuk output publik", + [("optimize.py", ' if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)):\n' + ' return repr(v)\n return "of type " + type(v).__name__', + " return repr(v)")], + """ + secret = "sk-synthetic-NOTAREALKEY-0123456789" + bad = rec(100, {"Bash": 10.0, "Read": 10.0}, scope="/home/synthetic/.ssh/id_ed25519") + bad["schema_version"] = secret + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + p = w(os.path.join(D, "leak.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + assert all(secret[:12] not in k for k in rej), list(rej) + assert list(rej) == ["unsupported schema_version of type str"], list(rej) + """), + + ("M_REJECTED_SCOPE_TRUSTED: record yang gagal validasi bukan otoritas atas scope-nya sendiri", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " sid = (o or {}).get('scope_id')\n" + " if isinstance(sid, str) and 0 < len(sid) <= 64 and not carry.SAFE_LABEL.search(sid):\n" + " cuts_scope[sid] += 1\n" + " else:\n" + " cuts_file += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + bad = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="ghost") + p = w(os.path.join(D, "trusted.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, d + assert d["file_global"] == 1, d + assert len(optimize.active_records(recs, ep)) == 0, "populasi rev tak ikut terpotong" + """), + + ("M_REJECTED_SCOPE_GHOST_KEY: label dari baris yang ditolak tak boleh jadi kunci publik", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " cuts_scope[str((o or {}).get('scope_id') or 'default')] += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + for label in ("/etc/passwd.d/synthetic", "agent-b", "x" * 64): + bad = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope=label) + p = w(os.path.join(D, "ghost.jsonl"), [json.dumps(r) for r in rows] + [json.dumps(bad)]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, (label, d) + assert d["file_global"] == 1, (label, d) + """), + + ("M_REJECTED_SCOPE_CARDINALITY: regex 'label yang tampak aman' tetap mempercayai baris ditolak", + [("optimize.py", " if rejection_is_loss(why):\n cuts_file += 1", + " if rejection_is_loss(why):\n" + " sid = (o or {}).get('scope_id')\n" + " if isinstance(sid, str) and sid.isalnum() and len(sid) <= 32:\n" + " cuts_scope[sid] += 1\n" + " else:\n" + " cuts_file += 1")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + flood = [json.dumps(rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="s%04d" % i)) + for i in range(2000)] + p = w(os.path.join(D, "flood.jsonl"), flood + [json.dumps(r) for r in rows]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {}, len(d["scope_local"]) + assert d["file_global"] == 2000, d["file_global"] + """), + + ("M_GLOBAL_DAMAGE_NOT_CUT: kegagalan decode adalah kehilangan, bukan catatan kaki", + [("optimize.py", + 'NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares",\n "not our record_type")', + 'NOT_A_LOSS = ("not valid UTF-8",)')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + ghost = rec(100 + 9 * 604800, {"Bash": 10.0, "Read": 10.0}, scope="@@M@@") + bad = json.dumps(ghost).encode().replace("@@M@@".encode(), bytes([255, 254, 128])) + p = w(os.path.join(D, "notcut.jsonl"), [json.dumps(r) for r in rows] + [bad]) + recs, rej, _l, ep = optimize.load_history(p) + assert optimize.damage_summary(ep)["boundaries"] == 1, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 0, "epoch tak terpotong" + """), + + ("M_GLOBAL_DAMAGE_POISONS_FOREVER: potongan global memotong di POSISI, bukan selamanya", + [("optimize.py", " if r.get(EPOCH_KEY, key) == key:", + " if r.get(EPOCH_KEY, key) == key and not (epochs or {}).get('file_global'):")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="rev") + for i in range(6)] + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="rev") for i in range(6)] + p = w(os.path.join(D, "poison.jsonl"), + [json.dumps(r) for r in rows] + [chr(123) + '"torn":'] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 6, len(optimize.active_records(recs, ep)) + """), + + ("M_VALID_DEGRADED_POISONS_SCOPE_FOREVER: record sah yang kehilangan bukti membuka epoch", + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if verdict == RUN_ID_CONFLICT:")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a") + deg["unreadable"] = 3 + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "stuck.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg)] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("epoch pemulihan tak terbuka", len(cur)) + assert optimize.history_quality(optimize.comparable(cur)[0]) == "COMPLETE", "masih DEGRADED" + """), + + ("M_VALID_DEGRADED_CUTS_ALL_SCOPES: kehilangan milik satu scope hanya memotong scope itu", + [("optimize.py", " cuts_scope[scope_of(o)] += 1", + " cuts_file += 1")], + """ + a = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + b = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="b") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a") + deg["unreadable"] = 3 + p = w(os.path.join(D, "neighbour.jsonl"), + [json.dumps(r) for r in a] + [json.dumps(r) for r in b] + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + cur = [r for r in optimize.active_records(recs, ep) if optimize.scope_of(r) == "b"] + assert len(cur) == 6, ("tetangga kehilangan epoch-nya", len(cur)) + """), + + ("M_DEGRADED_RECORD_INCLUDED_POST_BOUNDARY: record yang melaporkan kehilangan ada di epoch LAMA", + [("optimize.py", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]\n" + " + (1 if degrades_scope(o) else 0))")], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(6)] + deg = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a", run_id="degraded-one") + deg["unreadable"] = 3 + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "order.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg)] + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert "degraded-one" not in [r.get("run_id") for r in cur], "record DEGRADED masuk epoch baru" + assert len(cur) == 6, len(cur) + """), + + ("M_DEGRADED_RETRY_CUTS_TWICE: satu boundary per run logis, bukan per salinan", + [("optimize.py", " if verdict == TRUE_RETRY:", " if False:")], + """ + def deg(seq, bash, run_id): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = 2 + return json.dumps(r) + healthy = [json.dumps(rec(100 + (20 + i) * 604800, + {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "retrycut.jsonl"), [deg(0, 30.0, "X")] + healthy + [deg(0, 30.0, "X")]) + recs, rej, _l, ep = optimize.load_history(p) + d = optimize.damage_summary(ep) + assert d["scope_local"] == {"a": 1}, d + assert len(optimize.active_records(recs, ep)) == 6, len(optimize.active_records(recs, ep)) + # kontrol: kehilangan kedua dari run yang BENAR-BENAR lain tetap memotong + p2 = w(os.path.join(D, "realsecond.jsonl"), + [deg(0, 30.0, "X")] + healthy + [deg(30, 30.0, "Y")]) + recs2, rej2, _l2, ep2 = optimize.load_history(p2) + assert optimize.damage_summary(ep2)["scope_local"] == {"a": 2}, optimize.damage_summary(ep2) + """), + + ("M_DEGRADED_RETRY_NO_RECOVERY: retry tak boleh mencuci kehilangan yang dilaporkan kembarannya", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (scope_of(rec), rec.get(EPOCH_KEY), rid)')], + """ + rows = [rec(100 + i * 604800, {"Bash": 30.0 + i * 3, "Read": 70.0 - i * 3}, scope="a") + for i in range(5)] + deg = rec(100 + 5 * 604800, {"Bash": 45.0, "Read": 55.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + clean_retry = rec(100 + 6 * 604800, {"Bash": 48.0, "Read": 52.0}, scope="a", run_id="X") + later = [rec(100 + (20 + i) * 604800, {"Bash": 48.0 + i * 3, "Read": 52.0 - i * 3}, + scope="a") for i in range(6)] + p = w(os.path.join(D, "retry.jsonl"), + [json.dumps(r) for r in rows] + [json.dumps(deg), json.dumps(clean_retry)] + + [json.dumps(r) for r in later]) + recs, rej, _l, ep = optimize.load_history(p) + # amandemen §7.6: kedua salinan BERBEDA, jadi yang belakangan adalah konflik identitas, + # bukan retry. Yang dijaga tetap sama: salinan bersih itu tak boleh jadi bukti epoch pulih. + assert ep["run_id_conflicts"] == 1, dict(rej) + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + cur = optimize.active_records(recs, ep) + assert len(cur) == 6, ("salinan bersih dari run yang sama ikut jadi bukti", len(cur)) + assert all(r.get("run_id") != "X" for r in cur), "salinan konflik masuk epoch pulih" + """), + + + # ------------------------------- run-id conflict (author-adversarial review of 71016ee) + ("M_RUNID_CONFLICT_TREATED_AS_RETRY: run_id sama bukan bukti pengamatan sama", + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return TRUE_RETRY")], + """ + def obs(seq, bash, run_id, unreadable=0, scanned=100): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + r["scanned"] = scanned + return json.dumps(r) + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "conflict.jsonl"), + [obs(0, 30.0, "X", unreadable=2)] + mid + + [obs(40, 66.0, "X", unreadable=9, scanned=150)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 1, dict(rej) + assert rej.get("duplicate run_id (retry)", 0) == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 2}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "bukti menyeberangi konflik identitas" + """), + + ("M_RUNID_CONFLICT_SECOND_LOSS_SUPPRESSED: konflik identitas memotong di posisinya sendiri", + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if degrades_scope(o) and verdict != RUN_ID_CONFLICT:")], + """ + def obs(seq, bash, run_id, unreadable=0): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + return json.dumps(r) + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "second.jsonl"), + [obs(0, 30.0, "X", unreadable=2)] + mid + [obs(40, 66.0, "X", unreadable=9)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert len(optimize.active_records(recs, ep)) == 6, "kehilangan kedua tak dipotong" + """), + + ("M_EXACT_RETRY_OPENS_SECOND_BOUNDARY: salinan PERSIS SAMA tetap satu kehilangan", + [("optimize.py", " return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT", + " return RUN_ID_CONFLICT")], + """ + deg = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "exact.jsonl"), [json.dumps(deg)] + mid + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 1}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "epoch pulih terhapus retry persis" + """), + + ("M_RUNID_CONFLICT_CROSSES_EPOCH: record yang mengontradiksi identitasnya ada di epoch LAMA", + [("optimize.py", " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)])", + " o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]\n" + " + (1 if retry_identity(o) in under else 0))")], + """ + def obs(seq, bash, run_id, unreadable=0): + r = rec(100 + seq * 604800, {"Bash": bash, "Read": 100.0 - bash}, scope="a", + run_id=run_id) + r["unreadable"] = unreadable + return json.dumps(r) + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "crosses.jsonl"), + [obs(0, 30.0, "X"), obs(40, 66.0, "X", unreadable=9)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + cur = optimize.active_records(recs, ep) + assert all(r.get("run_id") != "X" for r in cur), "record konflik masuk epoch yang ia buka" + assert len(cur) == 6, len(cur) + """), + + ("M_RUNID_CONFLICT_CROSS_SCOPE_COLLIDES: run_id sama di scope lain bukan pengamatan yang sama", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (rec.get(EPOCH_KEY, (0, 0))[0], rid)')], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="b", run_id="X") + p = w(os.path.join(D, "xscope.jsonl"), [json.dumps(a), json.dumps(b)]) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["boundaries"] == 0, optimize.damage_summary(ep) + assert len(recs) == 2, len(recs) + """), + + ("M_RUNID_CONFLICT_GLOBAL_EPOCH_COLLIDES: kehilangan global memutus kontinuitas identitas", + [("optimize.py", ' return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid)', + ' return (scope_of(rec), 0, rid)')], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="a", run_id="X") + p = w(os.path.join(D, "xglobal.jsonl"), + [json.dumps(a), chr(123) + '"torn":', json.dumps(b)]) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 0, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {}, optimize.damage_summary(ep) + assert len(recs) == 2, len(recs) + """), + + ("M_CLEAN_RUNID_CONFLICT_IGNORED: kontradiksi identitas tanpa kehilangan tetap integritas", + [("optimize.py", " if verdict == RUN_ID_CONFLICT or degrades_scope(o):", + " if degrades_scope(o):")], + """ + a = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + b = rec(200, {"Bash": 50.0, "Read": 50.0}, scope="a", run_id="X") + tail = [json.dumps(rec(100 + (50 + i) * 604800, + {"Bash": 72.0 + i * 3, "Read": 28.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "cleanconf.jsonl"), [json.dumps(a), json.dumps(b)] + tail) + recs, rej, _l, ep = optimize.load_history(p) + assert ep["run_id_conflicts"] == 1, dict(rej) + assert optimize.damage_summary(ep)["scope_local"] == {"a": 1}, optimize.damage_summary(ep) + assert len(optimize.active_records(recs, ep)) == 6, "bukti menyeberangi kontradiksi identitas" + """), + + ("M_PRIVATE_ANNOTATION_IN_FINGERPRINT: anotasi pembaca bukan bagian dari observasi", + [("optimize.py", " body = {k: v for k, v in rec.items() if k not in READER_PRIVATE}", + " body = dict(rec)")], + """ + deg = rec(100, {"Bash": 30.0, "Read": 70.0}, scope="a", run_id="X") + deg["unreadable"] = 2 + mid = [json.dumps(rec(100 + (10 + i) * 604800, + {"Bash": 36.0 + i * 3, "Read": 64.0 - i * 3}, scope="a")) + for i in range(6)] + p = w(os.path.join(D, "private.jsonl"), [json.dumps(deg)] + mid + [json.dumps(deg)]) + recs, rej, _l, ep = optimize.load_history(p) + assert rej.get("duplicate run_id (retry)") == 1, dict(rej) + assert ep["run_id_conflicts"] == 0, dict(rej) + """), ] diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 9e332d0..f2cc383 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -71,10 +71,36 @@ def main(): p = write(os.path.join(d, "corrupt.jsonl"), [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="a" * 32), "{tidak lengkap", "", "null", rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="a" * 32)]) # penulisan ULANG run yang sama - recs, rejected, lines = optimize.load_history(p) - check("JSONL rusak: satu record sah bertahan", len(recs), 1) + recs, rejected, lines, epochs = optimize.load_history(p) + # Sejak perbaikan B1: baris robek MEMOTONG riwayat di posisi fisiknya. Record sesudah potongan + # adalah pengamatan milik epoch BARU — termasuk bila run_id-nya sama — jadi keduanya bertahan + # dan tak ada yang dihitung sebagai retry. Yang dijaga: record SEBELUM potongan tidak ikut + # dianalisis, dan duplikat DI DALAM satu epoch tetap dibuang + menurunkan kualitas survivor. + check("JSONL rusak: record sebelum & sesudah potongan sama-sama terbaca", len(recs), 2) check("baris tak terparse dihitung", rejected["unparseable line"], 1) - check("run_id sama (retry) dihitung sekali", rejected["duplicate run_id (retry)"], 1) + check("run_id sama di SISI LAIN potongan bukan retry", rejected["duplicate run_id (retry)"], 0) + check("hanya epoch terbaru yang aktif", len(optimize.active_records(recs, epochs)), 1) + # Amandemen keputusan-beku (docs/V142_COUNTEREXAMPLES.md §7.6): run_id adalah KLAIM identitas, + # bukan bukti kesamaan. Salinan yang PERSIS sama = retry; salinan yang berbeda (di sini `ts`) + # = konflik identitas yang memotong di posisinya sendiri, dan tak pernah disebut retry. + p_same = write(os.path.join(d, "same_epoch.jsonl"), + [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32), + rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) + recs_same, rej_same, _l, ep_same = optimize.load_history(p_same) + check("salinan PERSIS SAMA di dalam satu epoch dihitung sekali", + (len(recs_same), rej_same["duplicate run_id (retry)"]), (1, 1)) + check("dan salinan persis sama tidak memotong apa pun", + optimize.damage_summary(ep_same)["boundaries"], 0) + p_conf = write(os.path.join(d, "conflict_epoch.jsonl"), + [rec(100, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32), + rec(101, {"Bash": 50.0, "Read": 50.0}, run_id="d" * 32)]) + recs_c, rej_c, _l, ep_c = optimize.load_history(p_conf) + check("salinan BERBEDA di bawah satu run_id adalah konflik, bukan retry", + (len(recs_c), rej_c.get("duplicate run_id (retry)", 0), ep_c["run_id_conflicts"]), + (2, 0, 1)) + check("dan konflik itu memotong di posisinya sendiri", + (optimize.damage_summary(ep_c)["boundaries"], len(optimize.active_records(recs_c, ep_c))), + (1, 0)) # Dua AGEN boleh menghasilkan metrik identik. Tanpa run_id itu dua pengamatan, bukan duplikat: # membuang salah satunya akan mengecilkan populasi yang sedang diukur. p2 = write(os.path.join(d, "twin.jsonl"), @@ -87,7 +113,7 @@ def main(): p = write(os.path.join(d, "shrink.jsonl"), [rec(100, {"Bash": 40.0, "Read": 60.0}, turns=10000), rec(200, {"Bash": 60.0, "Read": 40.0}, turns=1000)]) - recs, _, _ = optimize.load_history(p) + recs, _, _, _ = optimize.load_history(p) keep, dropped = optimize.comparable(recs) check("korpus menyusut 10x -> record lama TIDAK dibandingkan", (len(keep), len(dropped)), (1, 1)) check("jam mundur terdeteksi", diff --git a/tools/carry.py b/tools/carry.py index a67ca08..d527da1 100755 --- a/tools/carry.py +++ b/tools/carry.py @@ -35,6 +35,14 @@ # ASCII, and the constant was calibrated on the same len(). +# Evidence a sweep SELECTED and then could not turn into a measurement. A chosen bound is NOT one +# of them, which is the whole distinction `--accept-partial` rests on: a caller may adopt the bound +# it asked for, and may never adopt a file it could not read or a record it could not parse. +# One vocabulary, read by the producer below and by the optimizer's legacy law, so the two cannot +# drift into disagreeing about what "partial" meant. +LOSS_FIELDS = ("unreadable", "oversize", "malformed", "identity_changed", "conflicted_sources") +BOUND_FIELDS = ("skipped_by_limit",) + HISTORY_SCHEMA = 2 # the generation 1.3 wrote. Kept as the name older callers import. # 0 = pre-1.2 records with no schema field · 1 = + population identity # 2 = + run_id / scope_id / evidence quality (multi-agent safety) @@ -231,7 +239,50 @@ def _identity(path): "mtime_ns": getattr(st, "st_mtime_ns", int(st.st_mtime * 1e9))} -def accumulate(paths, min_turns=50, max_files=0): +def bounded_paths(paths, max_files): + """The ONE definition of a bounded sample: the newest `max_files` sources by mtime. + + A bound has to mean "the most RECENT N", not "the first N the filesystem listed". An + alphabetical prefix of a long-lived archive is a sample of whatever was created first, which + for a 24x7 population is the least informative slice there is. + + It lives here, alone, because a bounded run had two of these: the carry sweep took the newest N + and the skill-listing scan took a discovery-order slice of the same list, so one `--max-files 1` + run analysed two different single-source populations and reported them as one. + """ + paths = list(paths) + if not max_files or len(paths) <= max_files: + return paths + + def age(q): + # Per SOURCE, not per sweep: one file whose mtime cannot be read used to send the whole + # selection back to a discovery-order slice, which is the very sample this function + # exists to avoid. A source that cannot be dated simply cannot claim to be the newest. + # (cross-family review, round 1) + try: + return os.path.getmtime(q) + except OSError: + return float("-inf") + + return sorted(paths, key=age, reverse=True)[:max_files] + + +def sweep_label(facts): + """The 1.3-generation word for how a sweep went, from the counters the sweep itself kept. + + Four values, and PARTIAL has to carry two different meanings: a bound the caller ASKED for, and + evidence that was selected and then LOST. The label keeps the vocabulary a 1.3 reader knows; + a consumer that must tell the two apart reads LOSS_FIELDS and BOUND_FIELDS, which is exactly + what the optimizer's legacy law does — `--accept-partial` may adopt a bound, never a loss. + """ + if not facts.get("sessions"): + return "INVALID" if facts.get("scanned") else "EMPTY" + if any(facts.get(k) for k in LOSS_FIELDS + BOUND_FIELDS): + return "PARTIAL" + return "COMPLETE" + + +def accumulate(paths, min_turns=50, max_files=0, selected=None): carry, size, usage = collections.Counter(), collections.Counter(), collections.Counter() runtimes, models = collections.Counter(), collections.Counter() turns = sessions = 0 @@ -252,14 +303,10 @@ def accumulate(paths, min_turns=50, max_files=0): usage_conflicted_transcripts = identity_conflicted_transcripts = 0 usage_exact_measurement_excluded = identity_exact_measurement_excluded = 0 conflicted_sources = records_rejected = 0 - # A bound has to mean "the most RECENT N", not "the first N the filesystem listed". An - # alphabetical prefix of a long-lived archive is a sample of whatever was created first, which - # for a 24x7 population is the least informative slice there is. - if max_files and len(paths) > max_files: - try: - paths = sorted(paths, key=lambda q: os.path.getmtime(q), reverse=True)[:max_files] - except OSError: - paths = list(paths)[:max_files] + # `selected` lets a caller that must hand the SAME sample to another consumer choose once and + # pass it in; without it the sweep selects for itself. `total_paths` above is still what + # discovery found, so the bound is reported against the whole population either way. + paths = list(selected) if selected is not None else bounded_paths(paths, max_files) for p in paths: scanned += 1 try: @@ -372,29 +419,24 @@ def accumulate(paths, min_turns=50, max_files=0): "not_attempted": max(0, selected - accounted), "malformed": malformed + oversize, "records_rejected": records_rejected, "dirs_unreadable": vanished} - # Provenance, not decoration: an analyser that cannot tell a complete sweep from a sweep that - # hit unreadable files or a file cap will happily call a bounded corpus the population. - quality = "COMPLETE" - if unreadable or oversize or skipped_by_limit or conflicted_sources: - quality = "PARTIAL" - if sessions == 0: - quality = "INVALID" if scanned else "EMPTY" - return dict(sessions=sessions, turns=turns, lengths=sorted(lengths), - carry=carry, size=size, usage=usage, runtimes=runtimes, models=models, - unreadable=unreadable, short=short, scanned=scanned, oversize=oversize, - skipped_by_limit=skipped_by_limit, quality=quality, - # v1.4 acquisition facts. `quality` above stays for the 1.3 reader; it is NOT - # what the new record carries, and no record asserts it. - sources=sources, parsed=parsed_files, counters=counters, - sample_bound=max_files or 0, malformed=malformed, - identity_changed=identity_changed, empty_source=empty_source, - usage_conflicts=usage_conflicts, out_of_order=out_of_order, - identity_conflicts=identity_conflicts, - usage_conflicted_transcripts=usage_conflicted_transcripts, - identity_conflicted_transcripts=identity_conflicted_transcripts, - conflicted_sources=conflicted_sources, - usage_exact_measurement_excluded=usage_exact_measurement_excluded, - identity_exact_measurement_excluded=identity_exact_measurement_excluded) + facts = dict(sessions=sessions, turns=turns, lengths=sorted(lengths), + carry=carry, size=size, usage=usage, runtimes=runtimes, models=models, + unreadable=unreadable, short=short, scanned=scanned, oversize=oversize, + skipped_by_limit=skipped_by_limit, + # v1.4 acquisition facts. `quality` below stays for the 1.3 reader; it is NOT + # what the new record carries, and no record asserts it. + sources=sources, parsed=parsed_files, counters=counters, + sample_bound=max_files or 0, malformed=malformed, + identity_changed=identity_changed, empty_source=empty_source, + usage_conflicts=usage_conflicts, out_of_order=out_of_order, + identity_conflicts=identity_conflicts, + usage_conflicted_transcripts=usage_conflicted_transcripts, + identity_conflicted_transcripts=identity_conflicted_transcripts, + conflicted_sources=conflicted_sources, + usage_exact_measurement_excluded=usage_exact_measurement_excluded, + identity_exact_measurement_excluded=identity_exact_measurement_excluded) + facts["quality"] = sweep_label(facts) + return facts # Relative price of one token in each bucket, base input = 1.0. A bucket's share of the diff --git a/tools/optimize.py b/tools/optimize.py index b27fe0e..c0dca1f 100644 --- a/tools/optimize.py +++ b/tools/optimize.py @@ -36,7 +36,23 @@ import skills as skills_tool # noqa: E402 (listing usage — reused) OPTIMIZER_VERSION = "1.0" -OUTPUT_SCHEMA_VERSION = 1 # shape of --json; bump when a field's meaning changes +OUTPUT_SCHEMA_VERSION = 2 # shape of --json; bump when a field's meaning changes + # 2 (unreleased) = `evidence_quality` is DERIVED and may read + # DEGRADED/UNKNOWN · `history.quality` is the worst quality of + # the evidence ELIGIBLE FOR THIS ANALYSIS, and is EMPTY when + # nothing is comparable · `history.damage` counts the loss + # boundaries the file holds: `file_global` is every loss a + # rejected line represents, and `scope_local` maps a scope to + # the losses its OWN VALIDATED RECORDS reported — never a label + # read off a line that failed validation, so its keys are always + # scopes that also appear in `scope.known` · `scope.records_in_scope` + # counts the whole scope while `scope.records_in_epoch` counts what + # the current epoch contributes · `history.run_id_conflicts` + # counts accepted records that repeat a run_id with a DIFFERENT + # persisted observation, an integrity event that cuts like any + # other loss and is never reported as a retry · a CANDIDATE outranks + # PARTIAL_EVIDENCE, because each finding is gated on its own + # evidence before the run is summarised THRESHOLD_SCHEMA_VERSION = 1 # bump when any threshold below changes, with a reason and a test # ---------------------------------------------------------------- frozen thresholds @@ -62,6 +78,325 @@ SCHEMA_SUPPORTED = (0, 1, 2) # 0 = pre-1.2, 1 = + population identity, 2 = + run/scope/quality STATES = ("OBSERVED", "HYPOTHESIS", "CANDIDATE", "EXPERIMENTAL", "PROVEN", "REJECTED") +# ---------------------------------------------------------------- the legacy evidence-quality law +# ONE place decides what a piece of legacy evidence is worth, because the alternative was five: +# load_history checked a vocabulary, comparable() checked two values, analyse() read the live +# sweep's word for it, the trend read nothing at all, and emit_candidates() never asked. A finding +# could then be promoted from a corpus nobody had swept completely. +# +# Worst wins, never a majority: one bounded observation among nine complete ones still means part +# of the evidence was never swept, and nine neighbours cannot launder it. +QUALITY_RANK = {"COMPLETE": 0, "PARTIAL": 1, "DEGRADED": 2, "EMPTY": 3, "UNKNOWN": 4, "INVALID": 5} +SCHEMA_WITH_QUALITY = 2 # the first history schema that records how a sweep was taken +# Two different questions, and conflating them cost a HIGH in review: +# REQUIRED — what the schema-2 writer ALWAYS wrote, so a record that lacks them cannot attest +# completeness. Nothing is invented for an older schema; a record that never carried +# these is UNKNOWN rather than trusted. +# LOSS — every counter that, WHEN PRESENT, means evidence was selected and then lost. A +# reader must honour a loss it can see even if that counter came from a later writer +# (`malformed_lines` is the 1.3.1-era name, `malformed` the current one). +RECORD_REQUIRED_COUNTERS = ("unreadable", "oversize", "skipped_by_limit") +RECORD_LOSS_COUNTERS = ("unreadable", "oversize", "malformed", "malformed_lines", + "identity_changed", "conflicted_sources", "records_rejected") +RECORD_BOUND_COUNTERS = ("skipped_by_limit",) +RECORD_COUNTERS = tuple(dict.fromkeys(RECORD_REQUIRED_COUNTERS + RECORD_LOSS_COUNTERS + + RECORD_BOUND_COUNTERS)) +# Set by load_history() when a duplicate run_id shows a worse sweep than the record that survives. +# A private, in-memory annotation; nothing writes it back to a file. +QUALITY_FLOOR = "_evidence_quality_floor" + + +def worst_quality(qualities): + """The worst quality in the set. An empty set is COMPLETE: a finding that draws on no sampled + evidence is not degraded by sampling that had nothing to do with it.""" + worst = "COMPLETE" + for q in qualities: + if QUALITY_RANK.get(q, QUALITY_RANK["UNKNOWN"]) > QUALITY_RANK[worst]: + worst = q if q in QUALITY_RANK else "UNKNOWN" + return worst + + +def may_promote(quality, accept_partial): + """May a finding resting on evidence of this quality become a CANDIDATE? + + `--accept-partial` means one thing: the caller declares the bound they asked for to be the + intended corpus. It is not a switch for evidence that was lost (DEGRADED), for a record that + cannot attest itself (UNKNOWN), or for one the producer's own rules call INVALID/EMPTY. + """ + return quality == "COMPLETE" or (quality == "PARTIAL" and bool(accept_partial)) + + +def _derived_record_quality(rec, schema): + """What a record's OWN numbers prove, ignoring what it claims.""" + nums = {} + for k in RECORD_COUNTERS + ("sessions", "turns", "carry_bytes", "scanned"): + if k not in rec: + continue + v = rec[k] + if isinstance(v, bool) or not isinstance(v, int) or v < 0: + return "INVALID" # a count that is not a count: the record is not trustworthy + nums[k] = v + if nums.get("sessions", 1) == 0: + # the producer's own terms: a sweep that looked and found nothing usable is INVALID; one + # that had nothing to look at is EMPTY + return "INVALID" if nums.get("scanned", 0) > 0 else "EMPTY" + if "sessions" in nums and nums.get("turns", 1) == 0: + # Sessions were counted, so turns were counted. A record reporting sessions without them is + # not a quiet sweep, it is an impossible one. + return "INVALID" + shares = rec.get("shares") + if isinstance(shares, dict) and "carry_bytes" in nums: + # Zero carry is a REAL outcome, not a corrupt record: "A session whose every item lands on + # its final turn carries nothing" (tools/carry.py's own report). The 1.3 writer emits + # `shares: {}` for it and the current writer guards `if C else {}`. What cannot both be + # true is a share vector with no carry behind it, or carry with nothing to distribute. + if nums["carry_bytes"] == 0: + return "EMPTY" if not shares else "INVALID" + if not shares: + return "INVALID" + if "scanned" in nums and nums["scanned"] < nums.get("sessions", 0): + # A session is a transcript that was scanned AND cleared the turn floor, so the producer + # can never report more sessions than it scanned. Forty sessions out of one scanned file + # is not a sweep that went well; it is a record describing a sweep that cannot have + # happened. (cross-family review, round 1) + return "INVALID" + if any(nums.get(k, 0) for k in RECORD_LOSS_COUNTERS): + return "DEGRADED" + if any(nums.get(k, 0) for k in RECORD_BOUND_COUNTERS): + return "PARTIAL" + if not all(k in nums for k in ("sessions", "turns", "carry_bytes", "scanned")): + # Zero counters say "nothing went wrong"; they do not say a sweep happened. A record whose + # corpus fields are absent entirely has a share vector and no population behind it, and + # reading its zeroed counters as completeness would trust a measurement nobody took. + # (cross-family review, round 1) + return "UNKNOWN" + if schema >= SCHEMA_WITH_QUALITY and not all(k in nums for k in RECORD_REQUIRED_COUNTERS): + return "UNKNOWN" # claims a completeness it cannot show + return "COMPLETE" + + +def record_quality(rec): + """Evidence quality of ONE legacy history record: the worst of what it claims and what it can + show. + + A record may only ever describe itself as no better than its own numbers. Two ways it fails to + attest at all: the value is missing or unrecognised, or the record declares a schema older than + the field itself — schema 0/1 predates `evidence_quality`, so a schema-1 record carrying + COMPLETE is asserting something its own writer could not have known. That is a claim from a + hand-edited file or a back-filled migration, not provenance. The record stays readable and + stays in the report; it simply cannot carry a promotion. + """ + if not isinstance(rec, dict): + return "INVALID" + schema = rec.get("schema_version", 0) + if isinstance(schema, bool) or not isinstance(schema, int): + # "2" is a string, 2.0 is a float, True is neither. A version this reader cannot name is + # not a newer generation to be trusted; it is an older one to be doubted. + schema = 0 + claimed = rec.get("evidence_quality") + if not isinstance(claimed, str) or claimed not in QUALITY_RANK: + claimed = "UNKNOWN" + if schema < SCHEMA_WITH_QUALITY: + claimed = "UNKNOWN" + derived = _derived_record_quality(rec, schema) + if claimed == "PARTIAL" and derived == "COMPLETE": + # A bound the caller asked for shows up in `skipped_by_limit`; a loss shows up in a loss + # counter. A record that says PARTIAL while every counter it carries says nothing happened + # cannot say WHY it was partial — and `--accept-partial` adopts a bound, not a word. + # (cross-family review, confirmation round) + claimed = "UNKNOWN" + qualities = [claimed, derived] + floor = rec.get(QUALITY_FLOOR) + if floor: # absent is not UNKNOWN: most records carry no floor at all + qualities.append(floor) + return worst_quality(qualities) + + +def sweep_quality(live): + """Evidence quality of the LIVE sweep, from the counters carry.accumulate() kept. + + The producer reports one word for two different things (see carry.sweep_label): a bound the + caller asked for and evidence that was lost both read PARTIAL. Here they separate, because only + the first is something `--accept-partial` may adopt. The counters are the evidence and the + label is a summary of them, so the worst of the two governs. + """ + if not live: + return "COMPLETE" # no live sweep is not bad live evidence; it is none + claimed = live.get("quality") + if not isinstance(claimed, str) or claimed not in QUALITY_RANK: + claimed = "COMPLETE" + if not live.get("sessions"): + return "INVALID" if live.get("scanned") else "EMPTY" + if not live.get("turns") or not live.get("carry_bytes", 1): + # The same impossibility the record law refuses: sessions were counted, so turns were + # counted. A live dict that says otherwise is not a quiet sweep. `carry_bytes` is absent + # from the sweep dict itself (the caller sums it), so its absence is not the claim. + # (cross-family review, round 4) + return "INVALID" + derived = "COMPLETE" + if any(live.get(k) for k in carry.LOSS_FIELDS): + derived = "DEGRADED" + elif any(live.get(k) for k in carry.BOUND_FIELDS): + derived = "PARTIAL" + return worst_quality([claimed, derived]) + + +# A rejected line is a LOSS by default: the history held something that could not be read as a +# record, and a trend built from the survivors is built over a gap. The exceptions are named, and +# they are the only two things a rejection can be that are not a loss, plus the one shape that was +# never our record to begin with: +# * a refusal by design — current-generation evidence this optimizer does not read — which is a +# contract, not a loss, and would otherwise cut every history in the middle of a migration; +# * a deduplicated retry, which is bookkeeping (its quality already travels via QUALITY_FLOOR); +# * a well-formed JSON line that is not a carry record at all. In a shared file that is another +# tool's entry, and treating it as lost evidence would cut a history nothing happened to. +# Listed this way round on purpose: a reason added to valid_record() later defaults to LOSS rather +# than slipping through an allowlist nobody updated. (cross-family review, confirmation round) +NOT_A_LOSS = ("current-generation", "duplicate run_id", "not an object", "no shares", + "not our record_type") + +# The epoch a record belongs to: (file-global losses seen before it, losses seen before it that were +# attributed to ITS scope). A private, in-memory annotation; nothing writes it back to a file. +EPOCH_KEY = "_history_epoch" + + +def rejection_is_loss(reason): + """Did this rejected line cost the history a record? -> bool. + + A loss is not a verdict on the file. It is a BOUNDARY at that line's physical position: the + records before it and the records after it are two populations, and only the newest one may + support a promotion. The gap stays visible in `history.rejected` and `history.damage`. (The + first repair made damage permanent instead, and an independent acceptance review blocked it: + nothing in this product expires, rotates or repairs a history, and no flag adopts a loss, so a + single crash fragment disabled promotion for every scope, forever.) + """ + return not any(x in str(reason) for x in NOT_A_LOSS) + + +def degrades_scope(rec): + """Does this VALIDATED record's own canonical evidence prove that it lost records? -> bool. + + The one trusted source of a scope-local boundary, and the reason it is trusted is provenance, + not syntax: the record passed valid_record(), its `scope_id` is the same field every accepted + record already publishes through `scope.known`, and the damage fact comes from + RECORD_LOSS_COUNTERS rather than from guessing what a corrupt line meant. + + Only an actual loss qualifies. A bound the caller asked for (PARTIAL), a legacy schema that + cannot attest completeness (UNKNOWN) and a readable record nothing can be compared with + (INVALID/EMPTY) are not proof that a line went missing, and must not open an epoch. + """ + if not isinstance(rec, dict): + return False + schema = rec.get("schema_version", 0) + if isinstance(schema, bool) or not isinstance(schema, int): + schema = 0 + return _derived_record_quality(rec, schema) == "DEGRADED" + + +# The reader's own annotations. They are computed while reading, never persisted, and they must +# not take part in deciding what a record IS: otherwise the act of reading a file would make an +# identical retry look like a different observation. +READER_PRIVATE = (EPOCH_KEY, QUALITY_FLOOR) + + +def observation_digest(rec): + """A deterministic digest of the PERSISTED observation -> str. + + The whole record participates, not a hand-picked subset: the defect this exists to close was + caused by an identity that was too weak, and a fingerprint built from a chosen handful would be + the same mistake in a new spelling. Keys are sorted, so JSON key order and whitespace cannot + make two semantically identical observations differ, and a persisted field this reader does not + know still changes the digest — conservatively non-equivalent, which is fail-closed and + recoverable. + """ + body = {k: v for k, v in rec.items() if k not in READER_PRIVATE} + return hashlib.sha256(json.dumps(body, sort_keys=True, separators=(",", ":"), + default=str).encode("utf-8")).hexdigest() + + +def retry_identity(rec): + """The identity two accepted records must SHARE before they can be the same run -> key | None. + + Scope, because two agents legitimately write at once. The FILE-GLOBAL half of the epoch, + because across an unattributable loss the reader cannot establish continuity at all and both + copies stand. Not the scope-local half: a trusted boundary leaves the file intact, so identity + survives it. A record without a run_id cannot be shown to be anyone's retry and is always its + own observation. + """ + rid = rec.get("run_id") + if not (isinstance(rid, str) and rid): + return None + return (scope_of(rec), rec.get(EPOCH_KEY, (0, 0))[0], rid) + + +# `run_id` is an identity CLAIM, not proof of semantic equality. Two records are the same run only +# when their identity AND their persisted observation match; a same-identity pair whose observations +# differ is an integrity event, not bookkeeping. ONE classification, so the boundary, the +# deduplication, the quality floor and the diagnostics can never disagree about the same pair. +TRUE_RETRY, RUN_ID_CONFLICT, FIRST_SIGHTING = "retry", "conflict", "first" + + +def classify_repeat(prior_digest, digest): + """-> TRUE_RETRY | RUN_ID_CONFLICT | FIRST_SIGHTING""" + if prior_digest is None: + return FIRST_SIGHTING + return TRUE_RETRY if prior_digest == digest else RUN_ID_CONFLICT + + +def active_epoch_key(scope, epochs): + """The epoch a record of `scope` must carry to be part of the CURRENT analysis.""" + e = epochs or {} + return (e.get("file_global", 0), (e.get("scope_local") or {}).get(scope, 0)) + + +def active_records(recs, epochs): + """The records after the newest loss that applies to their own scope. + + A record built in memory rather than read from a file carries no epoch annotation; it belongs + to the current epoch, because no loss was observed around it. + """ + out = [] + for r in (recs or []): + key = active_epoch_key(scope_of(r), epochs) + if r.get(EPOCH_KEY, key) == key: + out.append(r) + return out + + +def damage_summary(epochs): + """What the FILE holds, as diagnostics — never a gate. A historical gap can stay true while the + evidence after it is independently complete. + + `file_global` is every loss a rejected line represents: a rejected line cannot say whose record + it was, so it cuts every scope at its position. `scope_local` therefore has exactly one source — + records that PASSED validation and whose own canonical counters reported a loss — and its keys + are always scopes that also appear in `scope.known`. No attacker-controlled string from a + rejected line can add a key here, whatever it looks like, and the map's cardinality is bounded + by the real scopes in the file rather than by the corrupt lines in it.""" + e = epochs or {} + local = dict(e.get("scope_local") or {}) + return {"boundaries": e.get("file_global", 0) + sum(local.values()), + "file_global": e.get("file_global", 0), "scope_local": local} + + +def history_quality(records): + """The worst quality among the records ELIGIBLE to support a finding. + + Eligibility is not decided here: comparable() already decided it — same scope, same workload + class, compatible corpus size, quality that can carry a comparison at all. A partial record + belonging to another agent's scope is not evidence for this run and must not block it. That + half matters as much as the fail-closed half: a gate that blocks on evidence a finding never + used is not correct, it is merely stuck. + """ + if not records: + # M1 (acceptance review): a machine consumer must never read COMPLETE and conclude that + # usable history exists. Nothing eligible is EMPTY — the producer's own word for "there was + # nothing to look at" — and it is refused by the promotion gate like every other non- + # COMPLETE value. worst_quality([]) keeps meaning COMPLETE: a FINDING that rests on no + # sampled evidence is not degraded by sampling it never used. + return "EMPTY" + return worst_quality([record_quality(r) for r in records]) + # machine-readable outcome. The CLI exits 0 for every VALID run by default (a scheduler must not # treat "nothing to do" as breakage); --strict-exit maps the status to the exit code instead. STATUS = {"NO_ACTION": 0, "CANDIDATE": 10, "INSUFFICIENT_DATA": 20, "HOST_BEHAVIOR_SHIFT": 30, @@ -109,6 +444,20 @@ def safe_err(exc): # ---------------------------------------------------------------- history: load and validate +def safe_label(v): + """A value read off a rejected line, before it may appear in a public reason string. + + Only a number can be a schema version, so only a number is echoed. Anything else is a corrupt + line's own content, it carries no diagnostic value beyond its type, and a reason string is + public output: `history.rejected` keys reach --json, the human report and any log that keeps + them. Truncating such a value is not a bound — thirty characters of a credential is still the + credential — so it is named by TYPE and never by content. + """ + if v is None or (isinstance(v, (int, float)) and not isinstance(v, bool)): + return repr(v) + return "of type " + type(v).__name__ + + def valid_record(o): """-> (ok, reason). Malformed input must not poison a trend; it must be counted and dropped.""" if not isinstance(o, dict): @@ -119,24 +468,46 @@ def valid_record(o): # the exact "a record gains trust by defaulting" failure, in the direction nobody watches. # Refuse it by NAME, and count the refusal, until the optimizer is ported. if isinstance(o.get("envelope"), dict): - return False, ("unsupported schema_version %r: current-generation (v1.4) evidence, " + return False, ("unsupported schema_version %s: current-generation (v1.4) evidence, " "not read by this optimizer" - % o["envelope"].get("schema_version")) + % safe_label(o["envelope"].get("schema_version"))) + # Computed here and used twice: a rejection's damage class depends on whether the line claimed + # to be one of OUR records at all. + claims_ours = (o.get("record_type") == "carry_run" or "schema_version" in o + or "run_id" in o or "carry_bytes" in o) if o.get("record_type") not in (None, "carry_run"): - return False, "unknown record_type" + # A record_type this reader cannot name, on a line that ALSO carries our fields, is a + # corrupted record of ours and therefore a loss. On a line that carries none of them it is + # another tool's entry in a shared history, and reading it as lost evidence would cut a + # history nothing happened to — the same distinction `no shares` already makes one check + # further down. (§6 of the trusted-boundary task: a migration must not read as damage.) + return False, ("unknown record_type" if claims_ours else "not our record_type") sv = o.get("schema_version", 0) if not isinstance(sv, int) or isinstance(sv, bool) or sv not in SCHEMA_SUPPORTED: - return False, f"unsupported schema_version {sv!r}" + # The reason string is public: it becomes a key of `history.rejected`. A rejected line's + # own content therefore goes through safe_label() before it can be echoed there. + return False, f"unsupported schema_version {safe_label(sv)}" sh = o.get("shares") - if not isinstance(sh, dict) or not sh: - return False, "no shares" + if not isinstance(sh, dict): + # Two different lines, and the damage classification depends on which one this is: a line + # that claims to be one of OUR records is a corrupted record (a loss), a line that claims + # nothing is another tool's entry in a shared file (not ours to lose). + return False, ("carry record without shares" if claims_ours else "no shares") + if not sh: + # An EMPTY share map is what the producer writes for a sweep that measured no carry. It is + # readable evidence with nothing to compare — record_quality() calls it EMPTY and + # comparable() leaves it out — and reading it as corruption would cut a history that is + # perfectly intact. A bare `{"shares": {}}` with no sign of being ours is still foreign. + if not claims_ours: + return False, "no shares" + sh = {} for k, v in sh.items(): if not isinstance(k, str) or not isinstance(v, (int, float)) or isinstance(v, bool): return False, "non-numeric share" if v < 0 or v > 100.5: return False, "share out of range" tot = sum(sh.values()) - if not (95.0 <= tot <= 105.0): + if sh and not (95.0 <= tot <= 105.0): return False, f"shares sum to {tot:.1f}, not ~100" for k in ("ts", "turns", "sessions", "carry_bytes"): v = o.get(k) @@ -147,6 +518,20 @@ def valid_record(o): q = o.get("evidence_quality") if q is not None and q not in ("COMPLETE", "PARTIAL", "INVALID", "EMPTY"): return False, "unknown evidence_quality" + sid = o.get("scope_id") + if sid is not None and (not isinstance(sid, str) or len(sid) > 64): + # `scope_id` is an ATTRIBUTION AUTHORITY: a validated record's own scope decides which + # population a loss cut applies to, so the field has to be checked before the record is + # accepted rather than after. The producer writes exactly one shape — + # `str(scope_id or "default")[:64]` (tools/carry.py) — so another type or another length + # was not written by it. Accepting one let `["rev"]` become the scope `"['rev']"`: the cut + # landed on a population nobody has and the real `rev` records kept crossing the loss, + # which is the blocked defect rebuilt through the one door this repair opened. + # (cross-family author review of the trust repair, round 2) + # The reason is STATIC on purpose: it is a rejected line's own content and must not be + # echoed. A control character inside a label the PRODUCER wrote is still accepted — it is + # already published through `scope.known` and calling it corruption would invent damage. + return False, "scope_id is not a label this producer writes" return True, "" @@ -155,45 +540,108 @@ def scope_of(r): def load_history(path): - """-> (records, rejected counter, lines). Deduplication is by run_id ONLY: two agents can + """-> (records, rejected counter, lines, epochs). Deduplication is by run_id within one epoch: two agents can legitimately produce the same timestamp, turn count and shares, and discarding one of them would undercount the population. A record without a run_id (schema 0/1) cannot be deduplicated and is kept as it is.""" recs, rejected = [], collections.Counter() + cuts_file, cuts_scope = 0, collections.Counter() + conflicts, under = 0, {} if not path or not os.path.exists(path): - return recs, rejected, 0 + return recs, rejected, 0, {"file_global": 0, "scope_local": {}, "run_id_conflicts": 0} lines = 0 try: - fh = open(path, encoding="utf-8", errors="replace") + # BYTES, decoded strictly per physical line. `errors="replace"` destroyed the evidence that + # a decode had failed: undecodable bytes inside `scope_id` arrived as U+FFFD, passed for a + # readable label, and cut a scope that does not exist — while the real population kept + # crossing the loss. A line the reader cannot decode is named and counted as a loss; the + # reader then continues at the next line rather than abandoning the file. + fh = open(path, "rb") except OSError as e: rejected[f"history unreadable: {safe_err(e)}"] += 1 - return recs, rejected, 0 + # A file that would not open is a loss nobody can attribute: every scope starts a new epoch + # with no records in it, which is the fail-closed answer. + return recs, rejected, 0, {"file_global": 1, "scope_local": {}, "run_id_conflicts": 0} with fh: - for line in fh: - line = line.strip() - if not line: + for raw in fh: + raw = raw.strip() + if not raw: continue lines += 1 - if len(line) > carry.MAX_RECORD: - rejected["record above the size cap"] += 1 - continue - try: - o = json.loads(line) - except Exception: - rejected["unparseable line"] += 1 - continue - ok, why = valid_record(o) - (recs.append(o) if ok else rejected.__setitem__(why, rejected[why] + 1)) - seen, uniq = set(), [] - for r in recs: - rid = r.get("run_id") - if isinstance(rid, str) and rid: - if rid in seen: - rejected["duplicate run_id (retry)"] += 1 + o, ok, why = None, False, "" + if len(raw) > carry.MAX_RECORD: # the cap is on BYTES; so is the line that hit it + why = "record above the size cap" + else: + try: + text = raw.decode("utf-8") + except UnicodeDecodeError: + why = "line is not valid UTF-8" + else: + try: + o = json.loads(text) + except Exception: + why = "unparseable line" # a torn line cannot say whose record it was + else: + ok, why = valid_record(o) + if ok: + # Stamped at READ time, in physical order: the epoch is a position in the file, + # never a timestamp, because the clock is exactly what a damaged history cannot + # be trusted about. + o[EPOCH_KEY] = (cuts_file, cuts_scope[scope_of(o)]) + # ONE classification for the pair, read here and nowhere else. The boundary, the + # deduplication, the quality floor and the diagnostics all consume this verdict: + # the defect that produced it was a boundary layer and a dedup layer holding two + # notions of identity, one of them too coarse. + ident = retry_identity(o) + prior = under.get(ident) if ident is not None else None + digest = observation_digest(o) if ident is not None else None + verdict = classify_repeat(prior["digest"] if prior else None, digest) + if verdict == TRUE_RETRY: + # The same run reporting the SAME observation again. Dropped as an observation, + # counted, and its quality still travels to the copy that survives — a retry + # cannot launder a bounded or lossy sweep into a complete one. It opens no + # boundary: a repeated copy of one loss is one loss, and letting a late copy + # cut again let `6 healthy records + one more copy of X` erase a recovered + # epoch, on repeat, forever. + rejected["duplicate run_id (retry)"] += 1 + kept = prior["rec"] + worse = worst_quality([record_quality(kept), record_quality(o)]) + if worse != record_quality(kept): + kept[QUALITY_FLOOR] = worse + continue + recs.append(o) + if ident is not None: + # The NEWEST observation is what this identity currently says, so a later exact + # copy is that one's retry rather than the original's conflict. + under[ident] = {"rec": o, "digest": digest} + if verdict == RUN_ID_CONFLICT: + # Not bookkeeping. The file states two different things under one identity and + # the reader cannot tell which one the population it is about to compare + # belongs to. Counted as a CONFLICT — reporting it as a retry would be a false + # statement — and cut at this record's own physical position, so evidence from + # before it is never combined with evidence after it. Recoverable like every + # other boundary here: enough clean later evidence promotes normally. + conflicts += 1 + if verdict == RUN_ID_CONFLICT or degrades_scope(o): + # The record belongs to the epoch it CLOSES: stamped first, counter moved + # after it. A population that recovers is not founded on the observation that + # reported the loss, nor on the one that contradicted its own identity. + cuts_scope[scope_of(o)] += 1 continue - seen.add(rid) - uniq.append(r) - return uniq, rejected, lines + rejected[why] += 1 + # EVERY loss a rejected line represents is file-global. A record that failed validation + # is not a trustworthy authority for its own scope attribution: the field that would + # name the population is part of the line this reader just refused to believe. Reading + # it anyway is the fail-open half — it invents a scope that may not exist and leaves + # the damaged one uncut. This over-blocks on purpose, and the over-block is temporary + # because epochs recover; a crossing of a real loss is not. + if rejection_is_loss(why): + cuts_file += 1 + # Deduplication already happened, in the read pass, in physical order, from the SAME verdict + # the boundary used. A second pass with its own notion of identity is exactly how the two + # layers came to disagree about one pair. + return recs, rejected, lines, {"file_global": cuts_file, "scope_local": dict(cuts_scope), + "run_id_conflicts": conflicts} def by_scope(recs): @@ -203,17 +651,41 @@ def by_scope(recs): return dict(out) +def eligible_anchor(recs): + """The newest record that may speak for a population -> record or None. + + Eligibility comes FIRST. The anchor used to be the newest record of any kind, and the very next + line threw it away for being INVALID — after it had already decided which scope, which workload + class and which corpus size every other record was measured against. One broken sweep at the + top of the file could strand an entire eligible population. + """ + usable = [r for r in recs if record_quality(r) not in ("INVALID", "EMPTY")] + return max(usable, key=lambda r: r.get("ts") or 0) if usable else None + + def comparable(recs): - """Records that may be compared with the newest one: same scope, same workload class, - compatible corpus size, usable evidence quality.""" + """Records that may be compared with the newest ELIGIBLE one: same scope, same workload class, + compatible corpus size, quality that can carry a comparison at all. + + INVALID and EMPTY records are dropped here; PARTIAL, DEGRADED and UNKNOWN ones stay, because + they ARE part of the population and the report should show them. What they cannot do is carry + a promotion — that is the promotion gate's job, and it reads history_quality() over exactly the + records this function kept. + """ if not recs: return [], [] - newest = max(recs, key=lambda r: r.get("ts") or 0) + newest = eligible_anchor(recs) + if newest is None: + return [], [(r, "evidence quality " + record_quality(r)) for r in recs] n_turn = newest.get("turns") or 0 n_scope = scope_of(newest) n_work = str(newest.get("workload_class") or "") keep, dropped = [], [] for r in recs: + q = record_quality(r) + if q in ("INVALID", "EMPTY"): + dropped.append((r, "evidence quality " + q)) + continue if scope_of(r) != n_scope: dropped.append((r, f"scope {scope_of(r)!r} vs {n_scope!r}")) continue @@ -221,9 +693,6 @@ def comparable(recs): dropped.append((r, f"workload class {str(r.get('workload_class') or '')!r} vs {n_work!r} " "— WORKLOAD_SHIFT, not a trend")) continue - if r.get("evidence_quality") in ("INVALID", "EMPTY"): - dropped.append((r, "evidence quality " + str(r.get("evidence_quality")))) - continue t = r.get("turns") or 0 if n_turn and t and max(n_turn, t) / min(n_turn, t) > CORPUS_TURN_RATIO_MAX: dropped.append((r, f"corpus {t:,} turns vs {n_turn:,}")) @@ -279,7 +748,9 @@ def load_ledger(path): try: fh = open(path, encoding="utf-8", errors="replace") except OSError: - return None + # A ledger that EXISTS and cannot be read is not the same as no ledger at all: the guard's + # evidence is missing rather than absent by design, and `rejected` says so. + return {"writes": 0, "prevented": 0, "rejected": 1, "rate": 0.0, "unreadable": True} with fh: for line in fh: line = line.strip() @@ -298,6 +769,11 @@ def load_ledger(path): checked += 1 elif ev == "denied": denied += 1 + else: + # Neither a write nor a prevented write: a line this reader cannot account for. + # Counting it as nothing at all let a ledger full of unknown events look like a + # clean sample. (cross-family review, confirmation round) + rejected += 1 total = checked + denied return {"writes": total, "prevented": denied, "rejected": rejected, "rate": (100.0 * denied / total) if total else 0.0} @@ -336,8 +812,14 @@ def finding(cid, state, headline, evidence, scope="default", bucket=0, hypothesi def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): out = [] - quality = (live or {}).get("quality", "COMPLETE") if live else "COMPLETE" - usable = accept_partial or quality == "COMPLETE" + # Every candidate-producing path below states which evidence it rests on, and asks the same + # question about exactly that evidence. A finding drawing on the live sweep is not blocked by a + # partial history it never read, and a finding drawing on history is not waved through because + # today's sweep happened to be clean. + quality = sweep_quality(live) if live else "COMPLETE" + hist_q = history_quality(hist.get("comparable", [])) + usable = may_promote(quality, accept_partial) # findings that rest on the live sweep + hist_usable = may_promote(hist_q, accept_partial) # findings that rest on the history run_ids = [r.get("run_id") for r in hist.get("comparable", []) if r.get("run_id")] # 1. concentration — only sayable over a corpus, never over one session, and never from a @@ -395,19 +877,27 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): f"which is the same movement, not a second one" continue reported.append((k, per_month)) - if True: - cid = "trend-" + re.sub(r"[^a-z0-9]+", "-", k.lower()).strip("-") - spike = (" — but the last value sits z=%+.1f from its own history, so this slope " - "may be one spike rather than a shift" % t["z"]) if abs(t["z"]) >= TREND_MIN_Z else "" - out.append(finding(cid, "CANDIDATE", f"{k} share is moving", - f"{per_month:+.1f} pp/month over {t['n']} records in scope {scope!r}, " - f"last value z={t['z']:+.1f}{spike}", scope=scope, - # DIRECTION, not magnitude. A fitted slope decays as its window - # grows even when the world stopped moving, so bucketing the - # magnitude mints a fresh proposal every time the estimator - # settles. One sustained movement is one proposal; a REVERSAL - # is a new one, which is exactly when a human should look again. - bucket=1 if per_month > 0 else -1, + cid = "trend-" + re.sub(r"[^a-z0-9]+", "-", k.lower()).strip("-") + spike = (" — but the last value sits z=%+.1f from its own history, so this slope " + "may be one spike rather than a shift" % t["z"]) if abs(t["z"]) >= TREND_MIN_Z else "" + # DIRECTION, not magnitude. A fitted slope decays as its window grows even when the + # world stopped moving, so bucketing the magnitude mints a fresh proposal every time + # the estimator settles. One sustained movement is one proposal; a REVERSAL is a new + # one, which is exactly when a human should look again. + direction = 1 if per_month > 0 else -1 + ev = (f"{per_month:+.1f} pp/month over {t['n']} records in scope {scope!r}, " + f"last value z={t['z']:+.1f}{spike} (evidence {hist_q})") + if not hist_usable: + # The movement is real arithmetic over records that cannot say how they were + # acquired, so it is reported and never promoted. A trend over bounded sweeps is + # a trend in the sample, not in the population. + out.append(finding(cid, "OBSERVED", f"{k} share is moving", + ev + f" — history evidence is {hist_q}, not a population", + scope=scope, bucket=direction, + metric=f"slope of {k} share", invariants=(SCOPE_INVARIANT,))) + else: + out.append(finding(cid, "CANDIDATE", f"{k} share is moving", ev, scope=scope, + bucket=direction, hypothesis=f"the change in {k} is a shift, not a spike, and has a cause worth naming", metric=f"slope of {k} share", effect="unknown until measured", risk="a trend can come from the workload, not from SameWrite", @@ -415,6 +905,11 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): invariants=(SCOPE_INVARIANT,), benchmark="identify the cause before proposing a rule")) # 3. the guard: does it still pay for itself? (rule retirement is a first-class outcome) + # Its evidence is the hook's OWN ledger — writes it saw, no-ops it prevented — not a sample + # of transcripts. No sweep or history quality can make it better or worse, so the eligibility + # decision here is explicit and separate: the ledger's own sample size is its gate. + ledger_usable = (bool(ledger) and ledger.get("writes", 0) >= LEDGER_MIN_WRITES + and not ledger.get("rejected")) if ledger and ledger["writes"]: if ledger["writes"] >= LEDGER_MIN_WRITES: if ledger["rate"] >= LEDGER_MIN_NOOP_RATE: @@ -423,7 +918,9 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): f"({ledger['rate']:.1f}% >= {LEDGER_MIN_NOOP_RATE}%)", scope=scope, bucket=bucket_of(ledger["rate"]), metric="ledger deny rate")) else: - out.append(finding("noop-guard-retire", "CANDIDATE", "the no-op guard may have stopped paying", + out.append(finding("noop-guard-retire", + "CANDIDATE" if ledger_usable else "OBSERVED", + "the no-op guard may have stopped paying", f"{ledger['prevented']} of {ledger['writes']} writes prevented " f"({ledger['rate']:.1f}% < {LEDGER_MIN_NOOP_RATE}%)", scope=scope, bucket=bucket_of(ledger["rate"]), @@ -467,14 +964,35 @@ def analyse(live, hist, ledger, cold, scope="default", accept_partial=False): def overall_status(findings, hist, live, pop, accept_partial): + """The run's one word — and the thing that decides whether anything is written. + + CANDIDATE now precedes PARTIAL_EVIDENCE, which it did not before. That is not a loosening: a + finding only reaches CANDIDATE after its OWN evidence passed the promotion gate, so a run that + has one is a run whose promoted findings rest on evidence that was eligible. The old order + blocked a valid history-only finding because today's live sweep happened to be bounded — a + refusal by evidence the finding never used. + """ if pop.get("host_shift"): return "HOST_BEHAVIOR_SHIFT" - q = (live or {}).get("quality", "COMPLETE") if live else "COMPLETE" - if q in ("PARTIAL", "INVALID") and not accept_partial: - return "PARTIAL_EVIDENCE" if any(f["state"] == "CANDIDATE" for f in findings): return "CANDIDATE" - if not live and len(hist.get("comparable", [])) < MIN_HISTORY_FOR_TREND: + # Evidence that EXISTS and was refused is a different answer from evidence that is missing: + # PARTIAL_EVIDENCE tells the caller a flag or a full sweep would change the outcome, while + # INSUFFICIENT_DATA tells them to keep collecting. + refused = [] + if live: + refused.append(sweep_quality(live)) + comp = hist.get("comparable", []) + if comp: + refused.append(history_quality(comp)) + elif any(record_quality(r) == "INVALID" for r, _why in hist.get("dropped", [])): + # Evidence that exists and is refused. EMPTY is NOT one of these: a record that measured + # nothing is missing evidence, not bad evidence, and "keep collecting" is the honest word + # for it. A loss does not appear here at all any more — it moved the epoch instead. + refused.append("INVALID") + if any(not may_promote(q, accept_partial) for q in refused): + return "PARTIAL_EVIDENCE" + if not live and len(comp) < MIN_HISTORY_FOR_TREND: return "INSUFFICIENT_DATA" return "NO_ACTION" @@ -554,12 +1072,22 @@ def spec_text(f): """ -def emit_candidates(findings, outdir): - """Atomic, deduplicated, and never inside a governed tree by default. +def emit_candidates(findings, outdir, status): + """Atomic, deduplicated, never inside a governed tree by default — and only under a status that + permits promotion at all. -> (written, existing, failed). A candidate whose id already exists is NOT rewritten: the - evidence bucket is part of the id, so a file reappears only when the evidence actually moved.""" + evidence bucket is part of the id, so a file reappears only when the evidence actually moved. + + The status gate lives HERE rather than at the one call site that used to need it. A status that + refuses promotion while a file lands on disk is the worst outcome available: the operator reads + exit 30 or 40 and the next reader of the directory finds a specification that looks approved. + Putting the gate in the emitter means every caller — the CLI, the long-run simulator, a future + scheduler — routes through it, and no new caller can forget. + """ written, existing, failed = [], [], [] + if status != "CANDIDATE": + return written, existing, failed for f in findings: if f["state"] != "CANDIDATE": continue @@ -571,11 +1099,20 @@ def emit_candidates(findings, outdir): try: os.makedirs(d, exist_ok=True) tmp = p + ".tmp-%d" % os.getpid() - with open(tmp, "w", encoding="utf-8") as fh: - fh.write(spec_text(f)) - fh.flush() - os.fsync(fh.fileno()) - os.replace(tmp, p) # a crash leaves the old file or the new one, never half + try: + with open(tmp, "w", encoding="utf-8") as fh: + fh.write(spec_text(f)) + fh.flush() + os.fsync(fh.fileno()) + os.replace(tmp, p) # a crash leaves the old file or the new one, never half + except OSError: + # A write that failed mid-way must not leave its half behind: the next reader of + # this directory would find a file nobody promised. (cross-family review, round 4) + try: + os.unlink(tmp) + except OSError: + pass + raise written.append(f["candidate_id"]) except OSError as e: failed.append((f["candidate_id"], safe_err(e))) @@ -618,8 +1155,10 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop + (f" (history also holds: {', '.join(sorted(s for s in scopes_seen if s != scope))})" if len(scopes_seen) > 1 else "")) L.append("scope") + epoch_note = ("" if hist.get("in_epoch", hist["in_scope"]) == hist["in_scope"] + else f" ({hist.get('in_epoch', 0)} after the newest loss)") L.append(f" history : {sources['history'] or '(none)'} — {hist['total']} records, " - f"{hist['in_scope']} in this scope, {len(hist['comparable'])} comparable, " + f"{hist['in_scope']} in this scope{epoch_note}, {len(hist['comparable'])} comparable, " f"{sum(hist['rejected'].values())} rejected, time order {hist['time_order']}") for why, n in hist["rejected"].most_common(): L.append(f" rejected: {n} × {why}") @@ -631,7 +1170,22 @@ def render(live, hist, ledger, cold, pop, findings, sources, status, scope, scop L.append(f" live scan : {live['sessions']} sessions / {live['turns']:,} turns " f"({live['scanned']} transcripts, {live['short']} below the turn floor, " f"{live.get('unreadable', 0)} unreadable, {live.get('oversize', 0)} oversized lines)") - L.append(f" evidence : {live.get('quality', 'COMPLETE')}") + L.append(f" evidence : {sweep_quality(live)}" + + (f" (the sweep reports {live.get('quality')})" + if sweep_quality(live) != live.get("quality") else "")) + if hist["comparable"]: + L.append(f" history : {history_quality(hist['comparable'])} " + f"(worst of {len(hist['comparable'])} eligible records)") + dmg = hist.get("damage") or {} + if dmg.get("boundaries"): + L.append(f" damage : {dmg['boundaries']} loss boundary/ies in this file " + f"({dmg.get('file_global', 0)} unattributable, " + f"{sum((dmg.get('scope_local') or {}).values())} scope-local) — evidence from " + f"before the newest one is not combined with evidence after it") + if hist.get("run_id_conflicts"): + L.append(f" identity : {hist['run_id_conflicts']} record(s) repeat a run id with a " + f"different observation — an identity contradiction, counted as a conflict and " + f"cut where it appears, never reported as a retry") L.append("") if live and live["sessions"]: C = sum(live["carry"].values()) or 1 @@ -705,19 +1259,35 @@ def main(argv=None): "governed repository: the optimizer has no authority to change code.") a = ap.parse_args(argv) - recs, rejected, lines = load_history(a.history) + recs, rejected, lines, epochs = load_history(a.history) scopes = by_scope(recs) scopes_seen = sorted(scopes) or ["default"] + # The epoch is selected BEFORE anything is anchored: a record from before the newest loss must + # not choose the scope, the workload class, the corpus size or the trend for the population + # that came after it. + current = active_records(recs, epochs) if a.scope_id: - scope, scoped = a.scope_id, scopes.get(a.scope_id, []) + scope = a.scope_id elif recs: - scope = scope_of(max(recs, key=lambda r: r.get("ts") or 0)) - scoped = scopes[scope] + # Which scope gets analysed is an anchor too: taking the newest record of ANY quality let a + # single INVALID sweep in another agent's scope send the whole run to a population that was + # never going to be analysable. The newest record that CAN speak, IN THE CURRENT EPOCH, + # chooses; if none can, the newest record of that epoch still names it. + # Nothing from before the newest loss is consulted, not even as a label: the scope travels + # into every candidate id, so a stale population naming a ledger finding would attach a + # promotion to a population that no longer exists. (cross-family review of the B1 repair) + anchor = (eligible_anchor(current) + or (max(current, key=lambda r: r.get("ts") or 0) if current else None)) + scope = scope_of(anchor) if anchor is not None else "default" else: - scope, scoped = "default", [] + scope = "default" + scoped = [r for r in current if scope_of(r) == scope] keep, dropped = comparable(scoped) - hist = {"total": len(recs), "in_scope": len(scoped), "comparable": keep, "dropped": dropped, - "rejected": rejected, "lines": lines, "time_order": time_order(keep)} + damage = damage_summary(epochs) + hist = {"total": len(recs), "in_scope": len(scopes.get(scope, [])), "in_epoch": len(scoped), + "comparable": keep, "dropped": dropped, "rejected": rejected, "lines": lines, + "time_order": time_order(keep), "damage": damage, + "run_id_conflicts": (epochs or {}).get("run_id_conflicts", 0)} ledger = load_ledger(a.ledger) live = cold = None @@ -729,9 +1299,14 @@ def main(argv=None): except Exception: paths = [] if paths: - live = carry.accumulate(paths, min_turns=a.min_turns, max_files=a.max_files) + # Selected ONCE, then handed to both consumers. Calling bounded_paths() twice is two + # selections, and an mtime that becomes unreadable between them is two different samples + # again — the defect this function exists to close. (cross-family review, round 4) + selected = carry.bounded_paths(paths, a.max_files) + live = carry.accumulate(paths, min_turns=a.min_turns, max_files=a.max_files, + selected=selected) try: - listing, uses, sess = skills_tool.scan(paths[:a.max_files] if a.max_files else paths) + listing, uses, sess = skills_tool.scan(selected) if listing: ent = skills_tool.parse_listing(listing) rows = skills_tool.tally(ent, uses, sess) @@ -754,7 +1329,13 @@ def main(argv=None): if not got: status = "ALREADY_RUNNING" else: - written, existing, failed = emit_candidates(findings, a.emit_candidate) + written, existing, failed = emit_candidates(findings, a.emit_candidate, status) + if failed and not written: + # The run promised a candidate and nothing landed — an unwritable directory, a + # path that is a regular file, a full disk. Reporting CANDIDATE and exit 10 + # tells a scheduler a specification exists; the status has to say otherwise. + # (cross-family review, round 1) + status = "INTERNAL_ERROR" if a.json: print(json.dumps({ @@ -765,10 +1346,24 @@ def main(argv=None): "history_schema_supported": list(SCHEMA_SUPPORTED), "generated": int(time.time()), "status": status, "status_code": STATUS.get(status, STATUS["INTERNAL_ERROR"]), - "scope": {"analysed": scope, "known": scopes_seen, "records_in_scope": len(scoped), - "comparable": len(keep)}, - "evidence_quality": (live or {}).get("quality", "NO_SCAN") if live else "NO_SCAN", + "scope": {"analysed": scope, "known": scopes_seen, + "records_in_scope": len(scopes.get(scope, [])), + "records_in_epoch": len(scoped), "comparable": len(keep)}, + # DERIVED, not the producer's summary word: a sweep that lost records reads DEGRADED + # here even though the 1.3 label for it is PARTIAL (output_schema_version 2). + "evidence_quality": sweep_quality(live) if live else "NO_SCAN", "history": {"records": len(recs), "comparable": len(keep), + # the evidence eligible for THIS analysis, not a verdict on the file + "quality": history_quality(keep), + # ...and what the file holds regardless: a historical gap stays true while + # the evidence after it is independently complete + "damage": damage, + # An observable identity contradiction: two accepted records under one + # run_id whose persisted observations differ. A bounded COUNT, never a map + # keyed by anything a corrupt file chose — and separate from `rejected`, + # because a conflicting record is ACCEPTED and kept, not a line that failed + # to become one. + "run_id_conflicts": (epochs or {}).get("run_id_conflicts", 0), "rejected": dict(rejected), "time_order": hist["time_order"]}, "ledger": ledger, "live": ({"sessions": live["sessions"], "turns": live["turns"], "scanned": live["scanned"],