Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
bcf0fd8
test: freeze the v1.4.2 counterexample matrix, RED on current main
ipeterpetrus Sep 21, 2026
1c97208
fix: one legacy evidence-quality law, an eligibility-first anchor, on…
ipeterpetrus Sep 21, 2026
3386780
test: a mutant per repaired invariant, and CI runs the new suite
ipeterpetrus Sep 21, 2026
f2a4394
fix: four more ways evidence could claim more than it showed
ipeterpetrus Sep 21, 2026
13d2cde
docs: record the cases added after the matrix was frozen
ipeterpetrus Sep 21, 2026
909d105
fix: four HIGHs a second reviewer family found, all of them "evidence…
ipeterpetrus Sep 21, 2026
588de91
fix: narrow container damage to lines that were lost, not lines that …
ipeterpetrus Sep 21, 2026
486004e
test: pin what the scope fallback can and cannot do
ipeterpetrus Sep 21, 2026
42a98f0
style: keep the counter vocabulary and the dedup floor in their own b…
ipeterpetrus Sep 21, 2026
ef0c749
fix: three HIGHs and a MEDIUM from the confirmation round
ipeterpetrus Sep 21, 2026
5cfcf7e
docs: the confirmation round's cases, in the same post-freeze section
ipeterpetrus Sep 22, 2026
fe78cbc
fix: round-4 findings — one selection per run, one impossibility rule…
ipeterpetrus Sep 22, 2026
aadebfb
test: freeze the B1 epoch matrix, RED on this branch's current head
ipeterpetrus Sep 22, 2026
a3067dc
fix: a loss cuts the history at its position instead of poisoning the…
ipeterpetrus Sep 22, 2026
9596b9e
report: name the epoch in the human report too
ipeterpetrus Sep 22, 2026
302b20a
test: both line orders for a run_id that spans a loss, and three copi…
ipeterpetrus Sep 22, 2026
056e138
fix: two ways the epoch repair still consulted the wrong population
ipeterpetrus Sep 22, 2026
e6c1f4f
docs: the two findings the cross-family lane made in the epoch repair
ipeterpetrus Sep 22, 2026
443abaf
docs(v1.4.2): freeze the trusted damage-boundary matrix before implem…
ipeterpetrus Sep 22, 2026
24c81d4
test(v1.4.2): RED — TUTF/TSCOPE/VD/MSG cases for the trusted damage b…
ipeterpetrus Sep 22, 2026
7f37713
fix(optimizer): a rejected line is never an authority for its own scope
ipeterpetrus Sep 22, 2026
189db57
fix(optimizer): one boundary per logical run, and a foreign record_ty…
ipeterpetrus Sep 22, 2026
71016ee
fix(optimizer): scope_id is an attribution authority, so it is checke…
ipeterpetrus Sep 22, 2026
c0749e8
docs(v1.4.2): freeze the run-id conflict matrix before implementing it
ipeterpetrus Sep 22, 2026
b0624a7
test(v1.4.2): RED — RID_01..RID_15, the run-id conflict matrix
ipeterpetrus Sep 22, 2026
6ef9988
fix(optimizer): a run_id is an identity claim, not proof of semantic …
ipeterpetrus Sep 22, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,10 @@ jobs:
run: python3 tests/test_evidence_phase2.py
- name: offline optimizer (privacy canary, corruption, zero policy mutation)
run: python3 tests/test_optimize.py
- name: legacy evidence integrity - the frozen v1.4.2 counterexample matrix
# Ambiguous, incomplete, invalid, non-comparable or host-shifted evidence must not produce
# a candidate specification unless the bounded-evidence exception was explicitly asked for.
run: python3 tests/test_evidence_integrity.py
- name: the 1.4.0 release ships shadow evaluation and nothing that writes
# Sebuah rilis paling mudah "menyalakan" sesuatu tanpa sengaja saat versinya dinaikkan.
# Suite ini yang membuat kalimat "promotion/persistence tidak aktif" bisa diperiksa.
Expand Down
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -369,7 +369,7 @@ docs/VNEXT.md build report · docs/RELEASE_NOTES_1.1.0.md · _1.2.
docs/reference-audits/ — nine projects read at pinned commits
docs/MULTI_AGENT.md many agents, running all the time · docs/AI_VOS_PROFILE.md (one profile)
docs/EVIDENCE_CONTRACT_V1_4.md the frozen contract the kernel implements
tests/ 1201 assertions in sixteen suites, mutation-tested; CI on Python 3.9 and 3.12
tests/ 1601 assertions in seventeen suites, mutation-tested; CI on Python 3.9 and 3.12
plus two shell suites: the OpenClaw one-liner's cleanup, the OpenClaw host
```

Expand Down
627 changes: 627 additions & 0 deletions docs/V142_COUNTEREXAMPLES.md

Large diffs are not rendered by default.

3 changes: 1 addition & 2 deletions experiments/aivos/bash_residue.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,8 +44,7 @@ def main():
a = ap.parse_args()

paths, _ = profiles.resolve(a.scan or [])
if a.max_files and len(paths) > a.max_files:
paths = sorted(paths, key=os.path.getmtime, reverse=True)[:a.max_files]
paths = carry.bounded_paths(paths, a.max_files) # one definition of "the newest N"

sizes = collections.defaultdict(list)
total = collections.Counter()
Expand Down
19 changes: 14 additions & 5 deletions experiments/aivos/longrun.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,8 +30,12 @@


def rec(ts, shares, scope, turns=1000, workload="", runtimes=None):
# The acquisition counters a real 1.2/1.3 sweep always wrote. Since 1.4.2 a record claiming
# COMPLETE without them cannot carry a promotion, so a simulation that omitted them would be
# simulating a population no producer ever writes.
return {"schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(),
"ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7,
"ts": ts, "sessions": 40, "turns": turns, "carry_bytes": 10 ** 7, "scanned": 100,
"unreadable": 0, "oversize": 0, "skipped_by_limit": 0,
"scope_id": scope, "workload_class": workload, "evidence_quality": "COMPLETE",
"runtimes": runtimes or {"2.1.270": 40}, "models": {"m1": 40},
"shares": shares, "bpt": {k: 1.0 for k in shares}}
Expand Down Expand Up @@ -154,17 +158,22 @@ def longrun(days, cycles, out, plateau=20):

day_new = 0
for _cycle in range(cycles):
recs, _rej, _lines = optimize.load_history(hist)
scopes = optimize.by_scope(recs)
recs, _rej, _lines, ep = optimize.load_history(hist)
# the same epoch the CLI analyses: evidence from before a loss is not combined with
# evidence after it, here either (cross-family review of the B1 repair)
current = optimize.active_records(recs, ep)
scopes = optimize.by_scope(current)
for role in ROLES:
keep, dropped = optimize.comparable(scopes.get(role, []))
h = {"comparable": keep, "total": len(recs), "in_scope": len(scopes.get(role, [])),
"rejected": {}, "dropped": dropped, "time_order": optimize.time_order(keep)}
"in_epoch": len(scopes.get(role, [])), "rejected": dict(_rej),
"dropped": dropped, "time_order": optimize.time_order(keep),
"damage": optimize.damage_summary(ep)}
lv = live(int(30 + min(day, plateau) * (30.0 / plateau)))
f = optimize.analyse(lv, h, None, None, scope=role)
st = optimize.overall_status(f, h, lv, optimize.population(keep), False)
statuses[st] += 1
w, e, _fail = optimize.emit_candidates(f, cand)
w, e, _fail = optimize.emit_candidates(f, cand, st)
written += len(w)
existing += len(e)
day_new += len(w)
Expand Down
25 changes: 24 additions & 1 deletion experiments/aivos/readiness.py
Original file line number Diff line number Diff line change
Expand Up @@ -130,7 +130,11 @@ def main():
"schema_version": 2, "record_type": "carry_run", "run_id": carry.new_run_id(),
"ts": 1_750_000_000 + i * 604800, "sessions": 60, "turns": 3000,
"carry_bytes": 10 ** 8, "scope_id": "governed", "workload_class": "audit",
"evidence_quality": "COMPLETE", "runtimes": {"2.1.271": 60}, "models": {"m": 60},
"evidence_quality": "COMPLETE", "scanned": 120,
# the acquisition counters the producer writes; since 1.4.2 a COMPLETE claim
# without them cannot promote (tests/test_evidence_integrity.py)
"unreadable": 0, "oversize": 0, "skipped_by_limit": 0,
"runtimes": {"2.1.271": 60}, "models": {"m": 60},
"shares": {"Bash": 40.0 + i * 9, "Read": 60.0 - i * 9},
"bpt": {"Bash": 1.0, "Read": 1.0}}) + "\n")

Expand All @@ -153,6 +157,25 @@ def main():
check("JSON leaks no secret", CANARY in rj.stdout, False)
check("JSON leaks no path", ("/home/" in rj.stdout) or (ws in rj.stdout), False)

# The same population with the counters stripped claims a completeness it cannot show: it must
# refuse, and it must write nothing. Without this, the fixture edit above could hide the gate.
bare_hist = os.path.join(state, "bare_history.jsonl")
bare_cand = os.path.join(state, "bare_candidates")
with open(hist, encoding="utf-8") as fh, open(bare_hist, "w", encoding="utf-8") as out_fh:
for line in fh:
o = json.loads(line)
for k in ("unreadable", "oversize", "skipped_by_limit"):
o.pop(k, None)
out_fh.write(json.dumps(o) + "\n")
rb = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "optimize.py"),
"--history", bare_hist, "--ledger", os.path.join(state, "none.jsonl"),
"--scan", "--scope-id", "governed", "--json", "--strict-exit",
"--emit-candidate", bare_cand],
capture_output=True, text=True, timeout=900)
check("a record that cannot attest its own sweep does not promote", rb.returncode, 40)
check("and nothing is written for it",
[f for r_, _d, fs in os.walk(bare_cand) for f in fs], [])

specs = [os.path.join(r_, f) for r_, _d, fs in os.walk(cand) for f in fs if f.endswith(".md")]
check("a specification was written, outside the governed repository", bool(specs), True)
check("specifications live outside the workspace",
Expand Down
Loading
Loading