Skip to content
Open

Test #571

Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
83 commits
Select commit Hold shift + click to select a range
b74f8dd
feat(bundle): exclude filter + knowledge-only preset for export
plind-junior Jul 10, 2026
bd98f8a
feat(bundle): export_check surfaces manifest counts and excluded
plind-junior Jul 10, 2026
fdf1520
feat(bundle): expose exclude filter on kb.export rpc and CLI
plind-junior Jul 10, 2026
0181cc5
feat(hub): client module - link config, token store, push/pull/status
plind-junior Jul 10, 2026
4380920
feat(cli): vouch hub link/push/pull/status
plind-junior Jul 10, 2026
3ecde10
feat(hub): personal catch-all kb, fallback capture, vouch adopt
plind-junior Jul 20, 2026
439ea3c
docs: personal catch-all kb + adopt in readmes and changelog
plind-junior Jul 20, 2026
8f0f0aa
fix(adopt): retire only durable claims; honest fallback surfaces
plind-junior Jul 20, 2026
4c3f0d6
docs(changelog): match the post-review adopt and fallback semantics
plind-junior Jul 20, 2026
048dada
Merge pull request #534 from vouchdev/feat/personal-kb
plind-junior Jul 20, 2026
ec1c440
feat(hooks): prompt gate — recall earns its place in a turn
plind-junior Jul 20, 2026
8d07646
refactor(hooks): delegate the recall decision to the model
plind-junior Jul 21, 2026
cca342d
fix(hooks): keep multi-digit query tokens; never assert an empty kb o…
plind-junior Jul 21, 2026
4514a39
Merge pull request #535 from vouchdev/feat/prompt-gate
plind-junior Jul 21, 2026
a19beec
feat(hub): gated federation import lands inbound knowledge as proposals
plind-junior Jul 21, 2026
95e9390
feat(sync): gated sync mode files inbound knowledge as proposals
plind-junior Jul 21, 2026
3c8139a
feat(context): surface federated provenance at read time
plind-junior Jul 21, 2026
97720c2
chore(merge): merge test into feat/vouchhub-gated-import
plind-junior Jul 21, 2026
16c8848
chore(schemas): regenerate context schemas for the federated origin f…
plind-junior Jul 21, 2026
1c9ca7c
Merge pull request #536 from vouchdev/feat/vouchhub-gated-import
plind-junior Jul 21, 2026
0020ea8
feat(compile): structured pages, wiki index/moc, two-phase drafting
plind-junior Jul 22, 2026
285bf79
Merge pull request #544 from vouchdev/feat/compile-structured-pages
plind-junior Jul 22, 2026
e32737d
feat(admission): gate knowledge-shaped garbage at the proposal funnel
plind-junior Jul 23, 2026
3beed68
Merge pull request #546 from vouchdev/feat/admission-gate
plind-junior Jul 23, 2026
5e14dab
fix(admission): reject emphasis-wrapped heading labels
plind-junior Jul 24, 2026
cda87e4
feat(worthiness): add tier-2 advisory claim-worthiness scoring
plind-junior Jul 24, 2026
0698041
Merge pull request #552 from vouchdev/fix/admission-heading-labels
plind-junior Jul 24, 2026
32c53d2
Merge pull request #553 from vouchdev/feat/worthiness-tier
plind-junior Jul 24, 2026
a0773ac
feat(webapp): batch-approve pending proposals from the review queue
plind-junior Jul 24, 2026
40c9df0
Merge pull request #554 from vouchdev/feat/webapp-batch-approve
plind-junior Jul 24, 2026
89a5de9
fix(webapp): unwrap the {items} envelope for kb.list_* results
plind-junior Jul 24, 2026
7420085
feat(webapp): attribute console approvals to a reviewer identity
plind-junior Jul 24, 2026
74c0106
Merge pull request #556 from vouchdev/feat/console-reviewer-identity
plind-junior Jul 24, 2026
a54069a
feat(kb): strip dead claim references at approve time and kb-wide
plind-junior Jul 24, 2026
93c4724
Merge pull request #557 from vouchdev/feat/approve-dead-claim-refs
plind-junior Jul 24, 2026
61b1358
feat(capture): extract answer claims once per session, not per turn
plind-junior Jul 24, 2026
2b8d48e
feat(capture): enrich session pages with dream-style subject extraction
plind-junior Jul 27, 2026
65f4d9e
feat(bench): seeded judge-free memory benchmark over the real pipeline
plind-junior Jul 27, 2026
5192045
fix(context): make sub-day recency real and apply it on every backend
plind-junior Jul 27, 2026
864d5c0
feat(retrieval): log every context-pack build to a local events file
plind-junior Jul 27, 2026
3543b5b
docs(bench): record the reference baseline table in the module
plind-junior Jul 27, 2026
99bde14
feat(capture): supersede updated claims from enrichment-detected changes
plind-junior Jul 27, 2026
1ec183a
feat(retrieval): pages-first packs and the receipt-coverage fidelity …
plind-junior Jul 27, 2026
da4d61c
ci(bench): season scoring workflow and competition rules
plind-junior Jul 27, 2026
0e45e38
Merge pull request #561 from vouchdev/feat/session-answer-mode
plind-junior Jul 27, 2026
aed4093
feat(competition): koth ladder - auto-merge kit PRs scored by vouchbench
plind-junior Jul 27, 2026
81787ef
docs(competition): fix stale band value in kit comment
plind-junior Jul 27, 2026
57db093
feat(bench): verifiability categories and the five-tool memory contract
plind-junior Jul 27, 2026
7a62b73
docs(competition): refresh ladder baseline for the v2 bench contract
plind-junior Jul 27, 2026
8bcec71
feat(competition): engine lane - submit ranking code, scored in a san…
plind-junior Jul 27, 2026
5548413
fix(bench): drop strategy kwarg that references an uncommitted module
plind-junior Jul 27, 2026
7243bd5
chore(competition): seed a weak round-0 canary champion on the ladder
plind-junior Jul 27, 2026
8811272
fix(ci): pin the koth gate checkout to the PR base branch
plind-junior Jul 27, 2026
3f286c9
feat(competition): challenger kit restoring the auto backend (#565)
plind-junior Jul 27, 2026
bc5f23b
feat(bench): local strategy scoring via --strategy/--against
plind-junior Jul 27, 2026
fe7d1e9
docs(competition): engine lane is the competition, kit lane the warm-up
plind-junior Jul 27, 2026
6a992ba
ci(competition): koth-ledger appends the throne row on dethrone
plind-junior Jul 27, 2026
e4900e2
chore(competition): sync ladder with engine lane and ledger automation
plind-junior Jul 27, 2026
0ee9c5b
chore(competition): seed a weak round-1 canary champion on the ladder
plind-junior Jul 27, 2026
02fa528
feat(competition): round-1 challenger kit restoring the auto backend …
plind-junior Jul 27, 2026
ee1bd45
ci(competition): ledger sweeps merged prs - auto-merge pushes fire no…
plind-junior Jul 27, 2026
670e34a
ci(competition): ledger pushes with an owner token past the ruleset
plind-junior Jul 27, 2026
989fb6e
docs(competition): ledger rows from sweep
github-actions[bot] Jul 27, 2026
71ff0ff
feat(retrieval): strategy ranks an over-fetched pool - exclusion is real
plind-junior Jul 28, 2026
70c7f7f
feat(competition): provenance-rank engine-lane submission
plind-junior Jul 28, 2026
bc0f120
ci(competition): ledger sweeps merged prs - auto-merge pushes fire no…
plind-junior Jul 27, 2026
1fbfe1a
ci(competition): ledger pushes with an owner token past the ruleset
plind-junior Jul 27, 2026
1513912
feat(retrieval): strategy ranks an over-fetched pool - exclusion is real
plind-junior Jul 28, 2026
01a7c03
docs(contributing): competition pr shapes for both koth lanes
plind-junior Jul 28, 2026
f82a536
docs(contributing): competition pr shapes for both koth lanes
plind-junior Jul 28, 2026
29137c2
ci(competition): zizmor annotations and shellcheck note for the gates
plind-junior Jul 28, 2026
36e4b8f
ci(competition): zizmor annotations and shellcheck note for the gates
plind-junior Jul 28, 2026
cacf49d
Merge pull request #567 from vouchdev/engine/provenance-rank
plind-junior Jul 28, 2026
819fdba
Merge pull request #569 from vouchdev/feat/koth-engine-lane
plind-junior Jul 28, 2026
df5018e
Merge branch 'test' into koth-ladder
plind-junior Jul 28, 2026
5baf77c
ci(competition): engine gate review fixes - sha pins and delete-only prs
plind-junior Jul 28, 2026
1dd468c
ci(competition): engine gate review fixes - sha pins and delete-only prs
plind-junior Jul 28, 2026
f28022e
Merge pull request #570 from vouchdev/koth-ladder
plind-junior Jul 28, 2026
d0c6f8d
Merge branch 'main' into test
plind-junior Jul 28, 2026
6f95ad5
feat(retrieval): promote the provenance champion to shipped default
plind-junior Jul 28, 2026
b806941
Merge pull request #572 from vouchdev/feat/koth-engine-lane
plind-junior Jul 28, 2026
b436ee3
feat(webapp): sessions view for compiled conversation history
plind-junior Jul 28, 2026
1fe8fb5
Merge pull request #575 from vouchdev/feat/webapp-sessions-view
plind-junior Jul 28, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
108 changes: 108 additions & 0 deletions .github/scripts/koth_score.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
"""Paired champion-vs-challenger scoring for the koth ladder.

Runs vouchbench twice per seed — once with the reigning kit, once with the
challenger kit — over seeds derived from the base sha and the utc date, and
applies the dethrone test from docs/vouchbench-seasons.md:

dethroned iff mean(challenger - champion) >= max(0.007, 1.96 x SE)

where SE is the standard error of the per-seed paired differences (common
random numbers: both arms see identical generated sessions per seed, so the
difference cancels seed-to-seed variance).

Seeds are deterministic for a given (base sha, utc day): anyone can recompute
the day's seeds and reproduce the run locally. Same-day overfitting to known
seeds is bounded by the margin band and settled monthly by the season's
commit-reveal scored run, which uses seeds that do not exist until the
cutoff.

Usage:
python koth_score.py --champion kit.yaml --challenger kit-pr.yaml \
--base-sha <sha> [--date YYYY-MM-DD] [--out report.json]

Exit code 0 = dethroned (challenger wins), 3 = held (champion stands),
1 = error. The distinct win/hold codes let the workflow branch without
parsing the report.
"""

from __future__ import annotations

import argparse
import datetime as dt
import hashlib
import json
from pathlib import Path

from vouch import bench

# 12 paired seeds, not 4: the band is built from the standard error of the
# paired differences, and n=4 makes that SE noisy and its normal-z reading
# optimistic. z=1.96 is the two-sided 95% normal quantile; with a small n
# the paired diffs are not perfectly normal, so the floor below still does
# the real gatekeeping when the SE collapses on a deterministic overfit.
N_SEEDS = 12
FLOOR = bench.DETHRONE_FLOOR
Z = bench.DETHRONE_Z
# bench sleeps this long between session ingests to spread created_at
# timestamps; keep it small — it is wall-clock, not simulated time. the
# 3600s that lived here briefly would have out-slept the CI job timeout
# many times over.
SESSION_GAP_SECONDS = 2.0


def day_seeds(base_sha: str, date: str, n: int = N_SEEDS) -> list[int]:
"""Derive the day's seed list from the champion sha and the utc date."""
seeds = []
for i in range(n):
digest = hashlib.sha256(f"{base_sha}:{date}:{i}".encode()).hexdigest()
seeds.append(int(digest[:12], 16))
return seeds


def score(kit_text: str, seeds: list[int]) -> list[float]:
extra = kit_text if kit_text.strip() else None
return [
bench.run(
s, extra_config=extra, session_gap_seconds=SESSION_GAP_SECONDS
)["composite"]
for s in seeds
]


def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--champion", required=True)
parser.add_argument("--challenger", required=True)
parser.add_argument("--base-sha", required=True)
parser.add_argument("--date", default=None)
parser.add_argument("--out", default=None)
args = parser.parse_args()

date = args.date or dt.datetime.now(dt.UTC).strftime("%Y-%m-%d")
seeds = day_seeds(args.base_sha, date)

champion_text = Path(args.champion).read_text(encoding="utf-8")
challenger_text = Path(args.challenger).read_text(encoding="utf-8")

champion_scores = score(champion_text, seeds)
challenger_scores = score(challenger_text, seeds)
verdict = bench.paired_verdict(
champion_scores, challenger_scores, floor=FLOOR, z=Z,
)

report = {
"date": date,
"base_sha": args.base_sha,
"seeds": seeds,
"lane": "kit",
**verdict,
}
text = json.dumps(report, indent=1)
print(text)
if args.out:
Path(args.out).write_text(text + "\n", encoding="utf-8")
return 0 if report["dethroned"] else 3


if __name__ == "__main__":
raise SystemExit(main())
82 changes: 82 additions & 0 deletions .github/scripts/score_strategy.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,82 @@
"""Paired champion-vs-challenger scoring for the engine (strategy) lane.

Same dethrone test as the kit lane (docs/vouchbench-seasons.md), but the arm
is a pluggable ranking strategy instead of a config fragment: each submission
is loaded as an untrusted file and run through vouch.strategy.SandboxProxy, so
the scored code executes only inside the sandbox child.

dethroned iff mean(challenger - champion) >= max(0.007, 1.96 x SE)

Exit code 0 = dethroned, 3 = held, 1 = error. Unlike the kit lane, a winning
verdict here does NOT auto-merge: engine code ships only through human review.
"""

from __future__ import annotations

import argparse
import datetime as dt
import hashlib
import json
from pathlib import Path

from vouch import bench
from vouch.strategy import SandboxProxy

N_SEEDS = 8
FLOOR = bench.DETHRONE_FLOOR
Z = bench.DETHRONE_Z
SESSION_GAP_SECONDS = 2.0


def day_seeds(base_sha: str, date: str, n: int = N_SEEDS) -> list[int]:
seeds = []
for i in range(n):
digest = hashlib.sha256(f"{base_sha}:{date}:{i}".encode()).hexdigest()
seeds.append(int(digest[:12], 16))
return seeds


def score(path: str, seeds: list[int]) -> list[float]:
proxy = SandboxProxy(path)
return [
bench.run(s, strategy=proxy, session_gap_seconds=SESSION_GAP_SECONDS)[
"composite"
]
for s in seeds
]


def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--champion", required=True)
parser.add_argument("--challenger", required=True)
parser.add_argument("--base-sha", required=True)
parser.add_argument("--date", default=None)
parser.add_argument("--out", default=None)
args = parser.parse_args()

date = args.date or dt.datetime.now(dt.UTC).strftime("%Y-%m-%d")
seeds = day_seeds(args.base_sha, date)

champion_scores = score(args.champion, seeds)
challenger_scores = score(args.challenger, seeds)
verdict = bench.paired_verdict(
champion_scores, challenger_scores, floor=FLOOR, z=Z,
)

report = {
"date": date,
"base_sha": args.base_sha,
"seeds": seeds,
"lane": "engine",
**verdict,
}
text = json.dumps(report, indent=1)
print(text)
if args.out:
Path(args.out).write_text(text + "\n", encoding="utf-8")
return 0 if report["dethroned"] else 3


if __name__ == "__main__":
raise SystemExit(main())
84 changes: 84 additions & 0 deletions .github/scripts/update_leaderboard.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,84 @@
"""Append a dethrone row to competition/LEADERBOARD.md from a scorecard.

The ledger row was appended by hand at season close; every dethrone that
lands on the ladder branch now writes its own row. Input is the report
json both scorers emit (koth_score.py for kits, score_strategy.py for
strategies — same shape, ``lane`` distinguishes them).

Idempotent on the PR number: if the ledger already cites the PR, the
script exits 0 without writing, so the ledger workflow re-running on its
own push converges instead of looping.

Usage:
python update_leaderboard.py --report report.json --pr 123 \
--author somebody [--ledger competition/LEADERBOARD.md]

Exit code 0 = row appended or already present, 1 = error.
"""

from __future__ import annotations

import argparse
import json
import re
from pathlib import Path


def next_row_number(ledger_text: str) -> int:
rows = re.findall(r"^\| (\d+) \|", ledger_text, flags=re.M)
return max((int(r) for r in rows), default=-1) + 1


def format_row(
number: int, report: dict, pr: int, author: str
) -> str:
lane = report.get("lane", "kit")
date = report.get("date", "")
mean = report["challenger"]["mean"]
margin = report["mean_diff"]
return (
f"| {number} | {author} ({lane}) | #{pr} | {date} "
f"| {mean:.4f} | +{margin:.4f} |"
)


def append_row(ledger: Path, report: dict, pr: int, author: str) -> bool:
"""Add the row unless the PR is already cited. True = file changed."""
text = ledger.read_text(encoding="utf-8")
if re.search(rf"\| #{pr} \|", text):
return False
row = format_row(next_row_number(text), report, pr, author)
if not text.endswith("\n"):
text += "\n"
# the table is the last block before the payout footer; append right
# after the final table row so the footer prose stays at the bottom
lines = text.splitlines(keepends=True)
last_row = max(
i for i, line in enumerate(lines) if line.startswith("| ")
)
lines.insert(last_row + 1, row + "\n")
ledger.write_text("".join(lines), encoding="utf-8")
return True


def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--report", required=True)
parser.add_argument("--pr", required=True, type=int)
parser.add_argument("--author", required=True)
parser.add_argument(
"--ledger", default="competition/LEADERBOARD.md",
)
args = parser.parse_args()

report = json.loads(Path(args.report).read_text(encoding="utf-8"))
if not report.get("dethroned"):
print("report is not a dethrone; nothing to append")
return 0
changed = append_row(Path(args.ledger), report, args.pr, args.author)
print("row appended" if changed else f"pr #{args.pr} already in ledger")
return 0


if __name__ == "__main__":
raise SystemExit(main())
107 changes: 107 additions & 0 deletions .github/scripts/validate_kit.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,107 @@
"""Validate a ladder kit file against the strict allowlist.

The kit is the one file an untrusted PR may change, and it is applied as
extra_config inside a write-token CI context — so the surface must be data
only. In particular, config keys that name executables (compile.llm_cmd,
enrich.llm_cmd) must never be reachable from a kit: a kit that could set a
command string would execute it on the runner. The allowlist below is
therefore closed-world: unknown keys are an error, not a warning.

Usage: python validate_kit.py <kit.yaml>
Exits 0 and prints the canonical yaml on success; exits 1 with a reason
on any violation.
"""

from __future__ import annotations

import sys
from pathlib import Path
from typing import Any

import yaml

MAX_KIT_BYTES = 4096

BACKENDS = {"auto", "fts5", "embedding", "hybrid", "substring"}

# key path -> (type, validator)
BOUNDS: dict[tuple[str, ...], Any] = {
("retrieval", "backend"): lambda v: isinstance(v, str) and v in BACKENDS,
("retrieval", "default_limit"): lambda v: (
isinstance(v, int) and not isinstance(v, bool) and 1 <= v <= 100
),
("retrieval", "prompt_gate", "enabled"): lambda v: isinstance(v, bool),
("retrieval", "recency", "enabled"): lambda v: isinstance(v, bool),
("retrieval", "recency", "half_life_days"): lambda v: (
isinstance(v, (int, float))
and not isinstance(v, bool)
and 0.01 <= float(v) <= 3650.0
),
("retrieval", "rerank", "enabled"): lambda v: isinstance(v, bool),
("retrieval", "rerank", "top_k"): lambda v: (
isinstance(v, int) and not isinstance(v, bool) and 1 <= v <= 500
),
("retrieval", "pages_first", "enabled"): lambda v: isinstance(v, bool),
("retrieval", "pages_first", "boost"): lambda v: (
isinstance(v, (int, float))
and not isinstance(v, bool)
and 0.1 <= float(v) <= 10.0
),
}

ALLOWED_BRANCHES = {path[:i] for path in BOUNDS for i in range(1, len(path))}


def _walk(node: Any, path: tuple[str, ...], errors: list[str]) -> None:
if isinstance(node, dict):
for key, value in node.items():
if not isinstance(key, str):
errors.append(f"non-string key at {'.'.join(path) or '<root>'}")
continue
child = (*path, key)
if child in BOUNDS:
if not BOUNDS[child](value):
errors.append(f"{'.'.join(child)}: value {value!r} out of bounds")
elif child in ALLOWED_BRANCHES:
_walk(value, child, errors)
else:
errors.append(f"{'.'.join(child)}: key not in allowlist")
else:
errors.append(f"{'.'.join(path) or '<root>'}: expected a mapping")


def validate(text: str) -> list[str]:
# an empty kit must NOT validate: over the contents-api, a file larger
# than the api's inline limit comes back with empty content, which would
# otherwise decode to "" and pass vacuously — the PR would then be scored
# as engine defaults rather than its real, oversized contents. require a
# real retrieval mapping so "empty" can never mean "champion defaults".
encoded = text.encode("utf-8")
if not encoded.strip():
return ["empty kit (nothing to score — did the fetch truncate?)"]
if len(encoded) > MAX_KIT_BYTES:
return [f"kit larger than {MAX_KIT_BYTES} bytes"]
try:
data = yaml.safe_load(text)
except yaml.YAMLError as exc:
return [f"not valid yaml: {exc}"]
if not isinstance(data, dict) or "retrieval" not in data:
return ["kit must be a mapping with a top-level 'retrieval' section"]
errors: list[str] = []
_walk(data, (), errors)
return errors


def main() -> int:
text = Path(sys.argv[1]).read_text(encoding="utf-8")
errors = validate(text)
if errors:
for err in errors:
print(f"kit rejected: {err}", file=sys.stderr)
return 1
print(text)
return 0


if __name__ == "__main__":
raise SystemExit(main())
Loading
Loading