diff --git a/CHANGELOG.md b/CHANGELOG.md index 9f1b79a..a6c8522 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [1.2.1] - 2026-10-01 + +### Changed + +- Top regressions in the HTML report now read as expandable: the worst one + in each section starts open, each row has a boxed caret and a + "Show details" / "Hide details" hint, and the header reacts on hover and + shows a keyboard focus ring. Printing expands every regression in + Chromium-based browsers. + ## [1.2.0] - 2026-10-01 Shipped as a minor again. The `slices` removal below is breaking by the diff --git a/DOCS.md b/DOCS.md index ff79465..da3c322 100644 --- a/DOCS.md +++ b/DOCS.md @@ -12,7 +12,7 @@ hosted (opt-in) run history, diffs, PR gates The suite is the crux, so the capture SDK is the recommended way to build one: it records real production runs to disk and `evalshift capture sync` promotes them into golden suites. Hand-written suites are fully supported — see [The golden suite](#the-golden-suite). -- **Package name:** `evalshift` · **CLI entry point:** `evalshift` · **version:** 1.2.0 +- **Package name:** `evalshift` · **CLI entry point:** `evalshift` · **version:** 1.2.1 - **Python:** >= 3.11 · **License:** Apache-2.0 · **Status:** stable - **Local-first.** Runs, scores, stats, and reports all happen on your machine under `.evalshift/`. The only network calls are the model API calls you asked for — and, if you opt in, pushes to the hosted service. - **Four pieces:** CLI (this doc), SDK, GitHub Action, hosted server — each with its own machine-readable reference for AI tools. See [Ecosystem and AI-tool references](#ecosystem-and-ai-tool-references). diff --git a/llms-full.txt b/llms-full.txt index aae1cf2..18049cb 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1,7 +1,7 @@ # evalshift (CLI) — complete reference for AI tools Canonical hosted copy: https://www.evalshift.dev/cli-llms-full.txt -Package: evalshift (PyPI) | CLI entry point: evalshift | version: 1.2.0 +Package: evalshift (PyPI) | CLI entry point: evalshift | version: 1.2.1 Python: >=3.11 | license: Apache-2.0 | status: stable Install: pip install evalshift (or: uv pip install evalshift) diff --git a/pyproject.toml b/pyproject.toml index 6a8eb5a..0282394 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "evalshift" -version = "1.2.0" +version = "1.2.1" description = "Run your prompts on two LLMs and find out, with statistical confidence, what regressed." readme = "README.md" license = "Apache-2.0" diff --git a/src/evalshift_cli/reports/templates/report.css b/src/evalshift_cli/reports/templates/report.css index 2afa1e6..ed363db 100644 --- a/src/evalshift_cli/reports/templates/report.css +++ b/src/evalshift_cli/reports/templates/report.css @@ -556,6 +556,33 @@ details[open] > summary .caret::after { content: "\25BE"; } background: var(--strip); flex-wrap: wrap; } +/* The row must read as a control, not a static line: a boxed caret, a worded + hint, and a hover state that reacts across the whole header. */ +.reg-head .caret { + display: inline-flex; + align-items: center; + justify-content: center; + width: 22px; + height: 22px; + flex: none; + border: 1px solid var(--stroke); + background: var(--panel-2); +} +.reg-head .caret::after { font-size: 14px; width: auto; color: var(--body); } +.reg-toggle { + font-family: var(--font-mono); + font-size: 12px; + color: var(--dim); + min-width: 88px; + text-align: right; +} +.reg-toggle::after { content: "Show details"; } +.regression[open] > .reg-head .reg-toggle::after { content: "Hide details"; } +.reg-head:hover { background: var(--panel-2); } +.reg-head:hover .caret { border-color: var(--ok-edge); } +.reg-head:hover .caret::after, +.reg-head:hover .reg-toggle { color: var(--accent); } +.reg-head:focus-visible { outline: 2px solid var(--accent); outline-offset: -2px; } .reg-id { font-family: var(--font-mono); font-size: 14px; color: var(--fg); } .reg-what { font-size: 14px; @@ -762,6 +789,8 @@ footer p { margin: 0; } id instead. */ .reg-head { gap: 8px 12px; padding: 12px 14px; } .reg-what { flex: 1 1 60%; min-width: 0; white-space: normal; } + /* The boxed caret alone carries the affordance when space is tight. */ + .reg-toggle { display: none; } .reg-body { padding: 14px; } .trace-list { padding: 12px 12px 12px 26px; } .transcript-body { padding: 0 12px 12px 28px; } @@ -769,6 +798,10 @@ footer p { margin: 0; } @media print { body { background: #fff; color: #111; } + /* Paper can't be clicked: show every regression body. Best-effort: only + engines with ::details-content (Chromium 131+) honour it. */ + .regression::details-content { content-visibility: visible; } + .reg-toggle { display: none; } } /* --- Tone utilities ------------------------------------------------------ */ diff --git a/src/evalshift_cli/reports/templates/report.html.j2 b/src/evalshift_cli/reports/templates/report.html.j2 index a2f2fe1..d1ec7e1 100644 --- a/src/evalshift_cli/reports/templates/report.html.j2 +++ b/src/evalshift_cli/reports/templates/report.html.j2 @@ -613,7 +613,9 @@ {% if section.top_regressions %} {% for tr in section.top_regressions %} {% set reason = regression_reason(tr) %} -
+ {#- The worst regression starts open: seeing one expanded is what tells + the reader the rest expand too. -#} +
{{ tr.example_id }} @@ -622,6 +624,7 @@ {% if tr.truncated %}truncated (token cap){% endif %} {% if tr.target_empty_output %}empty output{% endif %} {{ "%+.3f"|format(tr.delta) }} +
diff --git a/tests/unit/test_reports.py b/tests/unit/test_reports.py index 251e6e8..6bd2525 100644 --- a/tests/unit/test_reports.py +++ b/tests/unit/test_reports.py @@ -724,6 +724,21 @@ def test_top_regression_renders_collapsed_input(self, tmp_path: Path) -> None: assert '
' in html assert "Greet the user named Alex." in html + def test_top_regressions_open_first_and_hint_expandable(self, tmp_path: Path) -> None: + cwd, run_id = _scaffold_full_run(tmp_path) + payload = build_report_payload(cwd / ".evalshift" / "runs" / run_id) + html = render_html(payload) + + opened = html.count('
') + closed = html.count('
') + # The fixture must yield several regressions for this to mean anything. + assert opened + closed >= 2 + # Only each section's worst regression starts expanded, so the reader + # sees one opened and learns the rest open too. + assert opened == sum(1 for ps in payload.prompt_sections if ps.top_regressions) + # Every row says it expands, in words, not just a caret glyph. + assert html.count(' None: cwd, run_id = _scaffold_full_run(tmp_path) payload = build_report_payload(cwd / ".evalshift" / "runs" / run_id)