From 00b56ba5f17ce0440f0d301ec39935180836c8fc Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Sat, 3 Oct 2026 22:36:21 +0200 Subject: [PATCH] chore: stop tracking docs/superpowers plans and specs docs/superpowers is already gitignored, but its files had been force-added. Remove the shipped plans and specs and untrack the one plan still in progress, which stays on disk only. Repoint the three references that named a deleted spec at the user docs that now carry the same content. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 - .../plans/2026-06-09-cli-agent-traces.md | 180 --- .../plans/2026-07-21-strong-default-init.md | 203 --- .../2026-08-23-temperature-value-rejection.md | 735 --------- .../2026-09-08-external-review-response.md | 488 ------ ...026-09-18-policy-single-source-of-truth.md | 327 ---- .../plans/2026-09-21-stale-docs-cleanup.md | 1120 ------------- .../plans/2026-09-30-deepseek-provider.md | 1430 ----------------- .../2026-06-09-cli-agent-traces-design.md | 332 ---- .../2026-07-21-strong-default-init-design.md | 179 --- .../2026-07-23-ci-llms-full-link-design.md | 67 - ...8-23-temperature-value-rejection-design.md | 126 -- .../2026-09-09-namespace-collision-design.md | 182 --- ...2026-09-09-teacher-forced-replay-design.md | 340 ---- pyproject.toml | 2 +- tests/unit/test_suite_models.py | 4 +- 16 files changed, 3 insertions(+), 5713 deletions(-) delete mode 100644 docs/superpowers/plans/2026-06-09-cli-agent-traces.md delete mode 100644 docs/superpowers/plans/2026-07-21-strong-default-init.md delete mode 100644 docs/superpowers/plans/2026-08-23-temperature-value-rejection.md delete mode 100644 docs/superpowers/plans/2026-09-08-external-review-response.md delete mode 100644 docs/superpowers/plans/2026-09-18-policy-single-source-of-truth.md delete mode 100644 docs/superpowers/plans/2026-09-21-stale-docs-cleanup.md delete mode 100644 docs/superpowers/plans/2026-09-30-deepseek-provider.md delete mode 100644 docs/superpowers/specs/2026-06-09-cli-agent-traces-design.md delete mode 100644 docs/superpowers/specs/2026-07-21-strong-default-init-design.md delete mode 100644 docs/superpowers/specs/2026-07-23-ci-llms-full-link-design.md delete mode 100644 docs/superpowers/specs/2026-08-23-temperature-value-rejection-design.md delete mode 100644 docs/superpowers/specs/2026-09-09-namespace-collision-design.md delete mode 100644 docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md diff --git a/CHANGELOG.md b/CHANGELOG.md index a6c8522..ada125e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -616,7 +616,6 @@ the key and the fix, which is the whole migration. that imported CLI internals (`from evalshift.models.client import …`) must import from `evalshift_cli`. No shim is possible — shipping any `evalshift/` file would recreate the collision — so this ships as a minor bump (0.14.0). - Design: `docs/superpowers/specs/2026-09-09-namespace-collision-design.md`. - Docs: the capture guides (`README.md`, `DOCS.md`, `docs/sdk.md`, `docs/getting-started.md`, `llms-full.txt`) now cover the SDK 0.4.0 provider client wrappers (`wrap_openai` / `wrap_anthropic` / `wrap_genai`), the diff --git a/docs/superpowers/plans/2026-06-09-cli-agent-traces.md b/docs/superpowers/plans/2026-06-09-cli-agent-traces.md deleted file mode 100644 index 7eb8caa..0000000 --- a/docs/superpowers/plans/2026-06-09-cli-agent-traces.md +++ /dev/null @@ -1,180 +0,0 @@ -# CLI Agent Traces Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. - -**Goal:** Add CLI-only bring-your-own-agent trace import, comparison, debugging, and report support. - -**Architecture:** Add a new `evalshift.traces` package for strict trace models, JSONL loading, pairing, and diffing. Add an `agent_trace` evaluator family that consumes imported `traces.jsonl` pairs and emits existing `EvalRecord`s so `evaluate -> analyze -> report` remains the pipeline. Extend debug commands and report payload/rendering to prefer imported agent traces when present. - -**Tech Stack:** Python 3.14, Typer, Pydantic v2, Rich, pytest, existing EvalShift runner/evaluator/report modules. - ---- - -### Task 1: Trace Models And Loader - -**Files:** -- Create: `src/evalshift/traces/__init__.py` -- Create: `src/evalshift/traces/models.py` -- Create: `src/evalshift/traces/loader.py` -- Test: `tests/unit/test_trace_models.py` - -- [ ] **Step 1: Write failing trace model tests** - -Create `tests/unit/test_trace_models.py` with tests for model validation, event ordering, duplicate `sequence_index`, invalid tool-result references, JSONL line-numbered load errors, and pair indexing. - -- [ ] **Step 2: Run tests to verify failure** - -Run: `uv run pytest tests/unit/test_trace_models.py -q` -Expected: FAIL because `evalshift.traces` does not exist. - -- [ ] **Step 3: Implement trace models and loader** - -Implement strict Pydantic models with discriminated event types, `TRACES_FILENAME = "traces.jsonl"`, `load_traces_jsonl`, `write_traces_jsonl`, `index_traces`, and `pairs_for_prompt_examples`. - -- [ ] **Step 4: Run tests to verify pass** - -Run: `uv run pytest tests/unit/test_trace_models.py -q` -Expected: PASS. - -### Task 2: Trace Import Command - -**Files:** -- Create: `src/evalshift/cli/commands/traces.py` -- Modify: `src/evalshift/cli/main.py` -- Test: `tests/unit/test_trace_import_command.py` - -- [ ] **Step 1: Write failing CLI import tests** - -Create tests that scaffold a completed run, import source/target trace JSONL files, assert `.evalshift/runs//traces.jsonl` is written, assert bad example ids fail, and assert `--strict` fails when a pair is missing. - -- [ ] **Step 2: Run tests to verify failure** - -Run: `uv run pytest tests/unit/test_trace_import_command.py -q` -Expected: FAIL because `evalshift traces` is not registered. - -- [ ] **Step 3: Implement import command** - -Add `traces_app = typer.Typer(...)`, `import_traces(...)`, run artifact validation via `read_state` and `iter_calls`, source/target role validation, normalized write, and Rich summary output. - -- [ ] **Step 4: Run tests to verify pass** - -Run: `uv run pytest tests/unit/test_trace_import_command.py -q` -Expected: PASS. - -### Task 3: Agent Trace Evaluator - -**Files:** -- Modify: `src/evalshift/config/models.py` -- Modify: `src/evalshift/evaluators/failures.py` -- Create: `src/evalshift/evaluators/agent_trace.py` -- Modify: `src/evalshift/cli/commands/evaluate.py` -- Test: `tests/unit/test_config_models.py` -- Test: `tests/unit/test_agent_trace_evaluator.py` -- Test: `tests/unit/test_evaluate_command.py` - -- [ ] **Step 1: Write failing config and evaluator tests** - -Add tests for parsing `evaluators.agent_trace`, scoring missing verification before dangerous tools, argument drift, extra dangerous tools, and evaluate failure when `agent_trace` is configured without imported traces. - -- [ ] **Step 2: Run tests to verify failure** - -Run: `uv run pytest tests/unit/test_config_models.py tests/unit/test_agent_trace_evaluator.py tests/unit/test_evaluate_command.py -q` -Expected: FAIL because config/evaluator support does not exist. - -- [ ] **Step 3: Implement config, failure constants, evaluator, and evaluate dispatch** - -Add `AgentTraceEvaluatorConfig`, failure constants, `AgentTraceEvaluator.score_trace_pair`, and a separate agent-trace branch in `run_evaluate` that loads `traces.jsonl` only when configured. - -- [ ] **Step 4: Run tests to verify pass** - -Run: `uv run pytest tests/unit/test_config_models.py tests/unit/test_agent_trace_evaluator.py tests/unit/test_evaluate_command.py -q` -Expected: PASS. - -### Task 4: Trace Diff And Debug Commands - -**Files:** -- Create: `src/evalshift/traces/diff.py` -- Modify: `src/evalshift/cli/commands/debug_artifacts.py` -- Modify: `src/evalshift/cli/commands/diff.py` -- Modify: `src/evalshift/cli/commands/inspect.py` -- Modify: `src/evalshift/cli/commands/replay.py` -- Test: `tests/unit/test_trace_diff.py` -- Test: `tests/unit/test_debug_commands.py` - -- [ ] **Step 1: Write failing diff/debug tests** - -Add tests for missing/extra/reordered trace diff items, argument field deltas, `diff case` preferring trace timelines, `inspect case` showing a trace summary, and `replay case --trace` printing normalized JSON. - -- [ ] **Step 2: Run tests to verify failure** - -Run: `uv run pytest tests/unit/test_trace_diff.py tests/unit/test_debug_commands.py -q` -Expected: FAIL because trace diff/debug support does not exist. - -- [ ] **Step 3: Implement trace diff and debug rendering** - -Add reusable diff models and renderers, load traces from debug helpers, and wire debug commands to fall back to existing text behavior when traces are absent. - -- [ ] **Step 4: Run tests to verify pass** - -Run: `uv run pytest tests/unit/test_trace_diff.py tests/unit/test_debug_commands.py -q` -Expected: PASS. - -### Task 5: Report And Docs - -**Files:** -- Modify: `src/evalshift/reports/json.py` -- Modify: `src/evalshift/reports/templates/report.html.j2` -- Modify: `src/evalshift/reports/templates/report.css` -- Modify: `docs/configuration.md` -- Modify: `docs/getting-started.md` -- Create: `docs/traces.md` -- Create: `examples/agent-traces/evalshift.yaml` -- Create: `examples/agent-traces/golden.jsonl` -- Create: `examples/agent-traces/source-traces.jsonl` -- Create: `examples/agent-traces/target-traces.jsonl` -- Test: `tests/unit/test_reports.py` -- Test: `tests/integration/test_agent_trace_pipeline.py` - -- [ ] **Step 1: Write failing report/integration tests** - -Add tests that report payload includes `source_agent_trace`, `target_agent_trace`, and `trace_diff` for `agent_trace` regressions, HTML renders trace rows, and `traces import -> evaluate -> analyze -> report` works on fixtures. - -- [ ] **Step 2: Run tests to verify failure** - -Run: `uv run pytest tests/unit/test_reports.py tests/integration/test_agent_trace_pipeline.py -q` -Expected: FAIL because report/docs fixtures are not implemented. - -- [ ] **Step 3: Implement report serialization/rendering and docs/examples** - -Load `traces.jsonl` in report payload, attach trace data to top regressions, add timeline rendering, document CLI trace usage, and add a minimal example project. - -- [ ] **Step 4: Run targeted tests to verify pass** - -Run: `uv run pytest tests/unit/test_reports.py tests/integration/test_agent_trace_pipeline.py -q` -Expected: PASS. - -### Task 6: Final Verification - -**Files:** -- Verify whole repo. - -- [ ] **Step 1: Run unit/integration targets** - -Run: `uv run pytest tests/unit/test_trace_models.py tests/unit/test_trace_import_command.py tests/unit/test_agent_trace_evaluator.py tests/unit/test_trace_diff.py tests/unit/test_debug_commands.py tests/integration/test_agent_trace_pipeline.py -q` -Expected: PASS. - -- [ ] **Step 2: Run lint, format check, and type check** - -Run: - -```bash -uv run ruff check . -uv run ruff format --check . -uv run mypy --strict src/evalshift -``` - -Expected: all PASS. - -- [ ] **Step 3: Update changelog if needed** - -Add a concise `CHANGELOG.md` entry under `## [Unreleased]` for CLI agent trace import/evaluation. diff --git a/docs/superpowers/plans/2026-07-21-strong-default-init.md b/docs/superpowers/plans/2026-07-21-strong-default-init.md deleted file mode 100644 index 90befce..0000000 --- a/docs/superpowers/plans/2026-07-21-strong-default-init.md +++ /dev/null @@ -1,203 +0,0 @@ -# Strong-Default `evalshift init` Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. - -**Goal:** Make the config `evalshift init` generates give honest verdicts with zero edits: advisory vs blocking evaluators, CI-aware small-n budgets, judge fixes, provider-aware scaffold, capture dedup. - -**Architecture:** Additive `blocking` flag threaded config → evaluator → `EvalRecord` → policy engine; Wilson-interval budget checks in `analysis/policy.py`; failure paths raise instead of neutral-scoring; `init` gains a provider prompt that parameterises the scaffold template; `capture sync` dedupes on `input_hash`. - -**Tech Stack:** Python 3.14, pydantic v2 (`extra="forbid"`), typer, litellm, pytest. - -**Spec:** `docs/superpowers/specs/2026-07-21-strong-default-init-design.md` - -## Global Constraints - -- Python 3.14+, `mypy --strict` clean, `ruff check` + `ruff format` clean. -- All modules `from __future__ import annotations`; Google-style docstrings. -- Config models stay `extra="forbid"`. -- New config field ⇒ doc entry in `docs/configuration.md`; user-visible change ⇒ `CHANGELOG.md` under `## [Unreleased]`. -- **Deviation:** NO git commits — the working tree holds the user's uncommitted in-flight work in overlapping files. Commit steps are replaced by full-suite verify checkpoints; user reviews the diff. - ---- - -### Task 1: `blocking` flag on evaluator configs - -**Files:** -- Modify: `src/evalshift/config/models.py` (7 evaluator config classes) -- Test: `tests/unit/test_config.py` - -**Interfaces:** -- Produces: `cfg.evaluators.semantic.blocking: bool` (etc. on all 7 evaluator config models), default `True`. - -- [ ] **Step 1: Write failing tests** — `blocking` parses per model, defaults `True`, typo still rejected: - -```python -def test_evaluator_blocking_flag_parses_and_defaults() -> None: - cfg = load_config(write_yaml(...""" -evaluators: - semantic: - embedding_model: gemini/gemini-embedding-001 - blocking: false - llm_judge: - - criterion_name: eq - criterion_prompt: which is better - blocking: false - structural: - - type: length - min_chars: 1 -""")) - assert cfg.evaluators.semantic.blocking is False - assert cfg.evaluators.llm_judge[0].blocking is False - assert cfg.evaluators.structural[0].blocking is True -``` - -- [ ] **Step 2: Run, verify FAIL** (`extra_forbidden` for `blocking`). -- [ ] **Step 3: Implement** — add to each of the 7 evaluator config models: - -```python - blocking: bool = True -``` - -with docstring line: `blocking: Whether regressions from this evaluator can fail the migration verdict. Advisory (false) evaluators still score and appear in reports.` - -- [ ] **Step 4: Run, verify PASS.** - -### Task 2: `EvalRecord.blocking` + stamping at evaluate time - -**Files:** -- Modify: `src/evalshift/evaluators/base.py` (EvalRecord) -- Modify: `src/evalshift/cli/commands/evaluate.py` (`_build_evaluators`, record construction sites) -- Test: `tests/unit/test_evaluate_command.py` (or existing evaluate test file) - -**Interfaces:** -- Produces: `EvalRecord.blocking: bool = True`; every record written by `run_evaluate` carries the config value. -- Mechanism: `_build_evaluators` sets `evaluator.blocking = ` attribute on each instance; all `EvalRecord(...)` constructions in `evaluate.py` pass `blocking=getattr(evaluator, "blocking", True)`. - -- [ ] **Step 1: Failing test** — config with `llm_judge.blocking: false` + fake evaluator run ⇒ judge records have `blocking is False`, others `True`; old scores.jsonl row without the key loads as `True`. -- [ ] **Step 2: Verify FAIL.** -- [ ] **Step 3: Implement** (field + stamping at all 5 `EvalRecord(` sites in evaluate.py). -- [ ] **Step 4: Verify PASS.** - -### Task 3: Failure paths raise; `drop_params` - -**Files:** -- Modify: `src/evalshift/evaluators/llm_judge.py` (remove neutral-tie fallback) -- Modify: `src/evalshift/evaluators/semantic.py` (remove 0/0 fallback) -- Modify: `src/evalshift/models/client.py` (add `"drop_params": True` to completion kwargs in `complete_messages` and `complete_messages_with_tools`) -- Modify: `src/evalshift/evaluators/base.py` (protocol docstring: evaluators may raise; harness records the error) -- Test: `tests/unit/test_llm_judge.py`, `tests/unit/test_semantic.py`, `tests/unit/test_client.py` - -**Interfaces:** -- Produces: judge/semantic `.score()` raises `EvaluatorError` on API/parse failure; `_score_one` (already) converts raises to `error=`-stamped records excluded by `_metrics`. - -- [ ] **Step 1: Failing tests** — judge with client that raises ⇒ `pytest.raises(EvaluatorError)`; same for semantic embedding failure; client kwargs include `drop_params: True`. -- [ ] **Step 2: Verify FAIL.** -- [ ] **Step 3: Implement:** - -```python - except Exception as exc: - raise EvaluatorError( - f"judge call failed for {prompt_id}/{example_id}: {exc}" - ) from exc -``` - -(semantic analogous: `embedding failed for ...`). Client: add `"drop_params": True` beside `"temperature"` in both kwargs dicts. - -- [ ] **Step 4: Verify PASS; check `_score_one` integration test writes `error=` record.** - -### Task 4: Policy engine — advisory partition + Wilson CI budgets - -**Files:** -- Modify: `src/evalshift/analysis/policy.py` -- Test: `tests/unit/test_policy.py` - -**Interfaces:** -- Produces: `MigrationDecision.advisory: PolicyMetricSummary | None = None`, `MigrationDecision.advisory_regressions: list[BlockingRegression]` (default empty), `BudgetResult.ci_low/ci_high: float | None = None`, `BudgetResult.conclusive: bool = True`; `_wilson_interval(count: int, n: int, z: float = 1.96) -> tuple[float, float]`. -- Verdict semantics: observed within budget → pass; breach with CI excluding budget → fail; breach with CI straddling → `inconclusive` + `reason`. - -- [ ] **Step 1: Failing tests:** - -```python -def test_advisory_records_do_not_gate(): ... # 4 advisory regressions, 0 blocking → verdict pass, advisory.regression_rate == 1.0 -def test_small_n_breach_is_inconclusive(): ... # n=8 blocking, 3 regressions (0.375 > 0.30), Wilson low < 0.30 → "inconclusive", reason mentions n -def test_large_n_breach_fails(): ... # n=400, 160 regressions (0.40), Wilson low > 0.30 → "fail" -def test_clean_small_n_passes(): ... # n=8, 0 regressions → "pass" despite wide CI -def test_advisory_comparisons_not_blocking(): ... # critical semantic comparison from advisory evaluator → advisory_regressions, verdict pass -``` - -- [ ] **Step 2: Verify FAIL.** -- [ ] **Step 3: Implement** — partition records on `r.blocking`; derive `advisory_evaluators = {r.evaluator_name for r in records if not r.blocking}`; filter comparisons for `_verdict_for`/`_blocking_regressions`; Wilson: - -```python -def _wilson_interval(count: int, n: int, z: float = 1.96) -> tuple[float, float]: - if n <= 0: - return (0.0, 1.0) - p = count / n - denom = 1 + z * z / n - centre = (p + z * z / (2 * n)) / denom - margin = (z / denom) * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) - return (max(0.0, centre - margin), min(1.0, centre + margin)) -``` - -Rate budgets get `ci_low/ci_high/conclusive`; `_verdict_for` gains the three-way logic; `reason` composed for inconclusive. `inconclusive_decision` unchanged. Slice decisions use the same logic. - -- [ ] **Step 4: Verify PASS; run full `pytest tests/unit/test_policy.py`.** - -### Task 5: `capture sync` dedup - -**Files:** -- Modify: `src/evalshift/cli/commands/capture.py` (`capture_sync`) -- Test: `tests/unit/test_capture_sync.py` (or existing capture test file) - -**Interfaces:** -- Produces: duplicate `(suite, input_hash)` captures skipped (first wins) with summary count; `--keep-duplicates` flag preserves old behavior. - -- [ ] **Step 1: Failing test** — two captures, same suite + input_hash ⇒ 1 promoted, summary contains "duplicate"; with `--keep-duplicates` ⇒ 2 promoted. -- [ ] **Step 2: Verify FAIL.** -- [ ] **Step 3: Implement** in the `records_by_suite` build loop: - -```python - seen_inputs: set[tuple[str, str]] = set() - skipped_duplicates = 0 - ... - key = (envelope.suite, envelope.input_hash) - if not keep_duplicates and envelope.input_hash and key in seen_inputs: - skipped_duplicates += 1 - continue - seen_inputs.add(key) -``` - -plus summary segment `f", skipped {skipped_duplicates} duplicate capture(s) (same input)"` and the typer option. - -- [ ] **Step 4: Verify PASS.** - -### Task 6: Provider-aware, judge-fixed init scaffold - -**Files:** -- Modify: `src/evalshift/cli/commands/init.py` (template → function of provider; `--provider` option; TTY prompt) -- Test: `tests/unit/test_init.py` - -**Interfaces:** -- Produces: `render_minimal_config(*, profile: str, provider: str = "gemini") -> str`; `--provider [gemini|openai|anthropic]`; non-TTY default `gemini`. -- Scaffold content changes: symmetric criterion prompt (spec §3), `blocking: false` on semantic + llm_judge with comment, provider model table (spec §5), judge-family comment. - -- [ ] **Step 1: Failing tests** — `--provider openai` config contains `gpt-5.4-mini`/`gpt-5.6-luna`/`text-embedding-3-small` and NOT `gemini-`; anthropic variant comments out semantic; default remains gemini; criterion has no "TARGET"/"SOURCE" tokens; `blocking: false` present twice; generated YAML round-trips `load_config`. -- [ ] **Step 2: Verify FAIL.** -- [ ] **Step 3: Implement** — `_PROVIDER_MODELS: dict[str, ...]` table; template with `{source_model}`/`{judge_model}`/`{semantic_block}` slots; prompt via `typer.prompt` guarded by `sys.stdin.isatty()`. -- [ ] **Step 4: Verify PASS; run existing `tests/unit/test_init.py` fully (template assertions there will need updating to the new text).** - -### Task 7: Docs + changelog + full verify - -**Files:** -- Modify: `docs/configuration.md` (blocking field entry; provider-aware init note) -- Modify: `CHANGELOG.md` (`## [Unreleased]`) - -- [ ] **Step 1: Write docs entry** for `evaluators.*.blocking` (semantics, default, scaffold defaults) and `capture sync --keep-duplicates`. -- [ ] **Step 2: CHANGELOG entries** (Added: blocking flag, provider prompt, dedup; Changed: judge/semantic failures excluded not neutral; Fixed: reasoning-model judges, scaffold criterion). -- [ ] **Step 3: Full verify:** `ruff check . && ruff format --check . && mypy --strict src/evalshift && pytest -m "not integration"` — all green. - -## Self-Review - -- Spec coverage: §1→T1+T2, §2→T4, §3→T3+T6(criterion), §4→T1(scaffold flag)+T4, §5→T6, §6→T5, testing→each task, docs/compat→T7. No gaps. -- Placeholders: none (each step has code or exact assertions). -- Type consistency: `blocking: bool` end-to-end; `_wilson_interval` returns `tuple[float, float]` used only in T4. diff --git a/docs/superpowers/plans/2026-08-23-temperature-value-rejection.md b/docs/superpowers/plans/2026-08-23-temperature-value-rejection.md deleted file mode 100644 index 526b00b..0000000 --- a/docs/superpowers/plans/2026-08-23-temperature-value-rejection.md +++ /dev/null @@ -1,735 +0,0 @@ -# Temperature Value Rejection — Runtime Adaptation Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. - -**Goal:** When a provider 400s the *value* of `temperature` (reasoning-tier models accept only their default), the client drops the parameter, redispatches, memoizes per model, and the run reports the model in the existing `non_deterministic_models` banner. - -**Architecture:** One detection predicate + adaptation inside `ModelClient._dispatch_with_retry` (the single choke point under all four `complete*` methods). Two merge points carry the client's rejected-model set into `state.non_deterministic_models`: the orchestrator's final state write (run phase) and the evaluate command's state write (judge phase). No cache, config, or registry changes. - -**Tech Stack:** Python 3.11+, litellm, pydantic, pytest (+pytest-asyncio), `mypy --strict`, ruff. - -**Spec:** `docs/superpowers/specs/2026-08-23-temperature-value-rejection-design.md` — read it first. - -## Global Constraints - -- `mypy --strict` clean; ruff clean (`make ci` is the gate, and the pre-push hook). -- Tests first — watch each fail before implementing (TDD). -- Conventional Commits. -- Detection predicate must require ALL of: exception type name `BadRequestError`, message contains `temperature` (case-insensitive), `"temperature" in kwargs` at the call site. Anything less is not intercepted. -- Adaptation must NOT consume a retry attempt. `AuthError` short-circuit, backoff, and exhaustion behavior unchanged. -- Cache keys unchanged. -- Exception matching is by type NAME (string), matching the existing `_map_exception` idiom at `src/evalshift/models/client.py:678` — do not import litellm exception classes. - ---- - -### Task 1: Client-level detection, adaptation, memoization - -**Files:** -- Modify: `src/evalshift/models/client.py` (constructor ~:229, kwargs comment blocks ~:331 and ~:455, `_dispatch_with_retry` ~:470-529, helpers section) -- Modify: `src/evalshift/models/capabilities.py` (module docstring lines 20-23) -- Test: `tests/unit/test_model_client.py` - -**Interfaces:** -- Consumes: existing `_patch_acompletion(monkeypatch, handler)` harness and `_FakeResponse` in `tests/unit/test_model_client.py`; existing `RetryPolicy`, `_map_exception`. -- Produces: `ModelClient.temperature_rejected_models` property → `frozenset[str]` of canonical model ids (Tasks 2 and 3 read it). Module-level helper `_is_temperature_value_rejection(exc: BaseException) -> bool`. - -- [x] **Step 1: Write the failing tests** - -Append to `tests/unit/test_model_client.py` (reuse the file's existing fakes/harness; `RateLimitError` import already exists at the top): - -```python -# --------------------------------------------------------------------------- -# Temperature value rejection (reasoning-tier models) -# --------------------------------------------------------------------------- - -# Name-based to match _map_exception's idiom: production raises -# litellm.BadRequestError; tests only need the type NAME to match. -_BadRequestError = type("BadRequestError", (Exception,), {}) - -_TEMP_400_MSG = ( - "OpenAIException - Unsupported value: 'temperature' does not support 0.0 " - "with this model. Only the default (1) value is supported." -) - - -class TestTemperatureValueRejection: - @pytest.mark.asyncio - async def test_adapts_resends_without_temperature_and_records_model( - self, - monkeypatch: pytest.MonkeyPatch, - caplog: pytest.LogCaptureFixture, - ) -> None: - calls: list[dict[str, Any]] = [] - - def handler(**kwargs: Any) -> Any: - calls.append(dict(kwargs)) - if "temperature" in kwargs: - raise _BadRequestError(_TEMP_400_MSG) - return _FakeResponse("adapted ok") - - _patch_acompletion(monkeypatch, handler) - client = ModelClient(retry_policy=RetryPolicy(max_attempts=2)) - with caplog.at_level("WARNING"): - result = await client.complete(model="gpt-4o", prompt="hi") - - assert result.text == "adapted ok" - assert len(calls) == 2 - assert "temperature" in calls[0] - assert "temperature" not in calls[1] - assert client.temperature_rejected_models == frozenset({"openai/gpt-4o"}) - warnings = [r for r in caplog.records if "temperature" in r.getMessage()] - assert len(warnings) == 1 - - @pytest.mark.asyncio - async def test_adaptation_does_not_consume_a_retry_attempt( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - # max_attempts=2. Sequence: temp-400 (adaptation), transient timeout - # (attempt 1), success (attempt 2). Only passes if the adaptation - # left the full retry budget intact. - calls: list[dict[str, Any]] = [] - timeout_error = type("Timeout", (Exception,), {}) - - def handler(**kwargs: Any) -> Any: - calls.append(dict(kwargs)) - if "temperature" in kwargs: - raise _BadRequestError(_TEMP_400_MSG) - if len(calls) == 2: - raise timeout_error("transient") - return _FakeResponse("ok") - - _patch_acompletion(monkeypatch, handler) - client = ModelClient(retry_policy=RetryPolicy(max_attempts=2)) - result = await client.complete(model="gpt-4o", prompt="hi") - assert result.text == "ok" - assert len(calls) == 3 - - @pytest.mark.asyncio - async def test_later_calls_omit_temperature_preemptively( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - calls: list[dict[str, Any]] = [] - - def handler(**kwargs: Any) -> Any: - calls.append(dict(kwargs)) - if "temperature" in kwargs: - raise _BadRequestError(_TEMP_400_MSG) - return _FakeResponse("ok") - - _patch_acompletion(monkeypatch, handler) - client = ModelClient() - await client.complete(model="gpt-4o", prompt="first") - await client.complete(model="gpt-4o", prompt="second") - # first call: 2 dispatches (reject + adapted); second call: exactly 1. - assert len(calls) == 3 - assert "temperature" not in calls[2] - - @pytest.mark.asyncio - async def test_non_temperature_400_keeps_existing_retry_then_raise( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - def handler(**kwargs: Any) -> Any: - raise _BadRequestError("Unsupported value: 'tool_choice'") - - _patch_acompletion(monkeypatch, handler) - client = ModelClient(retry_policy=RetryPolicy(max_attempts=2)) - with pytest.raises(ModelError): - await client.complete(model="gpt-4o", prompt="hi") - assert client.temperature_rejected_models == frozenset() - - @pytest.mark.asyncio - async def test_temperature_400_without_temperature_in_kwargs_raises( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - # A model that keeps 400ing about temperature even after the parameter - # is gone must surface as ModelError, not loop. After the one - # adaptation, kwargs no longer carry temperature, so the predicate's - # kwargs leg fails and the normal path takes over. - def handler(**kwargs: Any) -> Any: - raise _BadRequestError(_TEMP_400_MSG) - - _patch_acompletion(monkeypatch, handler) - client = ModelClient(retry_policy=RetryPolicy(max_attempts=2)) - with pytest.raises(ModelError): - await client.complete(model="gpt-4o", prompt="hi") - # The adaptation itself still fired once and recorded the model. - assert client.temperature_rejected_models == frozenset({"openai/gpt-4o"}) -``` - -Note: check how the file's existing async tests are decorated (`@pytest.mark.asyncio` vs asyncio_mode config) and match; adjust `_FakeResponse`/import names only if they differ from the harness shown at the top of the file. - -- [x] **Step 2: Run tests to verify they fail** - -```bash -pytest tests/unit/test_model_client.py::TestTemperatureValueRejection -v -``` - -Expected: all 5 FAIL — `AttributeError: ... no attribute 'temperature_rejected_models'` (and/or `ModelError` from the unadapted 400). - -- [x] **Step 3: Implement in `src/evalshift/models/client.py`** - -Constructor (~:229) — add the set: - -```python - def __init__(self, *, retry_policy: RetryPolicy | None = None) -> None: - self._retry = retry_policy or RetryPolicy() - # Canonical ids of models that 400ed the VALUE of ``temperature`` - # (reasoning-tier models accept only their default). Once listed, a - # model's calls omit the parameter entirely -- one failed call per - # model per process. See _is_temperature_value_rejection. - self._temperature_rejected: set[str] = set() -``` - -Property, right after `__init__`: - -```python - @property - def temperature_rejected_models(self) -> frozenset[str]: - """Canonical ids that rejected every non-default ``temperature`` value. - - Populated at dispatch time, from the provider's own 400. The - orchestrator and the evaluate command merge this into the run - state's ``non_deterministic_models`` so the report's banner covers - runtime discoveries as well as the run-start capability probe. - """ - return frozenset(self._temperature_rejected) -``` - -Module-level helper, in the Helpers section next to `_map_exception` (~:678): - -```python -def _is_temperature_value_rejection(exc: BaseException) -> bool: - """Report whether ``exc`` is a provider 400 rejecting ``temperature``'s value. - - Matched by type NAME like :func:`_map_exception`, so litellm's - ``BadRequestError`` is caught without importing its class. The caller - must additionally check that the outgoing kwargs actually carried - ``temperature`` -- a temperature-flavoured 400 on a call that never sent - the parameter is somebody else's bug and must surface. - """ - return type(exc).__name__ == "BadRequestError" and "temperature" in str(exc).lower() -``` - -Rewrite `_dispatch_with_retry` (~:470-529). The `for` loop becomes a `while` so the adaptation can rewind the attempt counter; everything else keeps its exact semantics: - -```python - async def _dispatch_with_retry( - self, - canonical: str, - kwargs: dict[str, Any], - *, - log_suffix: str, - ) -> tuple[Any, int]: - # (keep the existing docstring; append:) - # - # One adaptation is layered on top of the retry policy: a provider - # 400 that rejects the VALUE of ``temperature`` (reasoning-tier - # models accept only their default) pops the parameter and - # redispatches immediately, without consuming a retry attempt. The - # model id is memoized on the client so later calls omit the - # parameter before dispatch. - if canonical in self._temperature_rejected: - kwargs.pop("temperature", None) - last_exc: Exception | None = None - attempt = 0 - while True: - attempt += 1 - start = time.perf_counter() - try: - response = await litellm.acompletion(**kwargs) - except Exception as exc: - if "temperature" in kwargs and _is_temperature_value_rejection(exc): - kwargs.pop("temperature") - if canonical not in self._temperature_rejected: - self._temperature_rejected.add(canonical) - log.warning( - "model %s rejects non-default temperature values; " - "resending without temperature — sampling for this " - "model is not controlled and outputs are " - "non-deterministic", - canonical, - ) - attempt -= 1 # adaptation, not a retry - continue - mapped = _map_exception(exc) - # Auth errors are deterministic; don't waste retries on them. - if isinstance(mapped, AuthError): - raise mapped from exc - last_exc = mapped - if attempt >= self._retry.max_attempts: - raise mapped from exc - delay = self._retry.delay(attempt) - log.warning( - "model %s%s attempt %d failed (%s); retrying in %.2fs", - canonical, - log_suffix, - attempt, - mapped.__class__.__name__, - delay, - ) - await asyncio.sleep(delay) - continue - latency_ms = int((time.perf_counter() - start) * 1000) - return response, latency_ms -``` - -Delete the now-unreachable trailing `raise ModelError(f"exhausted retries...")` and the `# Loop exits cleanly...` comment (the `while True` only exits via `return`/`raise`; keep `last_exc` only if something still reads it — if nothing does, remove it too). - -Termination argument (for the reviewer): the adaptation branch requires `"temperature" in kwargs` and unconditionally pops it, so it fires at most once per call — no infinite loop. - -- [x] **Step 4: Fix the two false comments** - -`client.py` kwargs blocks (~:331 and ~:455) — replace the four-line comment above `"drop_params": True` in BOTH places with: - -```python - # drop_params only saves models LiteLLM's configs special-case - # (o-series names). Other reasoning-tier models reject - # temperature != 1 at the API with a 400; that case is handled - # at dispatch — see _is_temperature_value_rejection and the - # adaptation in _dispatch_with_retry. - "drop_params": True, -``` - -`capabilities.py` module docstring (lines 20-23) — replace the paragraph -"Note this detects *withdrawal*, not value constraints. ... already handled by ``drop_params`` in :mod:`evalshift.models.client`." with: - -``` -Note this detects *withdrawal*, not value constraints. Reasoning-tier models -such as ``gpt-5.6-terra`` advertise ``temperature`` while rejecting every -value except their default; ``drop_params`` does not cover them (LiteLLM -special-cases only o-series names). That case is detected from the -provider's own 400 at dispatch time and adapted per model — see -``ModelClient._dispatch_with_retry`` in :mod:`evalshift.models.client`. -``` - -- [x] **Step 5: Run the tests** - -```bash -pytest tests/unit/test_model_client.py -v -``` - -Expected: new class 5/5 PASS, every pre-existing test in the file still PASS (retry, auth short-circuit, mapping tests prove the unchanged semantics). - -- [x] **Step 6: Static checks** - -```bash -ruff check src/evalshift/models/ tests/unit/test_model_client.py && mypy --strict src/evalshift -``` - -Expected: clean. - -- [x] **Step 7: Commit** - -```bash -git add src/evalshift/models/client.py src/evalshift/models/capabilities.py tests/unit/test_model_client.py -git commit -m "fix(client): adapt to models that reject temperature values - -Reasoning-tier models (gpt-5.6-terra and kin) advertise temperature but -400 every value except the default; drop_params only saves o-series -names. Detect the provider's 400 at dispatch, resend without the -parameter, and memoize per model so later calls skip it preemptively. -The adaptation does not consume a retry attempt. Also corrects the two -comments that claimed drop_params already covered this." -``` - ---- - -### Task 2: Orchestrator merges runtime rejections into run state - -**Files:** -- Modify: `src/evalshift/runner/orchestrator.py` (`_process_work` final-state block, ~:814-818) -- Test: `tests/unit/test_orchestrator.py` - -**Interfaces:** -- Consumes: `ModelClient.temperature_rejected_models` (Task 1); `_process_work`'s existing `client: ModelClient` parameter and `state` object; `touch_checkpoint`; `RunState.non_deterministic_models` (`src/evalshift/runner/models.py:159`). -- Produces: final `state.json` whose `non_deterministic_models` is the probe result plus runtime discoveries, deduplicated, probe-order first then sorted runtime additions. - -- [x] **Step 1: Write the failing tests** - -Append to `tests/unit/test_orchestrator.py`, using the file's existing fixtures (`cache`, `_config`, `_suite`, `_writeable_paths`, `_make_fake_client`) and its existing state-reading idiom (the file already imports the checkpoint module — reuse its import name for `read_state`): - -```python -class TestRuntimeTemperatureRejectionReporting: - @pytest.mark.asyncio - async def test_rejected_models_land_in_final_state( - self, - monkeypatch: pytest.MonkeyPatch, - tmp_path: Path, - cache: CacheStore, - ) -> None: - _make_fake_client(monkeypatch) - config_path, suite_path, runs_base = _writeable_paths(tmp_path) - client = ModelClient() - # White-box seed: simulates a mid-run provider rejection without - # needing a live 400 through the faked complete(). - client._temperature_rejected.add("openai/gpt-5.6-terra") - result = await run_orchestrator( - config=_config(), - suite=_suite(), - config_path=config_path, - suite_path=suite_path, - runs_base=runs_base, - source="gpt-4o", - target="gpt-4o-mini", - yes=True, - client=client, - cache=cache, - ) - state = read_state(result.run_dir) - assert "openai/gpt-5.6-terra" in state.non_deterministic_models - - @pytest.mark.asyncio - async def test_merge_deduplicates_against_probe_result( - self, - monkeypatch: pytest.MonkeyPatch, - tmp_path: Path, - cache: CacheStore, - ) -> None: - _make_fake_client(monkeypatch) - monkeypatch.setattr( - orchestrator_module, - "detect_non_deterministic_models", - lambda *, source, target: ["openai/gpt-5.6-terra"], - ) - config_path, suite_path, runs_base = _writeable_paths(tmp_path) - client = ModelClient() - client._temperature_rejected.add("openai/gpt-5.6-terra") - result = await run_orchestrator( - config=_config(), - suite=_suite(), - config_path=config_path, - suite_path=suite_path, - runs_base=runs_base, - source="gpt-4o", - target="gpt-4o-mini", - yes=True, - client=client, - cache=cache, - ) - state = read_state(result.run_dir) - assert state.non_deterministic_models.count("openai/gpt-5.6-terra") == 1 -``` - -Match the surrounding tests for the exact `run_orchestrator(...)` keyword set — copy the call from `test_single_run_writes_state_and_raw` (~:156) and change only what these tests need. Import `ModelClient` and the orchestrator module alias if the file lacks them. - -- [x] **Step 2: Run tests to verify they fail** - -```bash -pytest tests/unit/test_orchestrator.py -k RuntimeTemperatureRejection -v -``` - -Expected: FAIL — `"openai/gpt-5.6-terra" not in []` (first test); dedup test fails the same way until the merge exists. - -- [x] **Step 3: Implement the merge** - -In `_process_work` (`src/evalshift/runner/orchestrator.py` ~:814-818), replace: - -```python - # Final checkpoint + status flip. - final_state = touch_checkpoint(state, completed).model_copy( - update={"status": "completed"}, - ) -``` - -with: - -```python - # Final checkpoint + status flip. Runtime-discovered temperature - # rejections join the probe-detected list here so the report's - # non-determinism banner covers both. Probe entries keep their order; - # runtime additions follow, sorted, deduplicated. - runtime_nondet = sorted( - set(client.temperature_rejected_models) - set(state.non_deterministic_models) - ) - final_state = touch_checkpoint(state, completed).model_copy( - update={ - "status": "completed", - "non_deterministic_models": [ - *state.non_deterministic_models, - *runtime_nondet, - ], - }, - ) -``` - -(Resume caveat, acceptable and documented by this comment: a resumed run re-discovers rejections on its live calls; a fully-cached resume makes no live calls and adds nothing new — the prior final state already carried them.) - -- [x] **Step 4: Run tests** - -```bash -pytest tests/unit/test_orchestrator.py -v -``` - -Expected: both new tests PASS, all existing orchestrator tests PASS. - -- [x] **Step 5: Commit** - -```bash -git add src/evalshift/runner/orchestrator.py tests/unit/test_orchestrator.py -git commit -m "feat(runner): report runtime temperature rejections in run state - -Models the client discovered mid-run join non_deterministic_models at -the final state write, deduplicated against the run-start probe, so the -report banner and JSON cover both detection paths." -``` - ---- - -### Task 3: Evaluate command shares one judge client and merges its discoveries - -**Files:** -- Modify: `src/evalshift/cli/commands/evaluate.py` (caller ~:167, `_build_evaluators` ~:283-334, `write_state` ~:207) -- Test: `tests/unit/test_evaluate_command.py` - -**Interfaces:** -- Consumes: `ModelClient.temperature_rejected_models` (Task 1); `PairwiseJudgeEvaluator(..., client=...)` (existing parameter, `src/evalshift/evaluators/llm_judge.py:65`). -- Produces: `_build_evaluators(cfg, project_root, judge_client: ModelClient)` — new required keyword; the state written after scoring carries judge-phase rejections. - -- [x] **Step 1: Write the failing tests** - -Add to `tests/unit/test_evaluate_command.py` (follow the file's existing config-building idiom for a cfg with one `llm_judge` entry; adapt fixture names to what the file already uses): - -```python -def test_build_evaluators_threads_shared_judge_client(tmp_path: Path) -> None: - cfg = _cfg_with_judge() # file-local helper; build EvalShiftConfig with one llm_judge entry - judge_client = ModelClient() - evaluators = _build_evaluators(cfg, tmp_path, judge_client=judge_client) - judges = [e for e in evaluators if isinstance(e, PairwiseJudgeEvaluator)] - assert judges, "config should have produced a judge" - assert all(j._client is judge_client for j in judges) -``` - -And the merge behavior at the state write — if the file has a command-level harness that runs scoring end-to-end, add an assertion there that a pre-seeded `judge_client._temperature_rejected` entry appears in the re-read state's `non_deterministic_models`; if no such harness exists, test the merge expression through the command function with mocked `_score_everything` following the file's established mocking pattern. The assertion that matters: - -```python - state_after = read_state(run_dir) - assert "openai/gpt-5.6-terra" in state_after.non_deterministic_models - assert state_after.evaluator_coverage == coverage # both updates land in ONE write -``` - -- [x] **Step 2: Run tests to verify they fail** - -```bash -pytest tests/unit/test_evaluate_command.py -k "judge_client or temperature" -v -``` - -Expected: FAIL — `_build_evaluators() got an unexpected keyword argument 'judge_client'`. - -- [x] **Step 3: Implement** - -`src/evalshift/cli/commands/evaluate.py`: - -Add import (top of file, with the other model imports): - -```python -from evalshift.models.client import ModelClient -``` - -Caller (~:167): - -```python - # One client shared by every judge so temperature rejections discovered - # while judging are collected in one place and merged into state below. - judge_client = ModelClient() - evaluators = _build_evaluators(cfg, project_root, judge_client=judge_client) -``` - -`_build_evaluators` signature (~:283): - -```python -def _build_evaluators( - cfg: EvalShiftConfig, project_root: Path, *, judge_client: ModelClient -) -> list[Evaluator]: -``` - -Judge construction (~:326-333): - -```python - for j in cfg.evaluators.llm_judge: - _add( - PairwiseJudgeEvaluator( - criterion_name=j.criterion_name, - criterion_prompt=j.criterion_prompt, - judge_model=j.judge_model, - client=judge_client, - ), - blocking=j.blocking, - ) -``` - -State write (~:207) — replace: - -```python - write_state(run_dir, state.model_copy(update={"evaluator_coverage": coverage})) -``` - -with: - -```python - # Judge calls can discover temperature-rejecting models after the run - # phase already wrote its state; merge them here so the report banner - # covers the judge model too. - runtime_nondet = sorted( - set(judge_client.temperature_rejected_models) - - set(state.non_deterministic_models) - ) - write_state( - run_dir, - state.model_copy( - update={ - "evaluator_coverage": coverage, - "non_deterministic_models": [ - *state.non_deterministic_models, - *runtime_nondet, - ], - }, - ), - ) -``` - -Fix every other `_build_evaluators(` call site the same way (grep for it; `evalshift all` may reach it through this module only — verify). - -- [x] **Step 4: Run tests** - -```bash -pytest tests/unit/test_evaluate_command.py tests/unit/test_all_command.py -v -``` - -Expected: new tests PASS; existing evaluate/all command tests PASS (they exercise `_build_evaluators` and will catch a missed call site). - -- [x] **Step 5: Commit** - -```bash -git add src/evalshift/cli/commands/evaluate.py tests/unit/test_evaluate_command.py -git commit -m "feat(evaluate): share one judge client and report its rejections - -All PairwiseJudgeEvaluators now dispatch through a single ModelClient; -temperature rejections it discovers merge into non_deterministic_models -alongside evaluator_coverage in the existing post-scoring state write." -``` - ---- - -### Task 4: Judge integration test — verdict instead of EvaluatorError - -**Files:** -- Test: `tests/unit/test_evaluators.py` (judge section, near the `PairwiseJudgeEvaluator` tests ~:616) - -**Interfaces:** -- Consumes: Task 1's adaptation through the real `ModelClient`; the file's existing judge test idiom (`criterion_name`/`criterion_prompt`, `score(...)` signature at ~:625-631) and `rng` seam. - -- [x] **Step 1: Write the test (should pass immediately — it locks the end-to-end behavior)** - -```python - @pytest.mark.asyncio - async def test_judge_survives_temperature_value_rejection( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - # End-to-end through the real ModelClient: first dispatch 400s the - # temperature value, the client adapts, the judge gets its verdict. - # Before the adaptation existed this raised EvaluatorError and a - # blocking judge poisoned the gate. - bad_request = type("BadRequestError", (Exception,), {}) - calls: list[dict[str, Any]] = [] - - async def fake_acompletion(**kwargs: Any) -> Any: - calls.append(dict(kwargs)) - if "temperature" in kwargs: - raise bad_request( - "'temperature' does not support 0.0 with this model." - ) - return _fake_llm_response('{"winner": "A"}') # reuse/adapt the file's response fake - - monkeypatch.setattr(client_module.litellm, "acompletion", fake_acompletion) - monkeypatch.setattr( - client_module.litellm, "completion_cost", lambda **_: 0.0 - ) - rng = random.Random(0) - rng.random = lambda: 0.0 # type: ignore[method-assign] # target shown as A - ev = PairwiseJudgeEvaluator( - criterion_name="equivalence", - criterion_prompt="Which is better?", - judge_model="gpt-4o", - client=ModelClient(), - rng=rng, - ) - score = await ev.score( - prompt_id="p", - example_id="e", - input_vars={}, - source_output="src", - target_output="tgt", - ) - assert score.target_score == 1.0 - assert len(calls) == 2 and "temperature" not in calls[1] -``` - -Reuse the file's existing fake-response helper for the judge (`_client_returning` internals show the shape); import `client as client_module` from `evalshift.models` and `random` if not present. `completion_cost` signature: match how `tests/unit/test_model_client.py` patches it (`lambda completion_response=None, **_: 0.0`). - -- [x] **Step 2: Run it** - -```bash -pytest tests/unit/test_evaluators.py -k temperature -v -``` - -Expected: PASS. If it fails, Task 1 has a bug — fix there, not here. - -- [x] **Step 3: Commit** - -```bash -git add tests/unit/test_evaluators.py -git commit -m "test(judge): lock verdict delivery through temperature adaptation" -``` - ---- - -### Task 5: Documentation - -**Files:** -- Modify: `DOCS.md:587` (Sampling control paragraph) -- Modify: `CHANGELOG.md` (Unreleased) -- Modify: `llms-full.txt:467` area (determinism text) - -**Interfaces:** none — prose only. Check whether `llms-full.txt` is generated (`grep -rn "llms-full" Makefile scripts/`); if generated, update the source and regenerate instead of editing the copy. - -- [x] **Step 1: DOCS.md** — extend the Sampling-control paragraph (`DOCS.md:587`). After the sentence ending "…the report shows a banner above the verdict plus a methodology note.", insert: - -```markdown -A second failure mode is caught at call time rather than run start: reasoning-tier models (for example `gpt-5.6-terra`) advertise `temperature` but reject every value except their default with a 400. The first such rejection makes EvalShift resend the call without `temperature` and stop sending it to that model for the rest of the process; the model joins `non_deterministic_models` and the same banner. One call per affected model fails and is retried adapted — nothing is lost, but sampling for that model is provider-default, not controlled. -``` - -- [x] **Step 2: CHANGELOG.md** — under `## [Unreleased]`, add: - -```markdown -### Fixed - -- Models that reject non-default `temperature` values (reasoning-tier models - such as `gpt-5.6-terra`) no longer fail every call. The client detects the - provider's 400, resends without `temperature`, memoizes the model for the - rest of the process, and reports it in `non_deterministic_models` alongside - probe-detected models. Previously a `blocking: true` judge on such a model - errored on every example after burning the full retry budget per call. -``` - -- [x] **Step 3: llms-full.txt** — update the determinism passage (~:467) with the same two-mode story (withdrawal probed at run start; value rejection adapted at call time), matching the file's prevailing terseness. If generated, regenerate. - -- [x] **Step 4: Commit** - -```bash -git add DOCS.md CHANGELOG.md llms-full.txt -git commit -m "docs: document temperature value-rejection adaptation" -``` - ---- - -### Task 6: Full verification gate - -- [x] **Step 1: Run the CI mirror** - -```bash -make ci -``` - -Expected: ruff, `mypy --strict`, and the full pytest suite all green. Fix anything red before claiming done (superpowers:verification-before-completion). - -- [x] **Step 2: Reproduce the original failure shape once, manually** — optional sanity: `evalshift test-call` (or the judge path) against a faked reject is already covered by Task 4; live verification happens in the personalButler CI once a release ships. diff --git a/docs/superpowers/plans/2026-09-08-external-review-response.md b/docs/superpowers/plans/2026-09-08-external-review-response.md deleted file mode 100644 index d561f6c..0000000 --- a/docs/superpowers/plans/2026-09-08-external-review-response.md +++ /dev/null @@ -1,488 +0,0 @@ -# External Review Response — Findings and Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. Do one task at a time; each task is independently shippable. - -**Goal:** Act on an external review of EvalShift (2026-09-08) that raised nine weaknesses. Each point was verified against the source of `evalshift-sdk` and `evalshift-cli`. This document records the verdict per point, then lays out the work in phases ordered by *how valid the point is* and *how much it matters*, cheapest-and-most-misleading fixes first. - -**Scope:** Two repositories. Tasks are tagged `[cli]` or `[sdk]`. Cross-repo tasks list both. - -**Tech Stack:** Python 3.10+ (sdk, stdlib-only runtime) / 3.11+ (cli), pydantic, litellm, pytest, `mypy --strict`, ruff. - ---- - -## Part 1 — Findings - -Verdicts, from most to least valid. Evidence lines are current as of 2026-09-08. - -| # | Review point | Verdict | Importance | Phase | -|---|---|---|---|---| -| 4 | Capture and replay disagree on scope | **Valid, understated** | High — a documented workflow is broken | 0, 2 | -| 3 | Tool calls recorded only from executed functions | **Valid** | High — argument-drift scoring uses the wrong ground truth | 3 | -| 8 | LiteLLM hides provider edges (tool_choice, strict, structured output) | **Valid** | High — replays silently drop constraints | 4 | -| 1 | `evalshift` namespace collision | **Valid** (author already deferred it as D1-followup) | Medium-high — biggest setup friction, but breaking to fix | 6 | -| 2 | No provider client wrappers | **Valid on substance**; wrong that tokens gate promotion, wrong that LlamaIndex is on a roadmap | Medium-high — adoption | 5 | -| 6 | One sample per example, no power warning | **Half valid**: single-sample design is real; "no underpowered warning" is false | Medium | 7 | -| 7 | Judge self-preference unaddressed | **Half valid**: no code mitigation; warning already in scaffold + docs | Medium-low | 7 | -| 5 | Drift not correctness; 0.95 floor false alarms | **Half valid**: drift framing right; default init writes 0.75 and makes semantic/judge advisory | Low — already mitigated | 1 | -| 9 | Config carries two mental models | **Mostly invalid**: `prompts` is required and on the hot path; `captures` is not a config block. Real residue: one docs default mismatch, no `suites` example | Low | 0, 1 | - -### Evidence summary per point - -**#4 Capture vs replay.** -- SDK `docs/DECISIONS.md:31-34` says tool results MUST be stored as fixtures keyed by `call_id` + input hash for CLI replay; `trace/serialize.py` `build_fixture_table` implements it. No CLI code reads it (`grep fixture_table evalshift-cli/src` is empty). -- CLI runner makes one model call per example, no loop: `runner/orchestrator.py:95` (`WorkItem`: "a single LLM call"), `docs/agents.md:167-170`. -- Promotion defaults to round one: `captures/promote.py:91` `rounds: Literal["first","all"] = "first"`; `suite/models.py:196-199` says later rounds are "retained for teacher-forced multi-round replay" which does not exist. -- **Broken doc:** `evalshift-sdk/examples/support_agent/README.md:25` runs `evalshift run --offline --fixtures fixtures.jsonl`. Neither flag exists (`cli/commands/run.py`, `all.py`). -- README `README.md:253-256` claims detection of "*how* it sequences" tools; in the capture→run path this is in-order matching within one response only. - -**#3 Executed-only tool calls.** -- Only recorder is `@capture.tool` → `capture/api.py:298-332` `_run_tool`, arguments bound from the Python signature (`_bind`, `:158-165`). -- `trace/models.py:24-45` `ModelCallEvent` has `input`/`output`/`tools_offered` but no requested-calls field. `grep 'tool_calls|AIMessage|function_call' evalshift-sdk/src` is empty. -- LangChain adapter records `on_tool_start`/`on_tool_end` (`adapters/langchain.py:493-528`), never `message.tool_calls` in `on_llm_end` (`:465-478`). -- CLI compensates with `promote.py:378-419` `_unwrap_recorded_arguments` ("No model can produce the recorded shape", `docs/agents.md:190-206`). - -**#8 LiteLLM.** -- Single chokepoint `models/client.py:665` `litellm.acompletion`, with `drop_params: True` (`:483`, `:599`) — unsupported params vanish silently. -- No `tool_choice`, `parallel_tool_calls`, or `strict` anywhere in `evalshift-cli/src` (only unrelated `optional_fields_scored: "strict"`). -- SDK allow-list `capture/generation.py:28-36` `GENERATION_KEYS` records temperature, top_p, response_*, max_tokens — never tool_choice. CLI translator `runner/generation.py:18-20` `_HANDLED_KEYS` consumes four keys, debug-logs the rest. - -**#1 Namespace.** -- `evalshift-sdk/pyproject.toml` `packages = ["src/evalshift"]`; `evalshift-cli/pyproject.toml` same. Both have a regular `evalshift/__init__.py`. -- `evalshift-sdk/docs/DECISIONS.md` D-pkg: "Co-installing ... clashes ... tracked as D1-followup (unify later: CLI depends on SDK, or a `[cli]` extra)." -- Two-venv instruction repeated in `README.md:50-53`, `:85`, `docs/getting-started.md:25-27`, `evalshift-sdk/README.md:41-42`. - -**#2 Provider wrappers.** -- `evalshift-sdk/src/evalshift/adapters/` contains only `langchain.py`; one optional extra in `pyproject.toml:33-36`. -- Stdlib-only rule: `docs/DECISIONS.md` D-deps; `capture/toolset.py:19-20`. -- `record_model_call` (`capture/api.py:225-235`): `input_tokens`/`output_tokens`/`cost_usd` default 0, `latency_ms` None. No pricing table in SDK; LangChain adapter never sets `cost_usd` (`langchain.py:474-477`). -- Corrections: missing tokens never block promotion; missing `toolset_ref` does (`promote.py:206-215`, `blocked_reason="no_toolset"`). LlamaIndex appears once in the monorepo, in `README.md:331` **Non-goals**. `DECISIONS.md:4` references `IMPLEMENTATION_PLAN.md`, which does not exist. - -**#6 Sampling.** -- No repeat option (`grep repeat|n_runs|trials` over config/runner/suite empty). `orchestrator.py:588-600` emits one source + one target item per pair. `config/models.py:603` `cache: bool = True`. `docs/methodology.md:469-472` states it. -- Dedup is by content key, deliberately **not** input_hash: `promote.py:990-1005`. -- Power warnings exist: `analysis/statistics.py:56-57` `MIN_N_FOR_TEST=5`, `MIN_N_RELIABLE=20`; `reports/html.py:82` `_SMALL_SAMPLE_THRESHOLD=10`; `policy.py:808-812` `inconclusive` when Wilson interval spans budget. - -**#7 Judge.** -- `config/models.py:28` `DEFAULT_JUDGE_MODEL = "gemini-3.1-flash-lite-preview"`. `init.py:47-57` scaffolds a same-provider judge on purpose (one API key). -- A/B randomisation: `llm_judge.py:132-136`. Family-bias warning already present: `init.py:140-141`, `docs/configuration.md:502`. No runtime check. -- Stale docstring `config/models.py:186` says "Defaults to a strong Anthropic model". - -**#5 Drift vs correctness.** -- `evaluators/semantic.py:146` fails below `min_similarity` (default 0.9, `config/models.py:174`); target compared to source only. No text evaluator reads `SuiteExample.expected` (`suite/models.py:189-191`, "unused by most evaluators"). -- Default init profile (`_scaffold.py:94-105`) writes `min_equivalence_rate: 0.75`; `init.py:72-81, 142-150` sets semantic and llm_judge `blocking: false`. The 0.95 floor is only `provider-switch` and the repo's stale root `evalshift.yaml:53-60`. -- Library default `SemanticEvaluatorConfig.blocking = True` (`config/models.py:177`) — hand-written configs that omit the key are blocking. - -**#9 Config.** -- `config/models.py:626-640`: `prompts` required (`min_length=1`), `suites` optional dict. No `captures` block. `python_string` is a literal-only AST parser (`parsers/python_string.py:134-198`), never executes. -- `prompts` on hot path: `orchestrator.py:361-366`, `validate.py:69`, `doctor.py:149`. Init's `replay` passthrough prompt is what makes promoted captures replayable. -- **Real mismatch:** `docs/configuration.md:84-89` says init writes `max_cost_increase: 0.50`, `max_latency_increase: 2.0`; init and schema write 0.30/0.30. `llms-full.txt:909-911` shows a third, obsolete block (0.03/0/0.95/0.01/0.03). -- `examples/simple`, `examples/agent` use `python_string`; `examples/agent-traces` uses `manual`; none use `suites`. -- `docs/configuration.md:302` says "Three sub-keys" under evaluators; seven are documented. - ---- - -## Part 2 — Plan - -### Ordering rationale - -Phases are ordered by (validity × user harm) ÷ cost. Phase 0 is all documentation that is currently *wrong* and cheap to fix. Phases 1–4 restore honesty and correctness in what exists. Phases 5–6 add capability or restructure packaging. Phase 7 holds the half-valid statistical points, which are already partly mitigated. - -### Global constraints - -- `mypy --strict` and ruff clean in both repos (`make ci`). -- TDD for every logic change: failing test first. -- Conventional Commits; one commit per task unless noted. -- SDK runtime stays stdlib-only (D-deps). Any provider integration is an import-guarded optional extra, like `adapters/langchain.py`. -- SDK capture schema changes bump `SCHEMA_VERSION` (`trace/schema.py:19`, currently `2.0.0`) and register a migration via `trace/migrate.py` `register_migration`. CLI must accept both old and new envelopes. -- Do not make previously-valid `evalshift.yaml` files fail validation. - ---- - -## Phase 0 — Fix documentation that is wrong today - -*Validity: full. Importance: high (users follow these). Cost: minutes each.* - -### Task 0.1 `[sdk]` Remove the non-existent `--offline --fixtures` workflow - -**Files:** `evalshift-sdk/examples/support_agent/README.md`, `evalshift-sdk/examples/support_agent/` (check for `fixtures.jsonl` and any script that references it). - -- [x] **Step 1:** Read the example README end to end and list every command it tells the user to run. -- [x] **Step 2:** Run each command against the current CLI in a scratch venv; record which fail (`--offline`, `--fixtures` are known to). -- [x] **Step 3:** Rewrite the walkthrough to use commands that exist: `evalshift capture sync`, `evalshift run` with real keys, or the mocked integration harness at `evalshift-cli/tests/integration/replay_client.py`. Remove `fixtures.jsonl` if nothing consumes it. -- [x] **Step 4:** Add a short note that tool-result fixtures are captured but not yet consumed by replay (links to Phase 2). -- [x] **Step 5:** Commit: `docs(examples): replace non-existent --offline flags with a working walkthrough`. - Done (sdk). Also found and fixed: example evalshift.yaml used removed keys `tools_path` and `tool_selection[].mode`, and lacked managed suites markers; `tools.yaml` removed with `fixtures.jsonl`; push host corrected to api.evalshift.dev. - -### Task 0.2 `[cli]` Correct migration-policy defaults in the config reference - -**Files:** `docs/configuration.md:77-89`, `llms-full.txt:905-915`, `DOCS.md` (verify only). - -- [x] **Step 1:** Change `max_cost_increase: 0.50` → `0.30` and `max_latency_increase: 2.0` → `0.30` in `docs/configuration.md`, and reword the trailing comments (they explain the 50%/200% values). -- [x] **Step 2:** Replace the "what init writes" block in `llms-full.txt` with the current `INIT_PROFILE_POLICIES["model-upgrade"]` from `src/evalshift/cli/commands/_scaffold.py:94-105`. -- [x] **Step 3:** Add a unit test that renders `INIT_PROFILE_POLICIES["model-upgrade"]` and asserts each `key: value` line appears verbatim in both `docs/configuration.md` and `llms-full.txt` (pattern: `tests/unit/test_init.py:148-161` already pins init ↔ code; extend to docs). -- [x] **Step 4:** Commit: `docs(config): align migration_policy defaults with init and schema`. - Done in cef4c85. Test also pins DOCS.md. - -### Task 0.3 `[cli]` Regenerate the committed root `evalshift.yaml` - -**Files:** `evalshift.yaml` (repo root). - -- [x] **Step 1:** Confirm it is a stale init output (0.03/0/0.95 policy, no `blocking: false`). -- [x] **Step 2:** Decide: either regenerate with `evalshift init --provider ` and preserve any project-specific values, or delete it if it is only a fixture. Check `git log --follow evalshift.yaml` and grep tests/CI for references first. -- [x] **Step 3:** Commit: `chore: drop stale root evalshift.yaml (unused leftover init output)` — deleted rather than regenerated: no consumers, no project-specific values (a2c8f24). - -### Task 0.4 `[cli]` Fix stale docstrings and doc counts - -**Files:** `src/evalshift/config/models.py:186`, `docs/configuration.md:302`, `docs/sdk.md:123-124`. - -- [x] **Step 1:** `config/models.py:186`: replace "Defaults to a strong Anthropic model" with the actual `DEFAULT_JUDGE_MODEL`, or reference the constant. -- [x] **Step 2:** `docs/configuration.md:302`: "Three sub-keys" → count the documented sub-keys and state that number, or drop the count. -- [x] **Step 3:** `docs/sdk.md:123-124`: LangChain is listed as "outside the SDK entirely" — correct to mention `EvalShiftCallbackHandler` (`evalshift-sdk[langchain]`). -- [x] **Step 4:** Commit: `docs: fix stale judge default, evaluator count, and LangChain adapter mention`. - Done in 1861439. - -### Task 0.5 `[sdk]` Fix dangling references in DECISIONS.md - -**Files:** `evalshift-sdk/docs/DECISIONS.md:4`, `:31-34`. - -- [x] **Step 1:** Line 4 references `IMPLEMENTATION_PLAN.md`, which does not exist. Remove the sentence or point at the real phase list. -- [x] **Step 2:** Lines 31-34 assert the CLI consumes the fixture table with a "halt-and-flag" default. Annotate: "Fixture table is written; CLI consumption is pending (see cli plan 2026-09-08-external-review-response.md Phase 2)." -- [x] **Step 3:** Commit: `docs(decisions): remove dangling plan reference, mark fixture consumption as pending`. - Done (sdk, 647e09f). Left alone: DECISIONS.md:51 still says SCHEMA_VERSION 1.0.0 (actual 2.0.0) — separate stale note. - -### Task 0.6 `[cli]` Add a `suites`-based example - -**Files:** new `examples/captured-suite/` (name to taste), `README.md` examples list. - -- [x] **Step 1:** Create an example whose `evalshift.yaml` is what `evalshift init` writes (passthrough `replay` prompt + managed `suites:` block) plus a small pre-promoted `golden.jsonl` under `.evalshift/suites/` (or wherever `capture sync` writes; confirm from `captures/promote.py`). -- [x] **Step 2:** Include one or two capture envelope files so `evalshift capture sync` can be demonstrated from scratch. -- [x] **Step 3:** Ensure `evalshift validate` and `evalshift doctor` pass on it; add to whatever test iterates examples (grep `examples/` in `tests/`). -- [x] **Step 4:** Commit: `docs(examples): add capture-first example using the suites block`. - Done. Named `examples/capture-first/`. Envelopes generated with the real SDK; suite produced by the real `capture sync`; a test re-derives golden.jsonl from the committed captures. - -### Task 0.7 `[cli]` Follow-up found during 0.6: `validate` has no `--suite-name` - -**Files:** `src/evalshift/cli/commands/validate.py`, `cli/commands/_suites.py` (`resolve_suite_path`), tests, `docs/`. - -`run`, `all`, `bundle` and `push` resolve suites through `_suites.resolve_suite_path`; `validate` hardcodes `--suite` defaulting to `./golden.jsonl`, so bare `evalshift validate` fails in every capture-first project. - -- [ ] **Step 1:** Failing test: `evalshift validate --suite-name oncall_triage` in `examples/capture-first` exits 0; bare `evalshift validate` in a project whose only suite is in the managed block also resolves it (single-suite default) or lists the names. -- [ ] **Step 2:** Implement via `resolve_suite_path`; keep `--suite ` working. -- [ ] **Step 3:** Update `examples/capture-first/README.md` "Check it" section and `docs/getting-started.md`. -- [ ] **Step 4:** Commit: `feat(validate): accept --suite-name and resolve suites from the config`. - ---- - -## Phase 1 — Honest framing in docs (no code) - -*Validity: full for #4 marketing; partial for #5, #9. Importance: medium. Cost: low.* - -### Task 1.1 `[cli]` Reword the "sequences" claim and state the single-round limit up front - -**Files:** `README.md:253-256`, `docs/agents.md:3-5`, `docs/agents.md:167-170`, `llms-full.txt` (agent section). - -- [x] **Step 1:** Replace "how it sequences them" with what is actually tested: which tools, what arguments, order and parallelism *within the first tool-emitting round*. -- [x] **Step 2:** Move the one-call-per-example statement from `docs/agents.md:167-170` to the top of the agents page, and add a one-liner in the README agent section. -- [x] **Step 3:** Commit: `docs(agents): state first-round-only replay scope where the claim is made`. - Done inside the Phase 2 docs pass (4c3db27): the single-shot default is stated at the top of `docs/agents.md` and in the README agent paragraph, alongside the `--rounds all` opt-in, so no caveat had to be written and then reverted. - -### Task 1.2 `[cli]` Document drift-vs-correctness and the `prompts`/`suites` relationship - -**Files:** `docs/evaluators.md` (semantic section, near `:57-61`), `docs/configuration.md` (top of `prompts` and `suites` sections), `docs/faq.md`. - -- [x] **Step 1:** Add a short "What 'expected' means" paragraph to the semantic evaluator docs: the yardstick is the source model, so a correct-but-reworded target reads as drift; that is why init ships it advisory; use the judge criterion for correctness. -- [x] **Step 2:** Add two sentences to `docs/configuration.md` explaining that `prompts` is the template axis and `suites` the dataset axis, that both are always present, and that the init `replay` prompt is the passthrough that makes captured inputs replayable. -- [x] **Step 3:** FAQ entry: "Why does a hand-written config block on semantic when init does not?" (library default `blocking: true`, `config/models.py:177`). - Done as part of Task 7.3 (c253027). -- [x] **Step 4:** Commit: `docs: explain drift vs correctness and prompts vs suites`. - Done (Steps 1, 2, 4). Also mirrored in `DOCS.md` (Prompts, Semantic) and `llms-full.txt` (prompts and semantic blocks). - ---- - -## Phase 2 — Make replay consume what capture records (#4) - -*Validity: full. Importance: high. Cost: medium-high. Depends on Phase 0.1 wording.* - -Goal: teacher-forced multi-round replay. For round *k* > 1, the candidate is given the recorded history through round *k-1* (including recorded tool results as fixtures) and asked for round *k*. Every round is scored against `expected_tool_rounds[k]`. - -### Task 2.1 `[cli]` Spec first - -**Files:** new `docs/superpowers/specs/2026-09-XX-teacher-forced-replay-design.md`. - -- [x] **Step 1:** Write the design: work-item shape (one `WorkItem` per round, or one per example with an inner loop), how recorded tool results are injected (provider-native `tool_result` messages built from `ToolResultEvent.result`, keyed by `call_id`), what happens when the candidate calls a tool with no fixture (halt-and-flag per SDK D-decision, or substitute a synthetic "unavailable" result), cache-key changes (round index), cost estimate changes (`utils/cost.py` multiplies by rounds), and policy/report changes (per-round divergence). -- [x] **Step 2:** Decide the default: `rounds: first` stays default for cost; `rounds: all` opts into teacher forcing. Confirm `promote.py:91` already has the enum. -- [ ] **Step 3:** Review with maintainer; then continue. - Spec: `docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md` (1c3abbe). Written and implemented without the maintainer review (they were away and asked for the phase to be done); the decisions to confirm are listed under **Maintainer decisions to confirm** below. Key choice: pure teacher forcing — the candidate's own calls are never fed back, so "candidate calls a tool with no fixture" cannot arise and no halt-and-flag policy was needed; self-conditioned replay is out of scope. Fixtures are positional (`tool_result_fixtures[k][i]` ↔ `expected_tool_rounds[k][i]`), not keyed by `call_id`, because `ExpectedToolCall` deliberately carries none. One `WorkItem` per example with an inner loop (one `raw.jsonl` row per example per role, resume per example). - -### Task 2.2 `[cli]` Fixture loading - -**Files:** `src/evalshift/captures/promote.py` (`_tool_rounds` `:457-483`, history recovery `:554-581`), `src/evalshift/suite/models.py` (`SuiteExample`), `tests/unit/test_promote.py`. - -- [x] **Step 1:** Failing test: promoting a two-round capture with `rounds: all` yields `expected_tool_rounds` of length 2 and a new `tool_result_fixtures: dict[call_id, result]` (or per-round list) on the example. -- [x] **Step 2:** Implement; keep v0.1–v0.3 suites loading unchanged. -- [x] **Step 3:** Commit: `feat(promote): carry recorded tool results as replay fixtures`. - Done: contract d20e2db (`SuiteExample.tool_result_fixtures`, `ToolResultFixture`, `rounds_to_replay()`; `ToolCall.round_index`, `ToolTrace.round_count`/`round()`/`rounds()`), promotion abbe3a5, example regenerated 2b6d153 (`capture sync --force`; case files also gained the `cost_usd`/`cost_source` fields 0638b61 never regenerated). Pairing: `call_id` first (executed and requested calls alike), then name within the same round; coverage stops at the first round with an unpaired call and warns. **Behaviour change:** `--rounds all` no longer flattens `expected_tools`; it is `expected_tool_rounds[0]` under both settings. - -### Task 2.3 `[cli]` Multi-round runner loop - -**Files:** `src/evalshift/runner/orchestrator.py` (`_build_work_list` `:588-600`, dispatch `:1021-1038`), `src/evalshift/runner/models.py`, `src/evalshift/models/client.py`, `tests/unit/test_orchestrator.py`, `tests/integration/`. - -- [x] **Step 1:** Failing integration test using the mocked client: candidate emits tool call in round 1, receives fixture result, emits round-2 call; both rounds recorded on the result. -- [x] **Step 2:** Implement loop with a hard cap equal to `len(expected_tool_rounds)`; unmatched fixture → record `fixture_missing` and stop the loop for that example. -- [x] **Step 3:** Extend cache key with round index; extend cost pre-flight with round count. -- [x] **Step 4:** Commit: `feat(runner): teacher-forced multi-round replay`. - Done in 14c57da (merged a288dfd) + 7034b71. Deviations from the step text, per the spec: the cap is `len(fixtures) + 1` (the answer round after the last covered round is replayed too, so text evaluators get the candidate's real answer), and there is no runtime `fixture_missing` — coverage is settled at promotion and validated at suite load. Round 0 dispatches byte-identically to before; rounds ≥ 1 go through `complete_messages_with_tools` with positional ids `call_r{j}_{i}`. Error in round k → `Call.error = "round k/n: …"`, no trace; single-shot errors keep the bare text. `cache_key(round_index=None)` keeps every existing key (tool path still bypasses the cache). `ReplayClient` fixtures accept an optional `"round"`. Cost prompt now prints the estimate's call count (counts rounds); `total_evaluations` and the progress bar stay per `Call` row. - -### Task 2.4 `[cli]` Score and report per round - -**Files:** `src/evalshift/evaluators/tool_selection.py`, `tool_trace_structure.py`, `analysis/policy.py`, `reports/html.py` + template, `docs/agents.md`, `docs/evaluators.md`. - -- [x] **Step 1:** Failing tests: tool_selection compares round *k* output to `expected_tool_rounds[k]`; report shows a per-round divergence row. -- [x] **Step 2:** Implement; policy budget `max_tool_divergence` counts an example as diverged if any replayed round diverges (document this). -- [x] **Step 3:** Update docs and revert the Phase 1.1 caveat to describe the new capability. -- [x] **Step 4:** Commit: `feat(evaluators): per-round tool scoring for multi-round replay`. - Done in c9e5a4b, ab2247e, dbc0400 (merged 38d83dc); docs 4c3db27. Shared helpers in `evaluators/tool_rounds.py`. Multi-round mode is entered only when a trace has `round_count > 1`, so a `--rounds first` suite that still carries `expected_tool_rounds` scores exactly as before. Scores are the mean over replayed rounds (a round with no ground truth and no calls on either side is skipped), per-round detail under `metadata.rounds`; top-level names stay flattened for existing consumers. No policy code change: the mean drops below 1.0 on any diverged round, pinned by `test_policy.py::TestAMultiRoundDivergenceCountsAsDiverged`. Report: one line per round in the tools column, `Round k:` trace-diff prefixes; bundle events carry the real `round`. - -#### Maintainer decisions to confirm (Phase 2) - -1. **Pure teacher forcing** (recorded calls + results fed back, never the candidate's own). Self-conditioned replay would need name+argument fixture lookup and a halt-and-flag policy; deferred. -2. **`--rounds all` no longer flattens `expected_tools`** (CHANGELOG *Changed*). The `agent_trace` evaluator is the path for externally produced multi-round traces. -3. **The answer round is replayed** (rounds = covered + 1), costing one extra call per fully covered example so text evaluators compare real final answers. -4. **`--rounds all` on a capture with no recorded results** warns and silently promotes a single-shot case (identical to `--rounds first`). Could be made a hard error in one line of `build_example_from_capture` if preferred. -5. **Progress bar counts `Call` rows, the cost estimate counts rounds**, so the two numbers differ on a multi-round suite. - ---- - -## Phase 3 — Record the model's requested tool calls (#3) - -*Validity: full. Importance: high. Cost: medium. Cross-repo; SDK first.* - -### Task 3.0 `[cli]` CLI trace model accepts `requested_tool_calls` (must land before 3.1) - -**Files:** `src/evalshift_cli/traces/models.py` (`ModelCallEvent`), `captures/reader.py` (version check), `tests/unit/test_trace_models.py`, `tests/unit/test_captures_reader.py`. - -Found while preparing Phase 3: the CLI's trace models inherit `extra="forbid"`, and the SDK dataclasses mirror them field for field (guarded by `evalshift-sdk/tests/conformance/test_parity.py` against the vendored copy in `cli_models_vendored.py`). If the SDK starts writing a new field before the CLI accepts it, every existing CLI install rejects every new capture at load time. So the plan's "SDK first" is inverted for the schema field: the CLI model lands first and is the contract. - -Contract (both repos, verbatim): `requested_tool_calls: list[RequestedToolCall] | None = None`, `RequestedToolCall = {name: str, arguments: dict[str, Any], call_id: str | None}`, positioned immediately after `tools_offered` so the parity test's field order holds. - -- [x] **Step 1:** Failing tests: `ModelCallEvent` accepts and round-trips the field; absent → `None`; a `2.1.0` envelope passes `captures/reader.py` (it checks the major only, `_SUPPORTED_MAJOR = 2`; lock that in with a test). -- [x] **Step 2:** Add `RequestedToolCall` (strict) and the field. -- [x] **Step 3:** Commit: `feat(traces): accept requested_tool_calls on ModelCallEvent`. - Done in 6b8f210 (cli). reader.py needed no change: major-only gate, now pinned by tests. - -### Task 3.1 `[sdk]` Schema: add `requested_tool_calls` to `ModelCallEvent` - -**Files:** `src/evalshift/trace/models.py:24-45`, `trace/schema.py`, `trace/serialize.py`, `trace/migrate.py`, `docs/SCHEMA.md`, `tests/`. - -- [x] **Step 1:** Failing tests: a `ModelCallEvent` round-trips a `requested_tool_calls: list[{name, arguments, call_id}] | None` field; a `2.0.0` envelope loads with the field absent → `None`. -- [x] **Step 2:** Add the field (default `None` so old writers/readers coexist). Bump `SCHEMA_VERSION` to `2.1.0`; register a no-op-with-default migration. Update `schema.py:80` fixed field set. -- [x] **Step 3:** Document in `docs/SCHEMA.md` and `docs/DECISIONS.md` (new D-requested: "requested ≠ executed; both recorded"). -- [x] **Step 4:** Commit: `feat(trace): record model-requested tool calls separately from executed ones`. - Done in aad7a21 (sdk). Also updated docs/REDACTION.md field table and tests/test_smoke.py's pinned version. - -### Task 3.2 `[sdk]` API: accept requested calls in `record_model_call` and the streaming recorder - -**Files:** `src/evalshift/capture/api.py:225-250` (`record_model_call`), `:529-546` (`set_usage` area of `capture.model_call`), `DOCS.md:278-340`, `llms-full.txt`. - -- [x] **Step 1:** Failing tests for `record_model_call(..., requested_tool_calls=[...])` and `rec.set_requested_tool_calls([...])`. -- [x] **Step 2:** Implement; redact arguments through the same redactor as `ToolCallEvent.arguments`. -- [x] **Step 3:** Add small stdlib-only helpers that extract requested calls from an already-serialised provider response dict (OpenAI `choices[0].message.tool_calls`, Anthropic `content[].type == "tool_use"`, Gemini `candidates[0].content.parts[].functionCall`). Pure dict walking, no provider import. -- [x] **Step 4:** Docs: explain the difference between "offered" (`tools=`), "requested" (new), and "executed" (`@capture.tool`). -- [x] **Step 5:** Commit: `feat(capture): requested_tool_calls on record_model_call and model_call recorder`. - Done in ba8bffa + ff7a272 (sdk; helpers merged in d69d397, vendored `name` min_length synced in 19964ee). Helper: `evalshift.capture.requested.extract_requested_tool_calls`; returns `[]` for a recognised response with no calls and `None` for an unrecognised one. - -### Task 3.3 `[sdk]` LangChain adapter: read `AIMessage.tool_calls` - -**Files:** `src/evalshift/adapters/langchain.py:465-478` (`on_llm_end`), `tests/adapters/`. - -- [x] **Step 1:** Failing test with a synthetic `LLMResult` whose generation message carries `tool_calls`. -- [x] **Step 2:** Populate `requested_tool_calls` from `generations[0][0].message.tool_calls` when present. -- [x] **Step 3:** Commit: `feat(langchain): capture requested tool calls from AIMessage`. - Done in 836ea92 (sdk). Reuses api's normaliser; chat message with no tool_calls records `[]`, plain text generation leaves the field unset. - -### Task 3.4 `[cli]` Promotion prefers requested calls as ground truth - -**Files:** `src/evalshift/captures/models.py`, `captures/promote.py` (`_tool_rounds` `:457-483`, `_unwrap_recorded_arguments` `:378-419`), `docs/agents.md:190-206`, tests. - -- [x] **Step 1:** Failing tests: when `requested_tool_calls` is present it becomes `expected_tools` / `expected_tool_rounds` verbatim and `_unwrap_recorded_arguments` is skipped; when absent, current executed-call behaviour is unchanged. -- [x] **Step 2:** Implement; add a `promotion_source: "requested" | "executed"` note to the promoted case metadata so reports can show which yardstick was used. -- [x] **Step 3:** Update `docs/agents.md` — the "No model can produce the recorded shape" section now applies only to legacy captures. -- [x] **Step 4:** Commit: `feat(promote): use model-requested tool calls as ground truth when captured`. - ---- - Done in 76a8c43 (cli). `promotion_source` on PromotedCase and BuiltExample; mixed captures fall back to executed for the whole capture with a warning; requested-vs-executed disagreement warns, requested wins. Checked-in capture-first cases gained `promotion_source: executed`. - -## Phase 4 — Stop LiteLLM from silently dropping migration-relevant params (#8) - -*Validity: full. Importance: high. Cost: medium. Cross-repo.* - -### Task 4.1 `[sdk]` Record `tool_choice`, `parallel_tool_calls`, and strictness - -**Files:** `src/evalshift/capture/generation.py:28-36` (`GENERATION_KEYS`), `adapters/langchain.py:101` (`_generation_config`), `capture/toolset.py` (strict flag on OpenAI function shape survives normalisation?), tests, `DOCS.md`. - -- [x] **Step 1:** Failing tests: `sanitize_generation_config({"tool_choice": ..., "parallel_tool_calls": False})` keeps both; a toolset with `function.strict: true` fingerprints differently from one without. -- [x] **Step 2:** Extend `GENERATION_KEYS`; verify `normalize_tools` does not prune `strict` (`DECISIONS.md:309-310` says it never prunes keys — add a test that locks that in). -- [x] **Step 3:** Commit: `feat(capture): record tool_choice, parallel_tool_calls, and strict schemas`. - Done in 781bf46 + 387f975 (sdk). The premise of Step 2 was wrong: `normalize_tools` *did* prune `strict` (it rebuilt every tool from name/description/parameters; D-toolset's "never prunes" is about `input_schema`). Contract, mirrored verbatim in the CLI: the canonical tool dict gains `"strict": true` only when the source declared it (OpenAI `function.strict` or top-level `strict`), so every existing fingerprint is byte-identical. `GENERATION_KEYS` also gains Gemini's `tool_config`; `jsonable` duck-types `model_dump` so a `ToolConfig` lands as a dict. No schema or sidecar version bump (additive optional key in a content-addressed file). LangChain adapter needed no change. - -### Task 4.2 `[cli]` Carry those params through replay - -**Files:** `src/evalshift/runner/generation.py:18-20` (`_HANDLED_KEYS`), `models/client.py:588-592` (tool serialisation), `evaluators/tool_models.py:60-77`, `tool_parser.py:39-64`, tests. - -- [x] **Step 1:** Failing tests: a captured `tool_choice` reaches the `litellm.acompletion` kwargs; `strict: true` survives `to_openai`; Anthropic equivalent (`tool_choice: {type: "tool", name}`) is produced for Anthropic targets. -- [x] **Step 2:** Implement translation per provider prefix. Where a target provider cannot express the constraint, record it (next task) rather than drop it. -- [x] **Step 3:** Commit: `feat(runner): pass tool_choice, parallel_tool_calls, and strict through replay`. - Done in e9894c9 (cli). `ToolSpec.strict` (read from `function.strict` or top-level `strict`, emitted only when true by both `to_openai` and `to_anthropic`, so the orchestrator's cache fingerprint matches the SDK sidecar). `translate_generation_config` normalises the OpenAI, Anthropic and Gemini (`tool_config`, snake or camel case) spellings into OpenAI-style `tool_choice` + `parallel_tool_calls`. No hand translation per provider: litellm 1.100.0 already maps OpenAI-style values for Anthropic (`_map_tool_choice`, incl. `disable_parallel_tool_use`) and Gemini (`toolConfig`), pinned by tests against the installed source. Sending the plan's "Anthropic equivalent" `{type: "tool", name}` would have been wrong — litellm's dict branch has no `"tool"` case and drops it. Un-expressible: Gemini `parallel_tool_calls` and Gemini `strict` (folded into Task 4.3); a `tool_choice` on a tool-less example is stripped with a warning. Ignored-key warnings deduped per key set. - -### Task 4.3 `[cli]` Surface dropped parameters instead of hiding them - -**Files:** `src/evalshift/models/client.py:483, :599` (`drop_params`), `models/capabilities.py`, `runner/generation.py:58-60`, `reports/html.py` + template (extend the `non_deterministic_models` banner pattern at `report.html.j2:49-68`), `analysis/policy.py`, docs. - -- [x] **Step 1:** Failing tests: when `litellm.get_supported_openai_params` says the target lacks `tool_choice` (or `response_format`), the run records `dropped_params[model] = {...}` and the report shows a "Constraints not honoured by target" banner. -- [x] **Step 2:** Implement using the existing capability probe (`capabilities.py:56-64`) before dispatch; keep `drop_params: True` so calls still succeed, but log at `warning` not `debug`. -- [x] **Step 3:** Optional policy knob `fail_on_dropped_params: bool = False` (document in `docs/configuration.md`). -- [x] **Step 4:** Commit: `feat(report): surface generation params the target model cannot honour`. - Done in 23f42c0 (merged 8e8f211) + d0f356f (cli). `unsupported_params` generalises the probe (`honors_temperature` now builds on it); `detect_dropped_params` runs at run start over the suite's recorded keys (mapped to litellm names: `response_mime_type`/`response_schema` → `response_format`, `tool_config` → `tool_choice`, `max_output_tokens` → `max_tokens`) and unions a hard-coded known-litellm-gaps table (`_KNOWN_LITELLM_GAPS`, cites litellm 1.100.0 and the file) for Gemini `parallel_tool_calls` and the pseudo-param `tools.strict`. Stored as `RunState.dropped_params`, rendered as a banner after the sampling banner, emitted in `report.json`, and gated by `migration_policy.fail_on_dropped_params` (top-level only). `temperature` is deliberately left to `non_deterministic_models`. Warnings once per (model, param) at run start; the client's duplicate Gemini warnings were removed. - ---- - -## Phase 5 — Thin provider client wrappers (#2) - -*Validity: substantive. Importance: adoption. Cost: medium per provider. Depends on Phase 3.2 helpers.* - -Constraint: stdlib-only runtime (D-deps). Each wrapper is an import-guarded optional extra like `adapters/langchain.py`, and wraps a *client instance* rather than monkeypatching the module. - -### Task 5.1 `[sdk]` Design note - - Done in 42e3c78 (sdk): D-wrappers in `docs/DECISIONS.md`, the three extras, and a shared base `adapters/_wrap.py` (ClientProxy, `instrument`, StreamProxy/AsyncStreamProxy) so each provider module only contributes `describe` / `complete` / `on_chunk`. One rule added beyond the plan: a wrapper always asserts `tools` per call (`[]` when the request carried none), never inherits the session's. - -**Files:** `docs/DECISIONS.md` (new D-wrappers). - -- [x] **Step 1:** Record: wrappers are proxies over the user's client object (`wrap_openai(client)`, `wrap_anthropic(client)`, `wrap_genai(client)`); they populate `model_id`, `tools`, `input`, `output`, `input_tokens`, `output_tokens`, `latency_ms`, `generation_config` (incl. Phase 4 keys), and `requested_tool_calls` (Phase 3); `cost_usd` stays 0 in the SDK — pricing belongs to the CLI (`utils/cost.py`, litellm's price table) and is applied at promote/report time. Streaming: wrap the iterator; usage taken from the final chunk. -- [x] **Step 2:** Decide extra names: `evalshift-sdk[openai]`, `[anthropic]`, `[google-genai]`. -- [x] **Step 3:** Record open-source coverage: no dedicated wrapper. Ollama, vLLM, llama.cpp server, LM Studio, TGI, Together, Groq, Fireworks and OpenRouter expose OpenAI-compatible endpoints, so `wrap_openai(OpenAI(base_url=...))` covers them unchanged; `model_id` is whatever string the caller passed, and a server that omits `usage` yields zero tokens (never gates promotion). Native non-OpenAI clients (the `ollama` package, in-process transformers) keep using `record_model_call` / a manual `model_call` span. Replay of open-model targets is the CLI's job via litellm prefixes and is independent of this phase. - -### Task 5.2 `[sdk]` OpenAI wrapper - - Done in 6098bdc (sdk). Responses-API `input`/`instructions` are recorded as a messages-style list (a dict would be taken verbatim by `_recover_inputs`); flat Responses tools are translated to the nested chat shape because `toolset._normalize_dict` reads `input_schema`. Chat-stream usage needs `stream_options={"include_usage": True}`. Base fixes found here and by 5.3: Stainless SDKs wrap async `create` in a sync `@required_args` wrapper, so `instrument` checks `inspect.unwrap` and also resolves a plain `def` that returns an awaitable. - -**Files:** new `src/evalshift/adapters/openai.py`, `pyproject.toml` extras, `tests/adapters/test_openai.py` (synthetic response objects; one `importorskip`-guarded real-client smoke test like langchain), `README.md`, `DOCS.md`. - -- [x] **Step 1:** Failing tests: sync `chat.completions.create`, async, and streaming each produce one `model_call` with usage, latency, offered tools, requested tool calls. -- [x] **Step 2:** Implement as a proxy that forwards everything and intercepts only `chat.completions.create` / `responses.create`. Fail-open: any wrapper error logs and returns the real response. -- [x] **Step 3:** Commit: `feat(adapters): OpenAI client wrapper`. - -### Task 5.3 `[sdk]` Anthropic wrapper - - Done in e062a58 (sdk). `system` is prepended as a `{"role": "system"}` message (the only place the CLI's replay looks for it). `messages.stream` is a manager proxy: yields the real `MessageStream` unchanged and records on `__exit__` from `get_final_message()` (or `current_message_snapshot` when the body raised). - -- [x] Same shape as 5.2 for `messages.create` and `messages.stream`. Tool calls from `content[].type == "tool_use"`. Commit: `feat(adapters): Anthropic client wrapper`. - -### Task 5.4 `[sdk]` Google GenAI wrapper - - Done in 9f58ab7 (sdk). `contents` + `system_instruction` are folded into a messages-style list (a dumped `Content` is `{role, parts}`, which `_looks_like_messages_list` rejects). Callable tools (AFC) are declared through the SDK's `FunctionDeclaration.from_callable_with_api_option` via a lazy guarded import; a built-in-only `Tool` leaves the toolset unstamped. Follow-up: `toolset._normalize_gemini_tool` ignores `parameters_json_schema`, so manual `record_model_call(tools=[Tool(...)])` callers get less fidelity than the wrapper. - -- [x] Same shape for `models.generate_content` (sync/async/stream). Reuse the existing duck-typed Gemini toolset handling in `capture/toolset.py`. Commit: `feat(adapters): google-genai client wrapper`. - -### Task 5.5 `[cli]` Fill cost at promotion when the capture has tokens but no cost - - Done in 0638b61 (cli). `PromotedCase` gains `cost_usd` and `cost_source: recorded | estimated | null` (not on `SuiteExample`: provenance of the run, not something replay reproduces). Per event: keep a non-zero recorded cost, else price by that event's own `model_id`; sum; mixed → `estimated`. `litellm.cost_per_token` is gated on a pure `litellm.model_cost` lookup because a blind call prints a provider banner for unknown ids and opens a socket to localhost for `ollama/` ids. - -**Files:** `captures/promote.py`, `utils/cost.py`, tests, `docs/agents.md`. - -- [x] **Step 1:** Failing test: capture with `input_tokens>0`, `cost_usd==0` is promoted with a cost estimate derived from litellm's price table for `model_id`, tagged `cost_source: "estimated"`. -- [x] **Step 2:** Failing test: capture whose `model_id` has no entry in litellm's price table (local / self-hosted, e.g. `llama3.1:8b`) promotes with `cost_usd` left at 0 and no `cost_source` tag, without a warning or error. A missing price is the normal case for open-source models, not a failure. -- [x] **Step 3:** Implement; leave recorded non-zero costs untouched. -- [x] **Step 4:** Commit: `feat(promote): estimate cost from tokens when the SDK recorded none`. - ---- - -## Phase 6 — Resolve the `evalshift` namespace collision (#1) - -*Validity: full. Importance: high friction. Cost: breaking change, needs a major/minor bump and a deprecation window. Do after Phases 2–4 so the SDK schema is stable before the packaging move.* - -### Task 6.1 Decide the shape - -**Files:** `evalshift-sdk/docs/DECISIONS.md` (D1-followup → resolved), new spec in `evalshift-cli/docs/superpowers/specs/`. - -Options already named by the author: -- **A. CLI depends on SDK.** `evalshift` (CLI) declares `evalshift-sdk` as a dependency; the SDK owns the top-level `evalshift` package; CLI code moves under `evalshift.cli`/`evalshift._cli` or a separate top-level `evalshift_cli` package with the `evalshift` console script. Pro: `import evalshift` unchanged for SDK users. Con: CLI must never break SDK's stdlib-only promise; two repos share one import root. -- **B. Rename the SDK import** to `evalshift_sdk` and keep a deprecated `evalshift` shim for one minor version. Pro: clean separation, no cross-repo package root. Con: every existing `from evalshift import capture` breaks after the shim window. -- **C. Single distribution, `[cli]` extra.** Merge repos; `pip install evalshift` is the SDK, `pip install "evalshift[cli]"` adds the CLI. Pro: simplest for users. Con: repo merge, AGPL (CLI) vs MIT (SDK) licensing has to be reconciled per subpackage. - -- [x] **Step 1:** Picked **A** in its second form: CLI import package `evalshift_cli`, distribution name and console script unchanged, `evalshift-sdk>=0.3.0` declared as a runtime dependency. The first form (CLI code under the SDK's `evalshift` root) needs a PEP 420 namespace package, which the SDK's re-exporting `__init__.py` rules out; editable co-installs would need it too. -- [x] **Step 2:** Spec: `docs/superpowers/specs/2026-09-09-namespace-collision-design.md` (3dbf245). No shim is possible — any `evalshift/` file shipped by the CLI recreates the clash — so the window is the CHANGELOG entry plus a minor bump. Found seven two-venv passages in the CLI and three in the SDK (Findings #1 listed four). - -### Task 6.2 Implement per the chosen spec - -- [x] `[cli]` Package move `src/evalshift` → `src/evalshift_cli`; imports rewritten in `src/`, `tests/`, `scripts/`; tooling paths (mypy, ruff first-party, coverage, Makefile, CI, pre-commit); `evalshift-sdk` dependency; console script → `evalshift_cli.cli.main:app`; deferred-warnings printer matches its own records under the new root (1e1de65). -- [x] `[cli]` `doctor` row `evalshift-sdk`: ok with the SDK version; warn when missing, when the import fails, or when an older CLI's files or a local `evalshift/` directory shadow it; never fails the command (5da36c1). -- [x] `[cli]` Docs: README, getting-started, sdk.md, DOCS.md, AGENTS.md, llms-full.txt, faq.md, examples/capture-first; CHANGELOG Breaking + Added entries (d9c4eba, 1e1de65, 5da36c1). -- [x] `[sdk]` DECISIONS.md D-pkg and D1-followup marked resolved; README, DOCS.md, support_agent example; vendored-mirror header path; CHANGELOG. No code change (ea54c0a). -- [ ] `[cli]` Version bump to 0.14.0 — left for the maintainer's `chore(release): 0.14.0` commit, since CONTRIBUTING makes that bump the release itself. The CHANGELOG entry already names 0.14.0. - - Verified: ruff, format, `mypy --strict`, and 1893 tests green with `evalshift-sdk` 0.3.0 installed from PyPI; the built wheel ships 103 `evalshift_cli/` files, no `evalshift/` entry, and `Requires-Dist: evalshift-sdk>=0.3.0`; `evalshift doctor` shows the row; the capture-first agent and `capture sync` run from the single CLI venv; SDK tests 510 green. - ---- - -## Phase 7 — Statistical rigour and judge hygiene (#6, #7, #5 residue) - -*Validity: partial. Importance: medium. Cost: low–medium.* - -### Task 7.1 `[cli]` Optional repeated sampling per example - -**Files:** `config/models.py` (`defaults`), `runner/orchestrator.py:588-600` (`_build_work_list`), cache key `:911-920`, `analysis/statistics.py`, `reports/html.py` + template, `docs/methodology.md:469-472`, `docs/configuration.md`, tests. - -- [x] **Step 1:** Failing tests: `defaults.samples_per_example: 3` emits three source and three target items per pair, cache key includes the sample index, and the per-pair score is the mean with within-pair variance recorded. -- [x] **Step 2:** Implement; default stays 1. Cost pre-flight multiplies by the sample count. -- [x] **Step 3:** Report: show "samples per example: N" next to the pair count; when N == 1 and any model is non-deterministic, reuse the existing banner text. -- [x] **Step 4:** Update `docs/methodology.md` (replace the "run multiple seeds upstream" advice). -- [x] **Step 5:** Commit: `feat(runner): samples_per_example for repeated sampling`. - Done in ba9a128 (cli). `Defaults.samples_per_example` (1–20). `WorkItem.sample_index` / `Call.sample_index` (defaulted, so old `raw.jsonl` resumes); resume key and `total_evaluations` include the sample; `RunState.samples_per_example` recorded at run start. `cache_key(sample_index=None)` follows the `round_index` inclusion rule and the orchestrator passes the real index only when N > 1, so N == 1 keys are byte-identical and an N > 1 run is never served one cached response N times. Source sample *i* is paired with target sample *i*; scoring runs per sample pair with the evaluators untouched, then `_reduce_sample_cells` folds them into **one** `EvalRecord` per (prompt, example, evaluator, kind): means over the successful samples, `metadata.samples = {n, scored, source_scores, target_scores, deltas, delta_variance}`, `error` only when every sample failed. So analysis, policy, slicing, report and bundle still see one row per example and statistical *n* stays the example count — repeats never inflate power. Downstream consumers of `raw.jsonl` (report example rows, hosted bundle, insights, `inspect`) use `representative_calls()` = sample 0; economics and policy sum over every row. - -### Task 7.2 `[cli]` Runtime warning when the judge shares a family with source or target - -**Files:** `cli/commands/doctor.py`, `cli/commands/validate.py`, `evaluators/llm_judge.py`, `reports/html.py` + template, tests. - -- [x] **Step 1:** Failing tests: `doctor`/`validate` emit a warning when `judge_model` resolves to the same provider prefix as `source_model` or `target_model`; the HTML report shows a one-line "judge shares a model family with the target" note. -- [x] **Step 2:** Implement using `models/registry.py` provider resolution. Warning only; never fail. -- [x] **Step 3:** Commit: `feat(doctor): warn when the judge shares a model family with a compared model`. - Done in 6e0bec7 (merged 0edcedf). New `models/family.py` (`shared_judge_family`, `judge_family_overlaps`, `configured_judge_models` — top-level plus every `suites:` override — and `describe_overlap`); provider `other` never matches. `doctor` prints one warn `judge family` row per overlapping judge and an `ok` "from a third family" row otherwise; no row when no `llm_judge` is configured or either `defaults.source_model`/`target_model` is unset (doctor has no `--from/--to`). `validate` prints the same sentence as a `⚠` line. The report adds a third banner and a `report.json` `judge_family_overlap` field, only for judges that actually wrote `scores.jsonl` rows. Deviations: `defaults.judge_model` is *not* treated as a judge — it only seeds `insights_model`, never a pairwise verdict; `evaluators/llm_judge.py` is unchanged because the evaluator never sees the arms at construction. - -### Task 7.3 `[cli]` Consider flipping the library default for semantic `blocking` - -**Files:** `config/models.py:177`, `docs/configuration.md:304-318`, CHANGELOG. - -- [x] **Step 1:** Decided **not to flip** in `version: 1`: the flip would silently turn a gating evaluator advisory for every hand-written config that omits the key, so a migration that failed yesterday would pass today with no config change. A loosened gate under a minor bump is worse than the init/library asymmetry. Revisit under a `version: 2` schema. -- [x] **Step 2:** Not flipped, so instead the asymmetry is documented: `blocking` docstrings on `SemanticEvaluatorConfig`/`LLMJudgeConfig`, `docs/configuration.md` (blocking, semantic and llm_judge sections), and the FAQ entry from Task 1.2 Step 3 (c253027). No CHANGELOG entry. - -#### Maintainer decisions to confirm (Phase 7) - -1. **Samples are collapsed at `evaluate` time** into one `scores.jsonl` row per example (means; per-sample scores under `metadata.samples`). `EvalRecord` has no `sample_index`; the hosted bundle and report example rows show sample 0 only. `BUNDLE_SPEC.md` has no sample concept, so shipping per-sample outputs is a server-side spec question. -2. **`delta_variance` is recorded but not rendered**: a "noisy example" marker in the HTML example table could use it. -3. **`defaults.judge_model` is exempt from the family warning** because it only seeds `insights_model`. -4. **Semantic `blocking` stays `True`** (Task 7.3) until a `version: 2` schema. -5. **`doctor` cannot warn when arms come only from `--from/--to` on `run`**; a warn at `run` start would close that gap. The report note also needs a loadable config at report time; recording `judge_model` in each `llm_judge` row's metadata would make it config-independent. - ---- - -## Not planned (review points judged invalid) - -- Treating `prompts` as a leftover or removing `python_string` detection (#9). Both are required by the recommended flow. Only the docs residue (Tasks 0.2, 0.6, 1.2) is actioned. -- Adding a LlamaIndex adapter (#2). It is listed as a non-goal in the CLI README and appears on no roadmap. Provider wrappers (Phase 5) are the higher-leverage substitute the reviewer suggested. -- Changing the default `min_equivalence_rate` (#5). Init already writes 0.75 and ships semantic/judge advisory. - -## Progress log - -| Date | Task | Note | -|---|---|---| -| 2026-09-08 | — | Plan written from verified findings. | -| 2026-09-08 | 0.1–0.6 | Phase 0 complete. cli: a2c8f24, cef4c85, 1861439, 54ff55f. sdk: 647e09f, 7853167. Found and logged Task 0.7 (validate lacks --suite-name). | -| 2026-09-09 | 6.1–6.2 | Phase 6 done out of order (before 2–4, at the maintainer's request). cli: 3dbf245, 1e1de65, 5da36c1, d9c4eba. sdk: ea54c0a. Version bump deferred to the 0.14.0 release commit. | -| 2026-09-09 | 5.1–5.5 | Phase 5 complete. Shared base written by the coordinator (sdk 42e3c78), then four parallel agents: openai, anthropic and genai wrappers in SDK worktrees, cost estimation on cli main. sdk: 42e3c78, e062a58, 6098bdc, 0afee20 (merge), 2d4e65b (pre-existing mypy failure in `tests/test_toolset.py`), 9f58ab7, 5ceb191, ad025f6 (docs). cli: 0638b61. Gates green in both repos (sdk 709 tests, cli 2049). Version bump deferred to the release commit (SDK 0.4.0: `[openai]`/`[anthropic]`/`[google-genai]` extras). Open: `toolset._normalize_gemini_tool` should read `parameters_json_schema`; `validate --suite-name` (Task 0.7) still missing. | -| 2026-09-09 | 4.1–4.3 | Phase 4 complete, run as three parallel Opus agents (sdk; cli main; cli worktree for 4.3) plus one follow-up to fold the Gemini gaps into `dropped_params`. sdk: 781bf46, 387f975. cli: e9894c9, 23f42c0, 8e8f211 (merge), d0f356f. End-to-end verified with the dev SDK: an OpenAI-shaped strict tool + `tool_choice: required` + `parallel_tool_calls: false` capture promotes with those keys and a `strict: true` sidecar; `translate_generation_config` emits both; `detect_dropped_params` reports `['parallel_tool_calls', 'tools.strict']` for a Gemini target and nothing for Anthropic. Open: `validate --suite-name` (Task 0.7) still missing; the capture-first example still records executed-only until SDK 0.4.0 ships. | -| 2026-09-09 | 2.1–2.4, 1.1 | Phase 2 complete. Spec + model contract by the coordinator (1c3abbe, d20e2db), then three parallel Opus agents: promotion on main (abbe3a5, 2b6d153), runner in a worktree (14c57da, merged a288dfd, follow-up 7034b71), scoring/report/bundle in a worktree (c9e5a4b, ab2247e, dbc0400, merged 38d83dc); docs pass 4c3db27 also closes Task 1.1. Gates green: ruff, format, `mypy --strict`, 2156 tests, pre-commit. End-to-end verified on the capture-first captures synced with `--rounds all` against a fake client: 10 model calls for 6 rows, round-2 messages carry the recorded calls plus fixture results, per-round scores in `scores.jsonl`, and the report reads "Round 1: Target omitted page_oncall / Round 2: Target added search_logs". Five maintainer decisions listed under Task 2.4. Open: Task 0.7 (`validate --suite-name`), Task 1.2, Phase 7; the capture-first example stays on `--rounds first`. | -| 2026-09-09 | 1.2 | Task 1.2 complete (docs only). Open: Task 0.7 (`validate --suite-name`), Phase 2/7 maintainer decisions, version bump at release. | -| 2026-09-09 | 7.1–7.3 | Phase 7 complete, run as two parallel agents (7.1 on main; 7.2+7.3 in a worktree, merged 0edcedf with additive conflicts in `reports/`). cli: ba9a128, 6e0bec7, c253027. Gates green: ruff, format, `mypy --strict`, 2219 tests, pre-commit. Smoke: `doctor` and `validate` on capture-first with `target_model` set print the judge-family warning. Task 1.2 Step 3 closed by the 7.3 FAQ entry. Open: Task 0.7 (`validate --suite-name`), Task 1.2 Steps 1–2, the five Phase 7 decisions above, version bump at release. | -| 2026-09-10 | docs audit | Doc sweep against Phases 0–7. Fixed: SDK `DECISIONS.md` §1 and `examples/support_agent/README.md` still said the CLI does not replay recorded tool results (Phase 2 shipped it); CLI `llms-full.txt` still told users to install the SDK in a separate venv (Phase 6); CLI `docs/sdk.md`, `DOCS.md`, `README.md`, `docs/getting-started.md` and `llms-full.txt` never mentioned the Phase 5 provider wrappers or the Phase 3 `requested_tool_calls` path; `docs/sdk.md` lacked `--rounds`; capture-first README now says why its cases are `promotion_source: executed`. Verified current: `validate` docs claim no `--suite-name` (Task 0.7 still open); doctor rows, `report.json` fields, `samples_per_example`, judge-family and dropped-params docs all present in DOCS.md, llms-full.txt and the `docs/` pages. | -| 2026-09-09 | 3.0–3.4 | Phase 3 complete, run as three parallel Opus agents (cli; sdk core; sdk helpers in a worktree) then one for LangChain. cli: 6b8f210, 76a8c43, 3222b6a. sdk: aad7a21, ba8bffa, ff7a272, d69d397, 19964ee, 836ea92. Task 3.0 added: the CLI's `extra="forbid"` trace model must accept the field before the SDK writes it. End-to-end verified with the dev SDK: 2.1.0 envelopes with requested calls promote with `promotion_source: requested`. Open: whole-capture fallback when only some model calls carry the field (per-round hybrid considered, not done); capture-first example still records executed-only until SDK 0.4.0 ships the kwarg. | diff --git a/docs/superpowers/plans/2026-09-18-policy-single-source-of-truth.md b/docs/superpowers/plans/2026-09-18-policy-single-source-of-truth.md deleted file mode 100644 index c75d0dd..0000000 --- a/docs/superpowers/plans/2026-09-18-policy-single-source-of-truth.md +++ /dev/null @@ -1,327 +0,0 @@ -# Migration Policy — Single Source of Truth (`evalshift.yaml` → push → server) - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. Do one task at a time; each task is independently shippable. Every logic task is TDD: write the failing test, watch it fail, then code. - -**Goal:** A project's migration policy is written once, in `evalshift.yaml`, travels with every pushed run, and is what every surface gates on: the local `compare --policy-gate`, the run's Policy tab, the hosted policy-check endpoint, the PR list "blocked" signal, and the GitHub Action's commit status. The web app displays the policy; it no longer edits it. - -**Scope:** Four repositories, in deploy order: `[server]` → `[cli]` → `[client]` → `[action]`. `thresholds` (the free-form, non-gating key) is explicitly **out of scope** and keeps its current push-sync behaviour. *(Superseded 2026-09-19 — Phase 5's `[cli]` item deleted `thresholds` outright; see the note there.)* - -**Tech Stack:** server — FastAPI, pydantic, raw SQL via SQLAlchemy `text()`, alembic, pytest on SQLite, `ruff` + `mypy --strict`. cli — Python 3.11+, pydantic, typer/rich, pytest, `mypy --strict`. client — React + TypeScript, vitest + testing-library. action — stdlib Python, pytest. - ---- - -## Part 1 — Why (verified 2026-09-18) - -Two independent policy engines gate the same run and can disagree: - -| Surface | Policy it reads | Engine | Evidence | -|---|---|---|---| -| `evalshift compare --policy-gate`, run Policy tab, bundle `decision` | `evalshift.yaml:migration_policy` (9 fields, CLI defaults) | `analysis/policy.py:evaluate_migration_policy` | `hosted/bundle.py:135-147`, `cli/commands/compare.py:744-752` | -| `GET /runs/{id}/policy-check`, PR list `latest_status`, GitHub Action commit status | `projects.migration_policy_json` (6 fields, no defaults, web-only editor) | `app/policy/service.py:evaluate_policy` | `app/policy/service.py:895-918`, `app/runs/service.py:1838-1906`, action `scripts/evalshift_action.py:911-964` | - -- The CLI never uploads `migration_policy`. Only `thresholds` ride on `POST /runs` (`hosted/client.py:102-128`, `app/runs/schemas.py:30-36`). -- A project nobody configured on the web returns `inconclusive`; the Action treats that as not-a-failure (`_policy_gating`, `status == "inconclusive"` → `should_fail = False`). **The PR gate is silently off** even when the repo's yaml has a strict policy. -- Schema drift: CLI has `max_tool_divergence`, `tool_argument_drift_floor`, `fail_on_dropped_params`; the server model is `extra="forbid"` without them (`app/policy/schemas.py:8-11` cites a `FEATURES_TODO.md` that does not exist in any repo). The server cannot evaluate tool divergence at all — `run_policy_metrics` has no divergence column (`app/policy/service.py:790-792`). -- `PolicyCheckResponse.verdict` vs `.status` is the documented "these can differ" seam (`app/policy/schemas.py:96-99`). - -## Part 2 — Decisions - -**D1 — The policy rides inside the bundle, as `decision.policy`.** -Not in `POST /runs` metadata. `BUNDLE_SPEC.md:10` and server `CLAUDE.md:12` already say the bundle is the source of truth for policy decisions; a decision that does not say which budgets produced it is incomplete. This also fixes a local gap: `migration_decision.json` becomes self-describing, so `report` can state the budgets the verdict was made under without re-reading the config. `decision.policy` is the **resolved** policy (CLI defaults applied, all nine top-level fields present, `slices` present even if empty) or `null` when `migration_policy` is unset. Server `Decision` is `extra="forbid"` (`app/runs/bundle.py:509`), so the field is added there as optional **before** any CLI emits it. - -**D2 — For a run that carries a policy, the server's answer *is* the CLI's decision.** -`policy-check` returns `status = runs.verdict`, the stored `run_budget_results` and `run_regressions`, and the bundle's `reason`, with `policy_source = "run_policy"`. No re-evaluation: the server engine covers six of nine budgets and would disagree with the bundle exactly where the extra three matter. `verdict` and `status` are equal by construction. Existing runs without a snapshot keep today's re-evaluation against the legacy project column, with `policy_source = "project_policy"`. The dual code path is the migration path and is removed once no legacy-policy project has been pushed to in 90 days (tracked as a follow-up, not in this plan). - -**D3 — Per-run snapshot, stored in a new nullable `runs.policy_jsonb` column.** -Written at finalize from `decision.policy`. A run is gated on the policy it was pushed with, forever; a later yaml change never rewrites history. Nothing is written to `projects.migration_policy_json` on push. That column becomes **legacy, read-only**: `PATCH /projects/{id}` rejects `migration_policy` with 422, the create endpoint stops accepting it, and the client's edit/create/reset flows are deleted. - -**D4 — "The project's policy" for display = the policy of the latest available run on the default branch, else the latest available run, else the legacy column, else none.** -Served by a new `GET /projects/{id}/policy`. The web card shows it read-only, names the run it came from, and offers "Copy as YAML". The legacy case shows a banner telling the owner to move it into `evalshift.yaml`, with the YAML pre-rendered. The empty case shows the starter template as YAML to paste. `GET /policy/template` stays (it feeds that snippet); the client keeps no copy of the numbers. - -**D5 — Server `MigrationPolicy` schema is widened, not required-nine.** -`max_tool_divergence`, `tool_argument_drift_floor` (top-level and per-slice) and `fail_on_dropped_params` (top-level only) are added as `Optional`, default `None`. Legacy six-field rows keep validating; a CLI snapshot always fills all nine. Bounds mirror the CLI (`config/models.py:498-525`). `extra="forbid"` stays. - -**D6 — No policy is loud, not silent.** -`evalshift push` prints a warning when the run carries no policy. `policy-check` gains `policy_source = "none"` reason text that names the fix. The Action emits a GitHub `::warning::` annotation and a distinct commit-status description when `policy_source` is `none`; it still does not fail the job (behaviour change to failing is a documented opt-in via a new `require-policy` input, default `false`). - -**D7 — Adoption hint for projects that only have a web policy.** -`RunUploadResponse` gains `legacy_project_policy: dict | null` (the column value, when set). If the CLI's config has no `migration_policy` and the server reports a legacy one, `push` prints it as a ready-to-paste `migration_policy:` YAML block after the upload. The legacy column is never cleared automatically. - -**D8 — Permissions.** Pushing a policy needs only `run:create` — it is evidence about the run, like the rest of the decision block. `policy:configure` remains for `thresholds` only. A PR can loosen its own gate by editing the yaml; that is visible in the diff and is how every CI config works. Org-level floors are a possible later layer, not part of this plan. **[Superseded 2026-09-19:** Phase 5 removed `thresholds` from the CLI, so `policy:configure` now gates nothing the CLI is able to send. A `[server]` cleanup of that permission and of the orphaned `canonical_thresholds` response field is unscheduled.**]** - -**Rollout order matters.** A new CLI emitting `decision.policy` against an old server is rejected at finalize (`extra="forbid"`). Ship and deploy `[server]` Phase 1 before releasing `[cli]` Phase 2. Version bumps are deferred to release per project convention. - ---- - -## Phase 1 — `[server]` accept, store, and answer from the run's own policy - -### Task 1.1 — Widen `MigrationPolicy` / `SliceMigrationPolicy` to the CLI's field set (D5) - -**Files:** `app/policy/schemas.py`, `tests/test_phase3_policy.py`, `tests/test_phase_d2_policy_no_default.py` - -- [x] Test: a nine-field policy (the CLI's resolved shape, including `slices` with `max_tool_divergence` / `tool_argument_drift_floor` overrides) validates; a six-field legacy policy still validates with the new fields `None`; an unknown key is still rejected; out-of-bounds values for the new fields (e.g. `tool_argument_drift_floor: 1.5`) are rejected. -- [x] Add to `SliceMigrationPolicy`: `max_tool_divergence: float | None` (0–1), `tool_argument_drift_floor: float | None` (0–1). -- [x] Add to `MigrationPolicy`: the same two, plus `fail_on_dropped_params: bool | None = None`. All optional. Keep the six existing budgets required. -- [x] Rewrite the module docstring: remove the `FEATURES_TODO.md §9a/§9b` references (file does not exist); state that the server **stores** the full CLI shape and **evaluates** only the six for legacy runs (D2). -- [x] `STARTER_POLICY_TEMPLATE` gains the CLI's defaults for the three new fields (`max_tool_divergence=0.20`, `tool_argument_drift_floor=0.9`, `fail_on_dropped_params=False`) so a pasted template is a complete yaml block. -- [x] `make lint && make test`. - -### Task 1.2 — Accept `decision.policy` in the bundle (D1) - -**Files:** `app/runs/bundle.py`, `schemas/bundle_manifest.schema.json` (via `make export-schemas`), `tests/test_bundle_schema_parity.py`, `BUNDLE_SPEC.md` - -- [x] Test (parity): a full bundle with `decision.policy` set to a nine-field policy validates in both validators; a bundle with `decision.policy: null` validates; a bundle **omitting** `decision.policy` validates (old CLIs); a policy with an unknown key is rejected by both. -- [x] Add `policy: MigrationPolicy | None = None` to `Decision`. Import from `app.policy.schemas` — check for an import cycle (`app/policy/schemas.py` imports `BudgetResult`/`BlockingRegression` from `app.runs.bundle`). If cyclic, move `MigrationPolicy`/`SliceMigrationPolicy` into `app/runs/bundle.py` and re-export from `app.policy.schemas`. -- [x] `make export-schemas`; commit the regenerated JSON schema. -- [x] `BUNDLE_SPEC.md`: under Decision, add `policy` — "the resolved migration policy the verdict was computed under; `null` when the CLI had none; absent in bundles from CLI < the release that ships Phase 2. The server stores it verbatim and gates on it (D2)." -- [x] `make lint && make test`. - -### Task 1.3 — Migration: `runs.policy_jsonb` - -**Files:** `migrations/versions/202609180100_phase17_run_policy_snapshot.py`, `app/runs/service.py` (`RUN_COLUMNS`, `_run_from_row`, the `Run` record model), `tests/test_db_harness.py` or the existing migration-smoke test - -- [x] Migration `202609180100`, `down_revision = "202609160100"`: `ALTER TABLE runs ADD COLUMN policy_jsonb JSONB NULL`. Docstring in the repo's style: why per-run (history never rewritten), why nullable (pre-Phase-2 CLIs, and runs with no policy), why the project column is not touched. -- [x] Add `policy_jsonb` to `RUN_COLUMNS`; expose as `policy: dict[str, Any] | None` on the run record; parse with the same `_coerce_json_object` used for `summary_jsonb`. -- [x] Test: the SQLite harness creates the column; `_run_from_row` round-trips a policy dict and a `NULL`. -- [x] `make lint && make test`. - -### Task 1.4 — Write the snapshot at finalize - -**Files:** `app/runs/service.py` (finalize UPDATE near `:835`), `app/runs/detail_writer.py` (optional: keep it in `service.py` beside `summary_jsonb`), `tests/test_phase4_runs_api.py` - -- [x] Test: finalizing a bundle with `decision.policy` stores it; `GET /runs/{id}` (or the run record) exposes it; finalizing a bundle without the field stores `NULL`; re-finalize after `_reset_failed_run` clears it (add `policy_jsonb = NULL` to the reset at `:714`). -- [x] Persist `bundle.decision.policy.model_dump(mode="json")` (or `None`) into `policy_jsonb` in the same UPDATE that writes `verdict` and `summary_jsonb`. -- [x] `make lint && make test`. - -### Task 1.5 — `policy-check` answers from the snapshot (D2) - -**Files:** `app/policy/service.py`, `app/policy/schemas.py` (`PolicySource`, `PolicyCheckResponse`), `tests/test_phase3_policy.py`, `tests/test_phase_d2_policy_no_default.py` - -- [x] Extend `PolicySource = Literal["run_policy", "project_policy", "none"]`. -- [x] Test: a run with a stored snapshot returns `status == run.verdict`, `policy_source == "run_policy"`, `policy` equal to the snapshot, `budgets` from `run_budget_results` (overall scope + slices, as the existing `RunBudgetsResponse` reads them), `blocking_regressions` from `run_regressions`, `reason` from the stored decision. Cover all four verdicts. A **legacy** project policy present on the project must be ignored when a snapshot exists. -- [x] Test: a run without a snapshot on a project with a legacy policy keeps today's behaviour and `policy_source == "project_policy"`. -- [x] Test: a run without a snapshot on a project with no policy → `inconclusive`, `policy_source == "none"`, and `reason` names the fix ("add `migration_policy` to evalshift.yaml and push again"). -- [x] Implement `evaluate_run_policy`: branch on `run.policy is not None`. Add `load_stored_verdict_evidence(session, run_id)` that reads `run_budget_results` (all scopes) and `run_regressions`; do not reuse `load_stored_decisions`, whose docstring explicitly excludes limits/pass-fail for the re-evaluation path. -- [x] Store `decision.reason` if it is not already persisted (check `summary_jsonb` / `run_recommendations`; if absent, add `reason` to the finalize UPDATE as a column — `runs.decision_reason TEXT NULL` in the Task 1.3 migration). -- [x] Update `PolicyCheckResponse` docstring: `status` equals `verdict` whenever `policy_source == "run_policy"`. -- [x] `make lint && make test`. - -### Task 1.6 — PR list `latest_status` uses the snapshot - -**Files:** `app/runs/service.py:list_pull_requests`, `tests/test_phase3_pull_requests.py` - -- [x] Test: a PR whose latest run has a snapshot reports `latest_status == latest run's verdict` regardless of the project's legacy policy; a PR whose latest run has no snapshot falls back to the legacy evaluation. -- [x] In the fold, when `record.policy is not None`, set `latest_status = record.verdict` and skip that run in the batched `load_stored_decisions` call. -- [x] `make lint && make test`. - -### Task 1.7 — `GET /projects/{id}/policy` (D4) - -**Files:** `app/projects/router.py`, `app/projects/schemas.py`, `app/orgs/service.py` or a new `app/projects/policy_service.py`, `tests/test_phase_a3_project_update.py` (or a new `tests/test_phase17_project_policy.py`), `tests/test_phase8_route_authz.py` - -- [x] Schema `ProjectPolicyResponse`: `source: Literal["run_policy", "legacy_project_policy", "none"]`, `policy: MigrationPolicy | None`, `run: RunSummary | None` (the run it came from), `template: MigrationPolicy` (the starter, so the client makes one call). -- [x] Test: default-branch run wins over a newer non-default-branch run; with no default-branch run the newest available run wins; with no snapshot runs the legacy column is returned with `source == "legacy_project_policy"`; empty project → `none`; requires `policy:read` (route-authz test). -- [x] Query: `SELECT policy_jsonb, {RUN_COLUMNS} FROM runs WHERE project_id = :p AND status = 'available' AND deleted_at IS NULL AND policy_jsonb IS NOT NULL ORDER BY (branch = :default_branch) DESC, created_at DESC LIMIT 1`. -- [x] `make lint && make test`. - -### Task 1.8 — Freeze the legacy column: reject writes (D3) - -**Files:** `app/projects/router.py:update_project`, `app/projects/schemas.py:ProjectPatch`, `app/orgs/router.py` / `app/orgs/service.py:create_project`, `tests/test_phase_a3_project_update.py`, `tests/test_phase8_route_authz.py` - -- [x] Test: `PATCH /projects/{id}` with `migration_policy` (any value, including `null`) → 422 with detail `"migration_policy is configured in evalshift.yaml and synced on push"`; a PATCH with only `name` still works. `POST /orgs/{org}/projects` with `migration_policy` → 422 same message. -- [x] Remove `migration_policy` from `ProjectPatch` and the create payload (pydantic `extra="forbid"` on those models yields the 422; if they are not `forbid`, add an explicit check so the message is the one above). Delete `update_migration_policy` plumbing from `organizations.update_project`; keep the read path and the audit-diff for the column so history still renders. -- [x] Keep `Project.migration_policy` in the public read model — the client shows it in the legacy banner. -- [x] Remove the `policy:configure` requirement from anything policy-related that remains (there should be nothing left; `thresholds` keeps it). *(Superseded 2026-09-19 — Phase 5's `[cli]` item deleted `thresholds` outright; see the note there.)* -- [x] `make lint && make test`. - -### Task 1.9 — Adoption hint in `RunUploadResponse` (D7) - -**Files:** `app/runs/schemas.py:RunUploadResponse`, `app/runs/service.py:initiate_run` / `_existing_run_response`, `tests/test_phase4_runs_api.py` - -- [x] Test: `POST /runs` on a project with a legacy policy returns `legacy_project_policy` equal to it; on a project without one returns `null`. -- [x] Add `legacy_project_policy: dict[str, Any] | None = None`; populate from `project.migration_policy` in both the fresh and the existing-run responses. -- [x] `make lint && make test`. - -### Task 1.10 — Docs `[server]` - -**Files:** `BUNDLE_SPEC.md` (done in 1.2), `CLAUDE.md` (only if a workflow note changes), `app/policy/schemas.py` docstrings, OpenAPI descriptions - -- [x] Describe `policy_source` values and the D2 rule in the `policy-check` route docstring. -- [x] Note in `BUNDLE_SPEC.md` "Compatibility" section: bundles lacking `decision.policy` are gated on the legacy project policy or not at all. - ---- - -## Phase 2 — `[cli]` put the resolved policy into the decision and the bundle - -### Task 2.1 — `MigrationDecision.policy` (D1) - -**Files:** `src/evalshift_cli/analysis/policy.py`, `tests/unit/test_policy.py`, `tests/unit/test_analyze_command.py` - -- [x] Test: `evaluate_migration_policy(...)` returns a decision whose `policy` equals `policy.model_dump()` (all nine fields + `slices`); `inconclusive_decision(...)` returns `policy is None`; `to_dict()`/`from_dict()` round-trip both; `from_dict()` of a pre-existing `migration_decision.json` without the key still loads (`policy=None`). -- [x] Add `policy: dict[str, Any] | None = None` to the dataclass (dict, not the pydantic model, so `asdict` stays trivial). Populate in both constructors. -- [x] `analyze` already writes the decision to `migration_decision.json`; assert the key is present in the written file. -- [x] `uv run ruff check && uv run mypy && uv run pytest`. - -### Task 2.2 — Bundle carries `decision.policy` - -**Files:** `src/evalshift_cli/hosted/bundle.py`, `tests/unit/test_bundle_shape.py`, the CLI's copy of `schemas/bundle_manifest.schema.json` if it vendors one (check `grep -rn bundle_manifest.schema src tests`) - -- [x] Test: a built bundle with `migration_policy` configured has `decision.policy` with nine top-level keys and `slices`; without it, `decision.policy is None`. If the CLI validates bundles against the vendored JSON schema, copy the regenerated schema from Task 1.2 and assert the built bundle validates. -- [x] No code change should be needed beyond Task 2.1 (`decision.to_dict()` already flows through). Verify, then close. - -### Task 2.3 — `push` warnings and the adoption hint (D6, D7) - -**Files:** `src/evalshift_cli/hosted/push.py`, `src/evalshift_cli/hosted/client.py` (no change expected), `tests/unit/test_hosted_cli.py`, `tests/unit/test_push_validation.py` - -- [x] Test: pushing a bundle whose `decision.policy` is `null` prints `! this run carries no migration policy; the hosted gate will report inconclusive — add migration_policy to evalshift.yaml` (once, yellow, same style as `_warn_threshold_drift`). -- [x] Test: when the config has no `migration_policy` and the initiate response carries `legacy_project_policy`, `push` prints a `migration_policy:` YAML block containing that policy, preceded by `this project has a policy configured in the web app; move it into evalshift.yaml:`; when the config **has** one, nothing is printed even if the server sends a legacy policy. -- [x] Implement `_warn_missing_policy(console, bundle)` and `_print_legacy_policy_hint(console, config_path, response)`. Render YAML with the project's existing YAML writer (the one `init` uses via `render_minimal_config`); do not hand-format. Legacy policies have six fields; validate through `MigrationPolicy` first so the printed block is exactly what the CLI will accept. -- [x] `uv run ruff check && uv run mypy && uv run pytest`. - -### Task 2.4 — Docs `[cli]` - -**Files:** `docs/hosted.md`, `docs/configuration.md`, `README.md` (policy section, if any), `docs/llms*.txt` if the docs sync script needs re-running - -- [x] `docs/configuration.md`: `migration_policy` is the single source of truth; it is snapshotted into every pushed run; the web app shows it and cannot edit it. -- [x] `docs/hosted.md` "What `push` sends": add `decision.policy` to the block table; document the two new warnings and the adoption hint; add a troubleshooting row "`this run carries no migration policy`". -- [x] Run the docs/llms sync if the repo has one (see release notes memory: "docs llms synced"). - ---- - -## Phase 3 — `[client]` display, never edit - -### Task 3.1 — API layer - -**Files:** `src/lib/api.ts`, `src/lib/api.test.ts` (if present) - -- [x] Add `ProjectPolicy` type and `api.projectPolicy(projectId)` → `GET /projects/{id}/policy`. -- [x] Extend `MigrationPolicy` type with the three optional fields; extend `PolicyCheck.policy_source` union with `"run_policy"`. -- [x] Remove `migration_policy` from `api.updateProject`'s body type. - -### Task 3.2 — Project settings card becomes read-only (D4) - -**Files:** `src/pages/app/project/ProjectSettings.tsx`, `src/pages/app/project/policyFields.ts`, delete `src/pages/app/project/EditPolicyDialog.tsx`, `src/pages/app/project/ProjectSettings.test.tsx` - -- [x] Tests (replace the edit/reset/create suites at `ProjectSettings.test.tsx:361-510`): - - `source: "run_policy"` renders all nine budgets, the line "From run `` on ``, pushed ``" linking to the run, and no Edit/Create/Reset buttons even for an owner. - - `source: "legacy_project_policy"` renders the six budgets plus a banner "Configured in the web app. Move it into `evalshift.yaml` — editing here is no longer possible." and a "Copy as YAML" button that writes the yaml block to the clipboard (mock `navigator.clipboard`). - - `source: "none"` renders the empty state with the starter template rendered as a `migration_policy:` YAML block and a copy button; the text says the gate reports `inconclusive` until a run is pushed with a policy. - - The card no longer depends on `policy:configure`; a member sees the same content as an owner. -- [x] `policyFields.ts`: extend `POLICY_FIELDS` to nine (add `max_tool_divergence` percent, `tool_argument_drift_floor` percent, `fail_on_dropped_params` boolean → render "yes/no"); delete `validatePolicy`, `toPolicyValues`, `percentMax`, `helpText` and everything only the dialog used. Add `toYaml(policy)` (small hand-rolled renderer for this flat shape plus one level of `slices`; do not add a YAML dependency). -- [x] Delete `EditPolicyDialog.tsx`, the `"policy"` edit target, `onResetPolicy`, `confirmReset`, and the `api.policyTemplate` call (the template now arrives inside `api.projectPolicy`). -- [x] Load `api.projectPolicy` on mount and on project change; loading/error states use the page's existing `LoadingState`/`ErrorState`. -- [x] `npm run lint && npm run typecheck && npm test`. - -### Task 3.3 — Run detail Policy tab shows its policy source - -**Files:** `src/pages/app/runs/detail/tabs/PolicyTab.tsx`, `src/pages/app/runs/detail/tabs/PolicyTab.test.tsx`, `src/pages/app/runs/detail/fetchers.ts` - -- [x] Test: when `policyCheck.policy_source === "run_policy"` the tab header reads "Gated under the policy pushed with this run"; `"project_policy"` reads "Gated under the project's legacy web policy"; `"none"` reads "Not gated — no policy was pushed with this run" with a link to the settings card. -- [x] Add `fetchers.policyCheck` (wire the already-existing `api.policyCheck`, which today has no non-test caller) and render the one-line source header above the budget table. Budgets keep coming from `api.runBudgets`. -- [x] `npm run lint && npm run typecheck && npm test`. - -### Task 3.4 — Onboarding checklist copy (only if it mentions the policy) - -- [x] `grep -rn "policy" src/pages/app/onboarding src/components/*Checklist*` — if a step says "create a policy in settings", reword to "add `migration_policy` to evalshift.yaml and push". - -### Phase 3 landed — deviations - -- **3.1 is not independently green.** Dropping `migration_policy` from `api.updateProject`'s body - type breaks its only two callers, which are exactly what 3.2 deletes and rewrites, so 3.1 and - 3.2 are one commit. `api.policyTemplate` was kept (no caller, like `api.policyCheck` before - 3.3) since it is the endpoint's only client-side name. -- **`MigrationPolicy.slices` is now typed** (`Record`, new exported - type) rather than `Record` — `toYaml` needs to walk it. -- **3.3: `RunFetchers.policyCheck` is optional and absent from `sharedRunFetchers`.** The server - mounts `policy-check` under `/runs/{id}` only; there is no `/share/{token}` counterpart, so the - share surface would have pointed at a 404. The tab renders no source line there. A failed or - in-flight policy check degrades to the budget table alone rather than blanking the tab. -- **3.4 was a no-op**, verified: no onboarding or checklist copy mentions creating a policy. -- **Four files outside the task list asserted the old model** and were corrected, since Phase 1/2 - had already made them false: - - `docs/pages/MigrationPolicy.tsx` — documented the Settings create/edit dialog and claimed - "editing the policy re-decides *past* runs", which D3 reverses. Rewritten around the yaml → - push → snapshot model; `#editor`/`#reeval` replaced by `#source`/`#snapshot`/`#display`/ - `#legacy` (no inbound referrers); nav blurb at `docs/data/nav.ts` updated with it. - - `docs/pages/Verdicts.tsx` — the "server-side enforcement" callout claimed tightening a budget - can flip a stored run to FAIL. - - `app/permissionCatalog.ts` — `policy:configure` was labelled "Edit the migration policy"; per - D8 it now guards `thresholds` only. - - `app/help/topics/Baselines.tsx` — the in-app guide drew a `DialogFigure` of the deleted - "Create migration policy" dialog, and imports `POLICY_FIELDS`, so widening it to nine silently - rendered three blank inputs. The figure is now the `migration_policy:` block itself, rendered - by the same `toYaml` the settings card copies; the `policy:configure` `CannotNotice` is gone. -- **Not touched, deliberately:** `compare/data/langfuse.ts` had one stale claim (corrected); blog - posts are dated artifacts and were left alone. -- Acceptance item 2's open question — whether "the previous run's policy is still shown as - current" confuses — is answered by the card naming the run and branch each policy came from. - ---- - -## Phase 4 — `[action]` loud when ungated (D6) - -### Task 4.1 — `policy_source: none` annotation and status text - -**Files:** `scripts/evalshift_action.py`, `tests/test_evalshift_action.py`, `action.yml`, `README.md` - -- [x] Test: `_policy_gating` with `status == "inconclusive"` and `policy_source == "none"` → `should_fail False`, summary `the gate is off — no migration policy was pushed with this run; add migration_policy to evalshift.yaml`, and a `::warning::` line on stdout (workflow annotation). With `policy_source == "run_policy"` no annotation. -- [x] Test: new input `require-policy: true` makes that same case `should_fail True`, conclusion `failure`; default `false` keeps today's behaviour. -- [x] Implement: read `policy_source` from the payload; add `REQUIRE_POLICY` input plumbing next to `fail-on`; add `require-policy` to `action.yml` (`default: "false"`). -- [x] README: in the `policy` mode table add the `none` row, document `require-policy`, and update line ~200 ("the CLI, the web app and this check all enforce one policy") to say the policy comes from `evalshift.yaml` via the pushed run. -- [x] `uv run pytest` (or the repo's test command) + the pin-consistency test. - ---- - -## Phase 5 — cleanup and follow-ups (not blocking release) - -- [ ] `[server]` After 90 days with no `policy_source == "project_policy"` answers in logs: drop `projects.migration_policy_json`, `load_policy`, `evaluate_policy`'s legacy path, and `_effective_slice_policy`. Add a structlog counter now so the decision can be made from data. - **Counter done 2026-09-19** (PR #8, branch `chore/p17-phase5-cleanup`): every answer emits - `policy_check_answered` with `run_id`, `project_id` and `policy_source`, so the window is a - query over one event and the `run_policy`/`none` answers give it a denominator. The drop - itself stays open until the window is clear — earliest **2026-12-18**, counting from the - day the counter ships, not from the day it was decided. -- [x] `[cli]` Fold `thresholds` into `migration_policy` or delete it; today it is free-form and gates nothing (`docs/configuration.md:53`). - **Done 2026-09-19 — deleted outright** (maintainer's call: not folded, no deprecation - period). Branch `chore/remove-thresholds`. The field, the push sync, `_thresholds_from_config`, - `_non_empty` and `_warn_threshold_drift` are gone; a config still setting `thresholds:` now - fails to load with a message naming the removal. Breaking by the letter of SemVer, but - **released as 1.1.0, not 2.0.0** (maintainer's call 2026-09-19): the key gated nothing and - reached no further than a project-settings blob, so a major would have signalled a migration - that, for anyone who never wrote `thresholds:`, does not exist. The CHANGELOG says so at the - release heading rather than leaving it to look like an oversight. **Two follow-ups this opened:** (a) **resolved 2026-09-19 — the rule was - amended, `version:` stays `1`.** The literal marks a config that is still valid but would be - read with the wrong meaning; a removal that fails the load while naming the key is the - opposite of that, and bumping would have forced an edit on every config, including the - majority that never set `thresholds`. `docs/configuration.md`, `DOCS.md`, `llms-full.txt` and - the CHANGELOG entry now say so. (b) `[server]` `canonical_thresholds` now has no consumer and - `policy:configure` (D8) guards nothing — **scheduled 2026-09-19 as the `[server]` bullet - below.** -- [x] `[server]` Retire the thresholds plumbing the CLI no longer feeds (follow-up (b) above): - `canonical_thresholds` on the upload response, `_sync_project_thresholds`, and the - `policy:configure` requirement on `POST /runs`. A CLI older than 1.1.0 still sends - `thresholds`, so the field keeps being *accepted* — what goes is the sync, the response - field, and the permission that gated a write nothing performs any more. - **Done 2026-09-19** (PR #8). `RunCreate.thresholds` is marked deprecated and ignored; - `PATCH /projects/{id}` is the only writer left. Breaking in contract terms — - `canonical_thresholds` is gone from the `POST /runs` response — but CLI ≤ 1.0.1 reads it - through `.get` and simply skips its drift warning, and it has no other consumer. - **What this opened:** `policy:configure` now guards nothing at all. Deleting the key is a - coordinated change across three places — the server catalog (token creation validates - scopes against it), the role map served by `GET /orgs/{slug}/permissions`, and the web - app's `permissionCatalog.ts` label — so it stays defined, as its own bullet below. -- [x] `[server]` + `[client]` Delete the `policy:configure` permission. **Done 2026-09-19** — - server in PR #8, client in evalshift-client#10 (`permissionCatalog.ts` label, the - `MigrationPolicy` docs callout, the `ProjectSettings` role fixture). Verified before - deleting: `normalize_scopes` is the only validator and runs once, at mint time - (`service_accounts/service.py`), while each request merely intersects — so a token that - already carries the key keeps working and the key just stops matching. The one behaviour - change is that minting a *new* token naming it is a 422. -- [ ] `[server]` Org-level policy floor (a minimum a pushed policy cannot go below) if governance becomes a customer ask. Design only after D8's acceptance is revisited. - ---- - -## Acceptance (end-to-end, run by hand before release) - -1. Fresh project, `evalshift.yaml` with a strict `migration_policy`; `evalshift run` + `push`. Web settings card shows nine budgets "From run …". `policy-check` returns `policy_source: run_policy`, `status == verdict`. Action commit status matches the CLI's `compare --policy-gate` exit code. -2. Same project, delete `migration_policy`, push again. `push` warns; settings card still shows the *previous* run's policy as current (latest run with a snapshot) — verify this is the intended reading of D4 and adjust the card copy if it confuses; Action posts the `::warning::`; run Policy tab says "Not gated". -3. Pre-existing project with a web policy and no yaml policy: `push` prints the YAML hint; settings card shows the legacy banner; `policy-check` for old runs still answers `project_policy`; `PATCH` with `migration_policy` → 422. -4. Old CLI (pre-Phase-2) bundle against new server: finalize succeeds; run has `policy_jsonb NULL`; behaviour identical to (3). diff --git a/docs/superpowers/plans/2026-09-21-stale-docs-cleanup.md b/docs/superpowers/plans/2026-09-21-stale-docs-cleanup.md deleted file mode 100644 index 5269196..0000000 --- a/docs/superpowers/plans/2026-09-21-stale-docs-cleanup.md +++ /dev/null @@ -1,1120 +0,0 @@ -# Stale Docs Cleanup Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. - -**Goal:** Remove five verified stale or false claims from the shipped docs of `evalshift-action` and `evalshift-cli`, and leave a regression guard behind each one so the same rot cannot recur silently. - -**Architecture:** Every fix is a docs edit plus a test that fails on the stale text. Three of the five files are mirrored to the marketing site by a plain `cp` (`npm run sync:llms`), so all source-repo edits land first and a single sync commit re-publishes them last. No product code changes except `scripts/bump_cli_pin.py`, which gains one more self-syncing version site. - -**Tech Stack:** Markdown + plain-text docs; `pytest` + `ruff` (both Python repos); `mypy --strict` (CLI only); `vitest` + `tsc` + `eslint` (client). Version-drift guards follow the existing `PIN_SITES` / `ACTION_VERSION_SITES` regex-table pattern in `evalshift-action/scripts/bump_cli_pin.py`. - -**Spec:** This plan is its own spec. Every claim below was verified against source on 2026-09-21; the "Evidence" line under each task records what was checked and where. - -## Global Constraints - -- Three independent git repos under a non-git parent `/home/lukas/repos/evalshift`. **Always pass `-C `** to git. All three are on `main`, clean, and in sync with `origin` as of 2026-09-21. -- This plan file lives in `evalshift-cli/docs/superpowers/plans/`, which matches a `.gitignore` pattern. Committing it needs **`git add -f`**. -- Conventional Commits with scopes, matching existing history (`fix(docs): ...`, `chore(pin): ...`, `docs: ...`). -- Branch per task, PR per task. Never commit directly to `main`. -- **Gates — `evalshift-action`:** `uv run pytest` · `uv run ruff check .` · `uv run pip-audit` -- **Gates — `evalshift-cli`:** `ruff check .` · `ruff format --check .` · `mypy --strict src/evalshift_cli` · `pytest` (2276 tests, ~19s) -- **Gates — `evalshift-client`:** `npm run lint` · `npm run typecheck` · `npm run build` -- `evalshift-action/llms-full.txt`, `evalshift-cli/llms-full.txt` and `evalshift-sdk/llms-full.txt` are copied verbatim into `evalshift-client/public/` by `npm run sync:llms`. **Never hand-edit the `public/*-llms-full.txt` copies** — fix the source repo, then sync (Task 7). -- **Tasks run in order and each merges before the next starts.** Tasks 2, 3 and 4 all append to the same new file, `evalshift-cli/tests/unit/test_docs_currency.py`, and each branches from a freshly pulled `main`; the expected pass counts quoted in their verification steps assume the earlier task has merged. Tasks 1 and 5 (action) are independent of 2/3/4 (CLI) and may be reordered between repos, but Task 7 is last. -- Do not bump any package version. These are docs-only changes; the action's `pyproject.toml` version stays `0.5.1` and the CLI's stays `1.1.0`. - ---- - -### Task 1: Delete the dead `thresholds:` / `policy:configure` flow from the Action docs - -**Why this is first:** it is the only issue where the docs instruct a user to do something that is now impossible. A config carrying `thresholds:` fails to load outright, and the permission the text cites does not exist. The text also ships to AI agents at `evalshift.dev/ci-llms-full.txt`. - -**Evidence (verified 2026-09-21):** -- `evalshift-server/app/authz/permissions.py:26` — comment: *"There is no `policy:configure`. It guarded one write — the thresholds sync a run upload used to perform — and went with it (P17 Phase 5)."* -- `evalshift-server/tests/test_phase8_authz.py:164` — asserts `"policy:configure" not in authz.ALL_PERMISSIONS`. -- `evalshift-cli/src/evalshift_cli/config/models.py:738` — a config with `thresholds:` raises *"`thresholds` was removed: it was free-form and gated nothing."* -- The string `Project owner role required` appears **nowhere** in `evalshift-server` (only `Organization owner role required`, in `app/orgs/service.py`, for org-level operations) and nowhere in the action's own `scripts/`. The troubleshooting entry documents an error the product cannot emit, so it is deleted rather than reworded. -- The same README contradicts itself at `README.md:200-204`: *"the action never re-implements a threshold."* - -**Files:** -- Modify: `evalshift-action/README.md:110-121` -- Modify: `evalshift-action/DOCS.md:176-188`, `evalshift-action/DOCS.md:870-874` -- Modify: `evalshift-action/llms-full.txt:492-498`, `evalshift-action/llms-full.txt:693-694` -- Create: `evalshift-action/tests/test_docs_currency.py` - -**Interfaces:** -- Consumes: nothing from earlier tasks. -- Produces: `tests/test_docs_currency.py::RETIRED_TERMS`, a `tuple[str, ...]` of exact substrings that must not appear in the action's three prose files. Task 5 does not touch it; later doc retirements append to it. - -- [ ] **Step 1: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-action checkout -b fix/drop-dead-thresholds-docs -``` - -- [ ] **Step 2: Write the failing test** - -Create `evalshift-action/tests/test_docs_currency.py`: - -```python -"""The docs must not describe flows the product has removed. - -`thresholds:` was deleted from `evalshift.yaml` in CLI 1.1.0 -- a config that -still sets it fails to load, by name -- and `policy:configure` was deleted from -the server's permission catalog along with the single write it guarded. This -repo's README, DOCS.md and llms-full.txt described both for five releases after -they were gone, and llms-full.txt is copied verbatim to -https://www.evalshift.dev/ci-llms-full.txt, so the stale instructions were being -served to coding agents as current guidance. - -Nothing else noticed, because every existing docs test checks a version literal -rather than a claim. This is the tripwire for claims: a term retired from the -product must not reappear in prose. -""" - -from __future__ import annotations - -import pytest - -from _manifest import REPO_ROOT - -#: Exact substrings that named a removed feature. Retiring something else from -#: the product? Append it here in the same commit that removes it. -RETIRED_TERMS: tuple[str, ...] = ( - "policy:configure", - "thresholds:", -) - -PROSE_FILES: tuple[str, ...] = ("README.md", "DOCS.md", "llms-full.txt") - - -@pytest.mark.parametrize("name", PROSE_FILES) -@pytest.mark.parametrize("term", RETIRED_TERMS) -def test_prose_does_not_describe_a_removed_feature(name: str, term: str) -> None: - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert term not in text, f"{name} still describes the removed {term!r}" -``` - -- [ ] **Step 3: Run the test to verify it fails** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest tests/test_docs_currency.py -v -``` - -Expected: 6 tests, all 6 FAIL — each with ` still describes the removed ''`. - -- [ ] **Step 4: Fix `README.md:110-121`** - -Replace this block: - -```markdown -Two things a correctly-scoped key deliberately cannot do: - -- **Auto-create the hosted project.** `project:create` is an owner permission and - a service account is never an owner. Create the project once in the web app and - set `create-project: false`, so a wrong project slug fails as a missing project - rather than looking like a credential problem. -- **Rewrite the project's gating thresholds.** `evalshift push` sends the - `thresholds:` block from your `evalshift.yaml` whenever one is present, and - rewriting a project's gating policy needs `policy:configure` — also owner-only. - Keep thresholds canonical in the web app and out of the config the CI job runs, - or the push fails with `Project owner role required`. -``` - -with: - -```markdown -One thing a correctly-scoped key deliberately cannot do: - -- **Auto-create the hosted project.** `project:create` is an owner permission and - a service account is never an owner. Create the project once in the web app and - set `create-project: false`, so a wrong project slug fails as a missing project - rather than looking like a credential problem. - -The gate itself needs no extra scope. Your `migration_policy` travels inside the -run bundle that `run:create` already uploads, so a member-role key both pushes -the policy and gates on it — see [`fail-on` modes](#fail-on-modes). -``` - -- [ ] **Step 5: Fix `DOCS.md:176-188`** - -Replace this block: - -```markdown -Two consequences of a correctly-scoped key, both by design: -``` - -with: - -```markdown -One consequence of a correctly-scoped key, by design: -``` - -Then delete this bullet entirely: - -```markdown -- **It cannot rewrite the project's gating thresholds.** `evalshift push` sends the - `thresholds:` block from your `evalshift.yaml` whenever one is present, and rewriting a - project's gating policy needs the owner-only `policy:configure`. Keep thresholds canonical in - the web app and out of the config the CI job runs, or the push fails with - `Project owner role required`. -``` - -and insert, as a new paragraph after the remaining bullet and before the `A denial is self-diagnosing:` paragraph: - -```markdown -The gate needs no scope beyond these two. The `migration_policy` block rides inside the run -bundle `run:create` already uploads, so a member-role key both pushes the policy and gates on -it. -``` - -- [ ] **Step 6: Delete the phantom troubleshooting entry at `DOCS.md:870-874`** - -Delete the heading and its body, leaving the surrounding entries untouched: - -```markdown -### `Project owner role required` - -`evalshift push` tried to rewrite the project's gating thresholds, which needs the owner-only -`policy:configure`. Remove the `thresholds:` block from the config the CI job runs and manage -thresholds in the web app. - -``` - -- [ ] **Step 7: Fix `llms-full.txt:492-498`** - -Replace this block: - -```text -Out of reach for a scoped key, by design: -- Auto-creating the hosted project (`project:create` is owner-only). Create the project in the - web app and set `create-project: false`. -- Rewriting gating thresholds (`policy:configure` is owner-only). `evalshift push` sends the - `thresholds:` block from `evalshift.yaml` whenever one is present, so keep thresholds - canonical in the web app and out of the CI config, or push fails - `Project owner role required`. -``` - -with: - -```text -Out of reach for a scoped key, by design: -- Auto-creating the hosted project (`project:create` is owner-only). Create the project in the - web app and set `create-project: false`. -Gating needs no further scope: `migration_policy` rides in the bundle `run:create` uploads, so a -member-role key both pushes the policy and gates on it. -``` - -- [ ] **Step 8: Delete the phantom error line at `llms-full.txt:693-694`** - -Delete exactly these two lines: - -```text -`Project owner role required` → push tried to rewrite gating thresholds (`policy:configure`, -owner-only); drop `thresholds:` from the CI config. -``` - -- [ ] **Step 9: Run the test to verify it passes** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest tests/test_docs_currency.py -v -``` - -Expected: 6 passed. - -- [ ] **Step 10: Confirm nothing else mentions the retired terms** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && grep -rn "thresholds\|policy:configure" README.md DOCS.md llms-full.txt action.yml -``` - -Expected: **no output.** (Before this task the same command printed 13 lines.) `README.md:200-204` says "never re-implements a threshold" — singular, no colon — and is correct; it is not matched by `thresholds`. - -- [ ] **Step 11: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest && uv run ruff check . -``` - -Expected: all tests pass, ruff clean. - -- [ ] **Step 12: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action -git add README.md DOCS.md llms-full.txt tests/test_docs_currency.py -git commit -m "fix(docs): stop documenting the removed thresholds gate - -CLI 1.1.0 deleted \`thresholds:\` from evalshift.yaml and the server deleted -\`policy:configure\` with the one write it guarded, but all three prose files -still told users to keep thresholds in the web app to avoid a -\`Project owner role required\` error the server never emits. The same README -already said the action never re-implements a threshold. - -llms-full.txt is mirrored to evalshift.dev/ci-llms-full.txt, so this was being -served to coding agents as current guidance. test_docs_currency.py fails on any -future mention." -git push -u origin fix/drop-dead-thresholds-docs -gh pr create --fill -``` - ---- - -### Task 2: Advertise `compare --push`, not the hidden legacy `all --push` - -**Evidence (verified 2026-09-21):** `compare` is the registered command (`src/evalshift_cli/cli/main.py:89`). `all` is registered at `main.py:92` with `hidden=True` and prints a rename notice to stderr (`cli/commands/compare.py:322-339`); `LEGACY_COMMAND_NAME = "all"` at `compare.py:319`. The alias is permanent, so nothing is broken — but nine doc sites advertise the hidden name as the thing to type. - -**Scope note — five sites must NOT change.** These correctly describe `all` as a legacy alias, or are unrelated: -- `llms-full.txt:75` — "Formerly `all`: hidden alias, still works." -- `llms-full.txt:417` — "`all` -> `compare` in 1.0.0" -- `DOCS.md:353` — a *slice* named `all`. Unrelated to the command. -- `DOCS.md:819` — "Formerly `all`; that name is hidden but still works" -- `DOCS.md:880` — "`evalshift all` became `evalshift compare` in 1.0.0" - -The literal `all --push` appears at exactly nine sites and at none of the five above, so a global substring replace is safe. Step 5 proves it. - -**Files:** -- Modify: `evalshift-cli/README.md:216`, `:306` -- Modify: `evalshift-cli/DOCS.md:720` -- Modify: `evalshift-cli/llms-full.txt:938`, `:1091` -- Modify: `evalshift-cli/docs/faq.md:15` -- Modify: `evalshift-cli/docs/hosted.md:253` -- Modify: `evalshift-cli/docs/configuration.md:70` -- Modify: `evalshift-cli/docs/index.md:89` -- Create: `evalshift-cli/tests/unit/test_docs_currency.py` - -**Interfaces:** -- Consumes: nothing from Task 1 (different repo). -- Produces: `tests/unit/test_docs_currency.py::PROSE_FILES`, a `tuple[str, ...]` of the CLI's seven prose files relative to the repo root. Tasks 3 and 4 add their own test functions to this same module and reuse `PROSE_FILES`. - -- [ ] **Step 1: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli checkout -b fix/advertise-compare-not-all -``` - -- [ ] **Step 2: Write the failing test** - -Create `evalshift-cli/tests/unit/test_docs_currency.py`: - -```python -"""The docs must advertise the current command name, not a hidden alias. - -`evalshift all` became `evalshift compare` in 1.0.0. The old name stays -registered forever -- scaffolded EVALSHIFT.md files in user repos reference it, -and removing it would itself be breaking -- but it is `hidden=True` and prints a -rename notice. Nine doc sites still told readers to type it, so the docs taught -a name that `evalshift --help` does not list. - -Prose that *describes* the alias ("formerly `all`", "`all` -> `compare` in -1.0.0") is correct and deliberately not matched here: the assertion is on the -exact string `all --push`, which only ever appeared as advertised usage. -""" - -from __future__ import annotations - -from pathlib import Path - -import pytest - -REPO_ROOT = Path(__file__).resolve().parents[2] - -#: Every file that documents the CLI in prose, relative to the repo root. -PROSE_FILES: tuple[str, ...] = ( - "README.md", - "DOCS.md", - "llms-full.txt", - "docs/faq.md", - "docs/hosted.md", - "docs/configuration.md", - "docs/index.md", -) - - -@pytest.mark.parametrize("name", PROSE_FILES) -def test_prose_advertises_compare_not_the_hidden_alias(name: str) -> None: - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert "all --push" not in text, ( - f"{name} advertises the hidden `all` alias; write `compare --push`" - ) -``` - -- [ ] **Step 3: Run the test to verify it fails** - -```bash -pytest tests/unit/test_docs_currency.py -v --no-cov -``` - -Expected: 7 tests, **7 failed** — `README.md`, `DOCS.md`, `llms-full.txt`, `docs/faq.md`, `docs/hosted.md`, `docs/configuration.md`, `docs/index.md` each report the alias. - -- [ ] **Step 4: Replace all nine occurrences** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -sed -i 's/all --push/compare --push/g' \ - README.md DOCS.md llms-full.txt \ - docs/faq.md docs/hosted.md docs/configuration.md docs/index.md -``` - -- [ ] **Step 5: Verify the replacement hit nine sites and spared the five alias descriptions** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -echo "--- should print 9 ---" -grep -rc "compare --push" README.md DOCS.md llms-full.txt docs/*.md | grep -v ':0' | awk -F: '{s+=$2} END {print s}' -echo "--- should print nothing ---" -grep -rn "all --push" README.md DOCS.md llms-full.txt docs/ -echo "--- these 5 must still be present ---" -grep -n "Formerly \`all\`" llms-full.txt DOCS.md -grep -n "\`all\` -> \`compare\`\|evalshift all\` became" llms-full.txt DOCS.md -grep -n "\`all\` and any slice" DOCS.md -``` - -Expected: `9`; no `all --push` hits; and five surviving lines — `llms-full.txt:75`, `DOCS.md:819`, `llms-full.txt:417`, `DOCS.md:880`, `DOCS.md:353`. - -- [ ] **Step 6: Run the test to verify it passes** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli && pytest tests/unit/test_docs_currency.py -v --no-cov -``` - -Expected: 7 passed. - -- [ ] **Step 7: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -ruff check . && ruff format --check . && mypy --strict src/evalshift_cli && pytest -``` - -Expected: ruff clean, mypy clean, 2283 passed. - -- [ ] **Step 8: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -git add README.md DOCS.md llms-full.txt docs/ tests/unit/test_docs_currency.py -git commit -m "docs: advertise \`compare --push\`, not the hidden \`all\` alias - -\`all\` became \`compare\` in 1.0.0. The alias stays registered forever, but it -is hidden from --help and prints a rename notice, so nine doc sites were -teaching a name the CLI does not advertise. Prose that describes the alias as -an alias is unchanged. - -test_docs_currency.py fails on any future \`all --push\` in prose." -git push -u origin fix/advertise-compare-not-all -gh pr create --fill -``` - ---- - -### Task 3: Stop capping the provider list at three - -**Evidence (verified 2026-09-21):** the CLI dispatches every model call through LiteLLM (`src/evalshift_cli/models/client.py:35`, `litellm.acompletion`). `src/evalshift_cli/models/registry.py:10` states outright: *"the authority on whether a model is callable is **LiteLLM**, not us"*, and `resolve_model` falls back to prefix inference so an unregistered id still dispatches. `Provider = Literal["anthropic", "openai", "google", "other"]` (`registry.py:36`) is a *curated registry* for `doctor` output and report rendering, not a capability boundary. The repo's own `docs/faq.md:50` already answers "which models?" with **"Anything LiteLLM supports."** - -Four doc sites present the three names as the boundary, contradicting `faq.md:50`. - -**Client scope — decided 2026-09-21.** `evalshift-client` repeats the triple at three sites. The maintainer chose **the two docs mirrors only**: `src/pages/docs/pages/Faq.tsx:21` and `src/pages/docs/pages/WhatGetsUploaded.tsx:94`, which hand-mirror `docs/faq.md` and `docs/hosted.md` and would otherwise contradict the pages they mirror. `src/pages/landing/sections/Hero.tsx:113` ("Works with Anthropic, OpenAI and Google models.", asserted by `src/pages/landing/Landing.test.tsx:253`) stays as marketing copy — **do not touch it**. Those two client edits are executed in **Task 7**, which already owns this plan's client-repo branch; this task stays CLI-only. - -**Files:** -- Modify: `evalshift-cli/README.md:301-303` -- Modify: `evalshift-cli/docs/index.md:86-88` -- Modify: `evalshift-cli/docs/faq.md:5-8` -- Modify: `evalshift-cli/docs/hosted.md:246-247` -- Modify: `evalshift-cli/tests/unit/test_docs_currency.py` (add one test) - -**Interfaces:** -- Consumes: `tests/unit/test_docs_currency.py::PROSE_FILES` and `REPO_ROOT` from Task 2. -- Produces: nothing later tasks depend on. - -- [ ] **Step 1: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli checkout main && git -C /home/lukas/repos/evalshift/evalshift-cli pull -git -C /home/lukas/repos/evalshift/evalshift-cli checkout -b docs/provider-breadth -``` - -- [ ] **Step 2: Write the failing test** - -Append to `evalshift-cli/tests/unit/test_docs_currency.py`: - -```python -#: Files that describe *which providers work*, as opposed to naming three as -#: examples. Each must name LiteLLM, because LiteLLM is the actual boundary: -#: `models/registry.py` says so in its module docstring, and `docs/faq.md` -#: already answers "which models?" with "Anything LiteLLM supports." -PROVIDER_SCOPE_FILES: tuple[str, ...] = ( - "README.md", - "docs/index.md", - "docs/faq.md", - "docs/hosted.md", -) - - -@pytest.mark.parametrize("name", PROVIDER_SCOPE_FILES) -def test_provider_scope_is_not_capped_at_three(name: str) -> None: - """The curated registry has three entries; the CLI calls far more than three. - - `Provider` is a Literal of three names plus "other" because those three have - pricing tables and env-var mappings worth curating. Every call still goes - through `litellm.acompletion`, and `resolve_model` never raises -- an - unregistered id is dispatched with a prefix-inferred provider. Prose that - lists the three without naming LiteLLM reads as a compatibility list and - undersells the tool. - """ - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert "LiteLLM" in text, f"{name} scopes providers without naming LiteLLM" -``` - -- [ ] **Step 3: Run the test to verify it fails** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli && pytest tests/unit/test_docs_currency.py -k provider_scope -v --no-cov -``` - -Expected: 4 tests, **3 failed** (`README.md`, `docs/index.md`, `docs/hosted.md`) and **1 passed** (`docs/faq.md` — it already names LiteLLM at line 50, which is exactly the contradiction being fixed). - -- [ ] **Step 4: Fix `README.md:301-303`** - -Replace: - -```markdown -Your prompts and suite stay local for `doctor`, `run`, `evaluate`, `analyze`, -and `report`. The only outbound calls in local mode are to the LLM providers -you configure (Anthropic, OpenAI, Google) using your own API keys. -``` - -with: - -```markdown -Your prompts and suite stay local for `doctor`, `run`, `evaluate`, `analyze`, -and `report`. The only outbound calls in local mode are to the LLM providers you -configure — any provider LiteLLM supports, called with your own API keys. -Anthropic, OpenAI and Google ids additionally get a curated pricing and -capability entry; everything else is passed through with the provider inferred -from the id. -``` - -- [ ] **Step 5: Fix `docs/index.md:86-88`** - -Replace: - -```markdown -The only outbound calls are to the LLM providers you configure (Anthropic, -OpenAI, Google) using your own API keys. `bundle` packages artifacts locally; -hosted upload happens only when you run `push` or `compare --push`. -``` - -with: - -```markdown -The only outbound calls are to the LLM providers you configure — any provider -LiteLLM supports — using your own API keys. `bundle` packages artifacts locally; -hosted upload happens only when you run `push` or `compare --push`. -``` - -(Note: `compare --push` here assumes Task 2 has merged. If Task 3 runs first, that line still reads `all --push` — leave it alone and let Task 2 fix it.) - -- [ ] **Step 6: Fix `docs/faq.md:5-8`** - -Replace: - -```markdown -**Not during local runs.** `doctor`, `run`, `evaluate`, `analyze`, and -`report` operate locally. Every provider API call goes directly from your -machine to the LLM provider you configured (Anthropic, OpenAI, Google) -using your own API keys. -``` - -with: - -```markdown -**Not during local runs.** `doctor`, `run`, `evaluate`, `analyze`, and -`report` operate locally. Every provider API call goes directly from your -machine to the LLM provider you configured — any provider LiteLLM supports — -using your own API keys. -``` - -- [ ] **Step 7: Fix `docs/hosted.md:246-247`** - -Replace: - -```markdown -1. **Your model providers** (Anthropic, OpenAI, Google — whichever you - configure), using your own API keys: `run` sends the rendered prompts and -``` - -with: - -```markdown -1. **Your model providers** (whichever you configure — any provider LiteLLM - supports), using your own API keys: `run` sends the rendered prompts and -``` - -- [ ] **Step 8: Run the test to verify it passes** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli && pytest tests/unit/test_docs_currency.py -v --no-cov -``` - -Expected: 11 passed (7 from Task 2 + 4 here). - -- [ ] **Step 9: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -ruff check . && ruff format --check . && mypy --strict src/evalshift_cli && pytest -``` - -Expected: all clean. - -- [ ] **Step 10: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -git add README.md docs/ tests/unit/test_docs_currency.py -git commit -m "docs: name LiteLLM as the provider boundary, not three brands - -models/registry.py says LiteLLM is the authority on whether a model is callable, -resolve_model never raises, and docs/faq.md already answers 'which models?' with -'Anything LiteLLM supports.' Four other sites listed Anthropic/OpenAI/Google as -though that were the compatibility list -- which contradicts faq.md and -undersells the tool. The three keep a curated pricing and capability entry; that -is what the Literal is for. - -Known drift: evalshift-client's Faq.tsx and WhatGetsUploaded.tsx hand-mirror -these two pages and still say the old thing." -git push -u origin docs/provider-breadth -gh pr create --fill -``` - ---- - -### Task 4: Quote the placeholder `init` actually writes - -**Evidence (verified 2026-09-21):** `src/evalshift_cli/cli/commands/init.py:116` contains `content: "{{input}}"`, but `_MINIMAL_YAML_BODY` is a `str.format` template rendered at `init.py:174` — the doubled brace is an escape, so the `evalshift.yaml` on disk contains `{input}`. The existing test `tests/unit/test_init.py::TestInitHappy::test_written_config_parses_via_load_config` already asserts `prompt.content == "{input}"`. - -Three doc sites quote the escaped source form and so tell readers — and agents — to write a placeholder the templating engine will never expand. The claim as reported named only `llms-full.txt`; two more were found. - -**Files:** -- Modify: `evalshift-cli/llms-full.txt:450` -- Modify: `evalshift-cli/DOCS.md:440` -- Modify: `evalshift-cli/docs/configuration.md:286` -- Modify: `evalshift-cli/tests/unit/test_docs_currency.py` (add one test) - -**Interfaces:** -- Consumes: `PROSE_FILES` and `REPO_ROOT` from Task 2. -- Produces: nothing later tasks depend on. - -- [ ] **Step 1: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli checkout main && git -C /home/lukas/repos/evalshift/evalshift-cli pull -git -C /home/lukas/repos/evalshift/evalshift-cli checkout -b fix/docs-quote-rendered-placeholder -``` - -- [ ] **Step 2: Write the failing test** - -Append to `evalshift-cli/tests/unit/test_docs_currency.py`: - -```python -@pytest.mark.parametrize("name", PROSE_FILES) -def test_prose_quotes_the_rendered_placeholder(name: str) -> None: - """Docs must quote the config `init` writes, not the format template. - - `_MINIMAL_YAML_BODY` in `cli/commands/init.py` is passed through - `str.format`, so its literal `{{input}}` is a brace escape that renders as - `{input}` on disk -- which is what `test_init.py` asserts the loaded config - contains. Three doc sites copied the escaped source form verbatim, which - reads as instructions to write a placeholder `templating.py` will never - expand. - """ - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert "{{input}}" not in text, ( - f"{name} quotes the escaped `{{{{input}}}}`; `init` writes `{{input}}`" - ) -``` - -- [ ] **Step 3: Run the test to verify it fails** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli && pytest tests/unit/test_docs_currency.py -k rendered_placeholder -v --no-cov -``` - -Expected: 7 tests, **3 failed** — `DOCS.md`, `llms-full.txt`, `docs/configuration.md`. - -- [ ] **Step 4: Fix all three sites** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -sed -i 's/{{input}}/{input}/g' llms-full.txt DOCS.md docs/configuration.md -``` - -- [ ] **Step 5: Verify each site now reads correctly** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -grep -n "{input}" llms-full.txt DOCS.md docs/configuration.md -``` - -Expected, exactly: -- `llms-full.txt:123` — `(id: replay, detection: manual, content: "{input}", variables: [input])` *(already correct before this task)* -- `llms-full.txt:450` — ``# `replay` prompt is a passthrough: content "{input}" echoes the promoted capture.`` -- `llms-full.txt:1110` — `content: "{input}"` *(already correct before this task)* -- `DOCS.md:440` — ``(`content: "{input}"`)`` -- `docs/configuration.md:286` — ``(`content: "{input}"`) is a passthrough`` - -- [ ] **Step 6: Run the test to verify it passes** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli && pytest tests/unit/test_docs_currency.py -v --no-cov -``` - -Expected: 18 passed (7 + 4 + 7). - -- [ ] **Step 7: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -ruff check . && ruff format --check . && mypy --strict src/evalshift_cli && pytest -``` - -Expected: all clean. `test_init.py::test_written_config_parses_via_load_config` must still pass — it asserts the rendered form and is the reason this fix is in this direction. - -- [ ] **Step 8: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -git add llms-full.txt DOCS.md docs/configuration.md tests/unit/test_docs_currency.py -git commit -m "fix(docs): quote the placeholder init writes, not the format escape - -init.py's _MINIMAL_YAML_BODY goes through str.format, so the literal -\`{{input}}\` in the source is a brace escape and the scaffolded evalshift.yaml -contains \`{input}\` -- as test_init.py already asserts. Three doc sites copied -the escaped form, telling readers to write a placeholder templating.py never -expands, while two other sites in the same file showed the correct one." -git push -u origin fix/docs-quote-rendered-placeholder -gh pr create --fill -``` - ---- - -### Task 5: Make the example `@vX.Y.Z` tag self-syncing - -**Evidence (verified 2026-09-21):** the "pin to an exact tag" example says `@v0.3.0` at `README.md:333`, `DOCS.md:944` and `llms-full.txt:721`. The action's current release is **v0.5.1** (`pyproject.toml:3`; `v0.5.1` exists on `origin`). Nothing breaks — it is illustrative syntax — but it reads as a recommendation two minors stale, and it went stale for the same reason the version headers did: nothing bumped it and nothing checked it. - -This repo already solved that class of problem twice. `scripts/bump_cli_pin.py` holds `PIN_SITES` (the CLI pin, sourced from `action.yml`) and `ACTION_VERSION_SITES` (the action's own version, sourced from `pyproject.toml`), and `tests/test_pin_consistency.py` reads the same tables to fail on any stale literal. This task adds a third table rather than hand-editing three files, so the example cannot drift again. - -**Files:** -- Modify: `evalshift-action/scripts/bump_cli_pin.py` (add `EXAMPLE_TAG_SITES`; extend `sync_action_version`) -- Modify: `evalshift-action/tests/test_pin_consistency.py` (add two tests) -- Modify: `evalshift-action/README.md:333`, `DOCS.md:944`, `llms-full.txt:721` (rewritten by the script, not by hand) - -**Interfaces:** -- Consumes: `bump_cli_pin.VERSION`, `bump_cli_pin.replace_pins`, `bump_cli_pin.find_pins`, `bump_cli_pin.current_action_version`, `_manifest.REPO_ROOT` — all already exist. -- Produces: `bump_cli_pin.EXAMPLE_TAG_SITES: Mapping[str, tuple[str, ...]]`, keyed by the same three filenames. `sync_action_version(root) -> list[Path]` keeps its signature and return type; it now also rewrites these sites. - -- [ ] **Step 1: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-action checkout main && git -C /home/lukas/repos/evalshift/evalshift-action pull -git -C /home/lukas/repos/evalshift/evalshift-action checkout -b chore/self-syncing-example-tag -``` - -- [ ] **Step 2: Write the failing tests** - -Append to `evalshift-action/tests/test_pin_consistency.py`, and add `EXAMPLE_TAG_SITES` to the existing `from bump_cli_pin import ...` line: - -```python -def test_example_tag_sites_cover_the_documented_files() -> None: - assert set(EXAMPLE_TAG_SITES) == {"README.md", "DOCS.md", "llms-full.txt"} - - -@pytest.mark.parametrize("name", sorted(EXAMPLE_TAG_SITES)) -def test_every_example_tag_matches_pyproject(name: str) -> None: - """The "pin to an exact tag" example must name a tag that exists and is current. - - It sat at v0.3.0 across five releases -- v0.3.x through v0.5.1 -- because it - was hand-written prose that no bump touched. Advice to pin is advice to pin - to *something*; two minors behind, the example reads as the recommendation. - """ - released = current_action_version(REPO_ROOT) - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - stale = [ - found for found in find_pins(text, EXAMPLE_TAG_SITES[name], label=name) if found != released - ] - - assert stale == [], ( - f"{name}'s example tag is @v{sorted(set(stale))}; pyproject.toml says {released}" - ) -``` - -- [ ] **Step 3: Run the tests to verify they fail** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest tests/test_pin_consistency.py -v -``` - -Expected: collection ERROR — `ImportError: cannot import name 'EXAMPLE_TAG_SITES' from 'bump_cli_pin'`. - -- [ ] **Step 4: Add the table to `scripts/bump_cli_pin.py`** - -Insert immediately after the `ACTION_VERSION_SITES` block: - -```python -# The "pin to an exact tag" example in the versioning prose. Same source of truth as -# ACTION_VERSION_SITES -- pyproject.toml -- in a different shape: a `@vX.Y.Z` git tag -# rather than a bare version. It sat at `@v0.3.0` from v0.3.x through v0.5.1 because -# no bump touched it and no test read it. Advice to pin has to name a tag that exists. -EXAMPLE_TAG_SITES: Mapping[str, tuple[str, ...]] = { - "README.md": (rf"exact tag such as `@v{VERSION}`",), - "DOCS.md": (rf"exact tag such as `@v{VERSION}`",), - "llms-full.txt": (rf"^`@v0` tracks the latest v0\.x\. `@v{VERSION}` pins exactly\.",), -} -``` - -- [ ] **Step 5: Extend `sync_action_version` to rewrite them** - -Replace the body of `sync_action_version` with: - -```python -def sync_action_version(root: Path = REPO_ROOT) -> list[Path]: - """Rewrite every advertised action version to match ``pyproject.toml``. - - Covers both shapes the version is written in: the prose version headers - (``ACTION_VERSION_SITES``) and the ``@vX.Y.Z`` example tag - (``EXAMPLE_TAG_SITES``). Call this AFTER ``pyproject.toml`` is written, so - both follow the bump. Returns the files actually changed, in order, deduped. - """ - released = current_action_version(root) - changed: list[Path] = [] - for table in (ACTION_VERSION_SITES, EXAMPLE_TAG_SITES): - for name, patterns in table.items(): - path = root / name - before = path.read_text(encoding="utf-8") - after = replace_pins(before, patterns, released, label=name) - if after == before: - continue - path.write_text(after, encoding="utf-8") - # Both tables name the same three files, so a path can already be here. - if path not in changed: - changed.append(path) - return changed -``` - -Also update the module docstring's last paragraph, replacing `` ``ACTION_VERSION_SITES`` enumerates those headers and `` … with: - -``` -``ACTION_VERSION_SITES`` enumerates those headers, ``EXAMPLE_TAG_SITES`` the ``@vX.Y.Z`` -example tag that drifted the same way, and ``sync_action_version`` rewrites both from -``pyproject.toml`` on every bump, so none of them can disagree. -``` - -- [ ] **Step 6: Run the tests to verify they now fail on the stale literal (not on the import)** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest tests/test_pin_consistency.py -v -``` - -Expected: `test_example_tag_sites_cover_the_documented_files` PASSES; the three `test_every_example_tag_matches_pyproject` cases FAIL with ``example tag is @v['0.3.0']; pyproject.toml says 0.5.1``. - -- [ ] **Step 7: Rewrite the three docs by running the syncer** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action -uv run python -c " -import sys; sys.path.insert(0, 'scripts') -from bump_cli_pin import REPO_ROOT, sync_action_version -for p in sync_action_version(REPO_ROOT): print(p.relative_to(REPO_ROOT)) -" -``` - -Expected output: `README.md`, `DOCS.md`, `llms-full.txt`. - -- [ ] **Step 8: Verify the edit hit only the example tag** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && git diff --stat && git diff -U0 | grep '^[-+]' | grep -v '^[-+][-+]' -``` - -Expected: exactly three changed lines, each `v0.3.0` → `v0.5.1`. Every `uses: babaliauskas/evalshift-action@v0` line must be **unchanged** — the floating major tag is correct and the regexes are scoped to the prose sentence, not to `@v`. - -- [ ] **Step 9: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action && uv run pytest && uv run ruff check . -``` - -Expected: all pass, including the pre-existing `test_bump_cli_pin.py` suite. - -- [ ] **Step 10: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-action -git add scripts/bump_cli_pin.py tests/test_pin_consistency.py README.md DOCS.md llms-full.txt -git commit -m "chore(docs): sync the example \`@vX.Y.Z\` tag from pyproject - -The 'pin to an exact tag' example said @v0.3.0 from v0.3.x through v0.5.1 -- -hand-written prose no bump touched and no test read, which is exactly how the -version headers drifted before PR #13. Rather than edit it again, add -EXAMPLE_TAG_SITES alongside ACTION_VERSION_SITES so sync_action_version rewrites -it on every bump and test_pin_consistency fails on any stale literal. - -The floating \`@v0\` in the usage examples is unchanged -- it is correct." -git push -u origin chore/self-syncing-example-tag -gh pr create --fill -``` - ---- - -### Task 6: Make the Status section's claims checkable — **needs a decision first** - -Two claims sit in `evalshift-cli/README.md:67-70`. They are different kinds of problem and only one has a mechanical fix. - -**6a — the coverage number is wrong and unguarded.** README says *"the test suite covers 92% of the source."* Measured on `main` at 2026-09-21: **94%** (`TOTAL 9219 434 2642 202 94%`, 2276 passed in 18.5s). There is no `fail_under` in `[tool.coverage.report]` (`pyproject.toml:113-121`), so nothing checks the number and it will drift again. The fix is to stop quoting a literal that rots: set a floor and describe the floor. - -**6b — decided 2026-09-21: keep "Stable and in production use." exactly as written.** It was raised as a judgment call, not an error: `llms-full.txt` is mirrored to `evalshift.dev/cli-llms-full.txt`, so AI engines will repeat the phrase as a customer claim rather than as a self-description. The maintainer is comfortable with that. **Preserve the sentence byte-for-byte** — this task ships 6a alone. - -**Files:** -- Modify: `evalshift-cli/README.md:67-70` -- Modify: `evalshift-cli/pyproject.toml` (`[tool.coverage.report]`, add `fail_under`) - -**Interfaces:** none — nothing depends on this task, and nothing it depends on. - -- [ ] **Step 1: No decision pending.** 6b was answered on 2026-09-21: keep the sentence as-is. Proceed straight to Step 2 and change only the coverage claim. - -- [ ] **Step 2: Create the branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli checkout main && git -C /home/lukas/repos/evalshift/evalshift-cli pull -git -C /home/lukas/repos/evalshift/evalshift-cli checkout -b docs/checkable-status-claims -``` - -- [ ] **Step 3: Add the coverage floor to `pyproject.toml`** - -In `[tool.coverage.report]`, add above `exclude_lines`: - -```toml -# The README quotes this floor rather than a measured percentage, so the claim -# is enforced instead of hand-maintained. Measured 94% on 2026-09-21; the floor -# is set below that deliberately, so an honest refactor does not fail CI while -# a real drop still does. Raise it when the margin gets comfortable. -fail_under = 90 -``` - -- [ ] **Step 4: Verify the floor holds and actually bites** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -pytest 2>&1 | tail -3 -``` - -Expected: `TOTAL ... 94%` and `2276 passed` with **no** `FAIL Required test coverage of 90% not reached`. - -Then prove the gate is live: - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -pytest tests/unit/test_init.py --cov-fail-under=99 2>&1 | tail -2 -``` - -Expected: `FAIL Required test coverage of 99% not reached`. This confirms `fail_under` is wired, not inert. - -- [ ] **Step 5: Rewrite `README.md:67-70`** - -Replace: - -```markdown -**Stable and in production use.** Every command in the pipeline is shipped and -the test suite covers 92% of the source. The CLI is published on PyPI as -`evalshift`, the capture SDK as `evalshift-sdk`, and the hosted service runs at -`api.evalshift.dev`. -``` - -with exactly this — the lead sentence is unchanged per the 6b decision, and only -the coverage clause moves from a literal to the enforced floor: - -```markdown -**Stable and in production use.** Every command in the pipeline is shipped and -CI enforces a 90% coverage floor on the source. The CLI is published on PyPI as -`evalshift`, the capture SDK as `evalshift-sdk`, and the hosted service runs at -`api.evalshift.dev`. -``` - -- [ ] **Step 6: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -ruff check . && ruff format --check . && mypy --strict src/evalshift_cli && pytest -``` - -Expected: all clean. - -- [ ] **Step 7: Commit and open the PR** - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -git add README.md pyproject.toml -git commit -m "docs: enforce the coverage claim instead of hand-writing it - -README said 92%; the suite measures 94%, and nothing checked either number -because [tool.coverage.report] had no fail_under. Quote an enforced floor -rather than a literal that rots between releases." -git push -u origin docs/checkable-status-claims -gh pr create --fill -``` - ---- - -### Task 7: Re-publish the mirrored `llms-full.txt` files to the site - -**Run this only after Tasks 1–6 have merged to `main` in their source repos.** `npm run sync:llms` is a plain three-way `cp` (`evalshift-client/package.json:17`) with no test guarding drift, which is how the Task 1 text reached `evalshift.dev/ci-llms-full.txt` in the first place: the 2026-09-20 sync faithfully copied a stale source. - -This task also carries the two client-side docs mirrors the maintainer approved for Task 3 (decision of 2026-09-21): they live in this repo, and this is the plan's only client-repo branch. - -**Files:** -- Modify: `evalshift-client/public/ci-llms-full.txt` (from `evalshift-action`, Tasks 1 + 5) -- Modify: `evalshift-client/public/cli-llms-full.txt` (from `evalshift-cli`, Tasks 2 + 3 + 4) -- Modify: `evalshift-client/src/pages/docs/pages/Faq.tsx:21` -- Modify: `evalshift-client/src/pages/docs/pages/WhatGetsUploaded.tsx:94` -- `evalshift-client/public/sdk-llms-full.txt` — expected unchanged; no task touched the SDK. -- **Do not touch** `src/pages/landing/sections/Hero.tsx` or `src/pages/landing/Landing.test.tsx` — the hero's three-brand line is deliberate marketing copy. - -**Interfaces:** -- Consumes: the merged `llms-full.txt` of `evalshift-action` and `evalshift-cli`. -- Produces: nothing. - -- [ ] **Step 1: Confirm every source repo is on a clean, merged `main`** - -```bash -for r in evalshift-action evalshift-cli evalshift-sdk; do - echo "== $r" - git -C /home/lukas/repos/evalshift/$r checkout main -q && git -C /home/lukas/repos/evalshift/$r pull -q - git -C /home/lukas/repos/evalshift/$r status --short --branch -done -``` - -Expected: each reports `## main...origin/main` with no trailing `[ahead/behind]` and no modified files. - -- [ ] **Step 2: Create the branch and sync** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -git checkout main && git pull -git checkout -b chore/sync-llms-after-docs-cleanup -npm run sync:llms -``` - -- [ ] **Step 3: Verify the copies now match their sources exactly** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -diff ../evalshift-action/llms-full.txt public/ci-llms-full.txt && echo "ci OK" -diff ../evalshift-cli/llms-full.txt public/cli-llms-full.txt && echo "cli OK" -diff ../evalshift-sdk/llms-full.txt public/sdk-llms-full.txt && echo "sdk OK" -``` - -Expected: three `OK` lines, no diff output. - -- [ ] **Step 4: Verify the stale claims are gone from what the site serves** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -echo "--- all four greps must print nothing ---" -grep -n "policy:configure\|thresholds:" public/ci-llms-full.txt -grep -n "@v0\.3\.0" public/ci-llms-full.txt -grep -n "all --push" public/cli-llms-full.txt -grep -n "{{input}}" public/cli-llms-full.txt -``` - -Expected: no output from any of them. - -- [ ] **Step 5: Fix the two hand-written docs mirrors** - -These mirror `docs/faq.md` and `docs/hosted.md`, which Task 3 rewrote. Read each -file first — the surrounding JSX differs from the markdown and the replacement has -to fit the existing element structure, so match the file's own wrapping and -``/`` usage rather than pasting markdown prose. - -In `src/pages/docs/pages/Faq.tsx:21`, the phrase `providers you configured -(Anthropic, OpenAI, Google) using your` becomes the equivalent of *"providers you -configured — any provider LiteLLM supports — using your"*, matching how Task 3 -reworded `docs/faq.md:5-8`. - -In `src/pages/docs/pages/WhatGetsUploaded.tsx:94`, `Your model providers -(Anthropic, OpenAI, Google — whichever you` becomes the equivalent of *"(whichever -you configure — any provider LiteLLM supports)"*, matching Task 3's `docs/hosted.md:246-247`. - -- [ ] **Step 6: Confirm the hero is untouched and the mirrors changed** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -echo "--- hero must still say the old thing (deliberate) ---" -grep -n "Works with Anthropic, OpenAI and Google models" src/pages/landing/sections/Hero.tsx -echo "--- both mirrors must now name LiteLLM ---" -grep -c "LiteLLM" src/pages/docs/pages/Faq.tsx src/pages/docs/pages/WhatGetsUploaded.tsx -echo "--- landing test must be unmodified ---" -git diff --name-only | grep -c "Landing.test.tsx" || echo "0 (correct)" -``` - -Expected: the hero line present; `1` for each mirror; `0 (correct)` for the landing test. - -- [ ] **Step 7: Run the full gate** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -npm run lint && npm run typecheck && npm run build && npx vitest run -``` - -Expected: all pass, including `Landing.test.tsx` — its assertion on the hero string is -untouched and must stay green. The `public/*.txt` files are served statically and are not -parsed by the build; that part is a regression check on the site, not on the copies. - -- [ ] **Step 8: Commit** - -```bash -cd /home/lukas/repos/evalshift/evalshift-client -git add public/ci-llms-full.txt public/cli-llms-full.txt src/pages/docs/pages/ -git commit -m "chore: sync llms-full after the docs cleanup - -Picks up the action's removed thresholds/policy:configure guidance and current -example tag, and the CLI's compare --push, LiteLLM provider scope and {input} -placeholder. sync:llms is a plain cp with no drift guard, so the 2026-09-20 sync -faithfully republished stale sources; the guards now live in the source repos." -git push -u origin chore/sync-llms-after-docs-cleanup -gh pr create --fill -``` - -- [ ] **Step 9: Confirm the deploy served the new text** *(after the user pushes and merges)* - -After the PR merges and the site deploys: - -```bash -curl -s https://www.evalshift.dev/ci-llms-full.txt | grep -c "policy:configure" -curl -s https://www.evalshift.dev/cli-llms-full.txt | grep -c "all --push" -``` - -Expected: `0` from both. - ---- - -## Commit the plan itself - -```bash -cd /home/lukas/repos/evalshift/evalshift-cli -git add -f docs/superpowers/plans/2026-09-21-stale-docs-cleanup.md -git commit -m "docs(plan): stale docs cleanup across action, cli and site" -``` - -The `-f` is required: `docs/superpowers/plans/` matches a `.gitignore` pattern, but the plan files are tracked. - -## Out of scope, recorded deliberately - -- **`evalshift-client`'s landing hero.** `src/pages/landing/sections/Hero.tsx:113` ("Works with Anthropic, OpenAI and Google models.", asserted by `Landing.test.tsx:253`) keeps the three-brand line: the maintainer ruled on 2026-09-21 that naming three recognisable brands is a legitimate marketing choice, distinct from a docs page stating a capability boundary. The two docs mirrors were brought in scope by the same decision and are executed in Task 7. -- **A cross-repo drift guard for `sync:llms`.** The honest guard — a client-side test diffing `public/*-llms-full.txt` against `../evalshift-*/llms-full.txt` — cannot run in the client's CI, where the sibling repos are not checked out. The guards this plan adds live in the source repos instead, which is where the rot starts. -- **`projects.thresholds`.** The server still stores a per-project `thresholds` blob, edited through `PATCH /projects/{id}` under `project:update` (`app/authz/permissions.py:30-32`). Only the push-time sync and its permission were removed. Nothing in this plan touches that field. diff --git a/docs/superpowers/plans/2026-09-30-deepseek-provider.md b/docs/superpowers/plans/2026-09-30-deepseek-provider.md deleted file mode 100644 index 5ab5dbd..0000000 --- a/docs/superpowers/plans/2026-09-30-deepseek-provider.md +++ /dev/null @@ -1,1430 +0,0 @@ -# DeepSeek as a Supported Provider Implementation Plan - -> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. - -**Goal:** Make DeepSeek a first-class EvalShift provider: CLI runs, captures replays, tool-call evals, `init`, doctor and the report handle it correctly. Every public surface that lists providers says so: CLI/SDK/action docs, the website docs and the landing page. - -**Architecture:** The CLI already sends every call through LiteLLM, which knows the `deepseek/` provider. The work is mostly teaching EvalShift's own provider taxonomy about DeepSeek: registry, key pre-check, tool-response parsing, pricing lookup and `init`. There is also one small module for DeepSeek's thinking-mode behaviour. The SDK needs no code change, because `wrap_openai(OpenAI(base_url=...))` already captures DeepSeek. The action, SDK and client changes are docs and copy only. - -**Tech Stack:** Python 3.11+, LiteLLM (`litellm>=1.77,<2`, 1.100.0 installed), Typer, pytest, mypy strict, ruff (CLI/SDK/action). Vite + React 19 + TypeScript + Vitest (client). - -**Spec:** This plan's own **Findings** and **Design decisions** sections below. They come from the 2026-09-30 audit, which included a live LiteLLM probe run in `evalshift-cli/.venv`. There is no separate spec doc. - ---- - -## Findings (why each task exists) - -Verified on 2026-09-30 against the checked-out code and LiteLLM 1.100.0: - -| # | Symptom today | Root cause | -|---|---|---| -| F1 | `deepseek/deepseek-flash` works for plain completions only by accident: provider is `"other"` | `registry._infer_provider_and_canonical` has no DeepSeek rule | -| F2 | A capture made with `OpenAI(base_url="https://api.deepseek.com")` records the bare id `deepseek-flash`. Replaying it fails: LiteLLM raises `BadRequestError: LLM Provider NOT provided` | Bare `deepseek-*` ids are not prefixed with `deepseek/` | -| F3 | Every tool-call eval on a DeepSeek arm raises `ModelError` | `tool_parser.detect_provider` raises `ToolParseError` for any id that isn't Anthropic, OpenAI or Gemini (`evaluators/tool_parser.py:39-64`) | -| F4 | A missing `DEEPSEEK_API_KEY` is not caught before a run. `doctor` doesn't show the key | `PROVIDER_ENV_VARS` has no DeepSeek entry, and `"other"` skips the pre-check (`cli/commands/run.py:222`) | -| F5 | A DeepSeek judge grading DeepSeek arms gets no self-preference warning | `models/family.py:65`: `"other"` never matches | -| F6 | `capture sync` prices a bare `deepseek-flash` call at $0 | `utils/cost._price_table_key` returns the bare key first. `litellm.cost_per_token("deepseek-flash")` raises (no provider); only `deepseek/deepseek-flash` prices | -| F7 | **Temperature is silently ignored**, and nothing warns | DeepSeek's current models (`deepseek-flash`, `deepseek-v4-pro`) run in *thinking mode by default*. Thinking mode "does not support `temperature`, `presence_penalty`, `frequency_penalty` … setting them will not trigger an error but will also have no effect" ([docs](https://api-docs.deepseek.com/guides/thinking_mode/)). LiteLLM still lists `temperature` as supported, so `honors_temperature` returns `True` | -| F8 | **Multi-round tool replay against DeepSeek returns HTTP 400** | With `tools` on the request, DeepSeek requires the `reasoning_content` of every earlier assistant turn, or it returns 400. `orchestrator.build_round_messages` (`runner/orchestrator.py:966-985`) and `_dispatch_message` (`:1031-1047`) build assistant turns without it. LiteLLM backfills a `" "` placeholder only when `thinking={"type":"enabled"}` is passed explicitly (`litellm/llms/deepseek/chat/transformation.py:214-227`). EvalShift never passes it, and our floor `litellm>=1.77` may predate that backfill anyway | -| F9 | The landing page says "Works with Anthropic, OpenAI and Google models." The CLI, action and site docs list only three key env vars | Copy | - -DeepSeek API model ids as of 2026-09-30 ([pricing page](https://api-docs.deepseek.com/quick_start/pricing)): -- `deepseek-flash` (DeepSeek-V4.1-Flash, 1M context) -- `deepseek-v4-pro` (DeepSeek-V4-Pro-0813, 1M context) - -The legacy `deepseek-v4-flash` is still accepted and served by Flash. LiteLLM 1.100.0 prices both current ids, reports `supports_reasoning=True` and `supports_function_calling=True`, and reads `DEEPSEEK_API_KEY` / `DEEPSEEK_API_BASE`. - -## Design decisions - -1. **DeepSeek becomes a registry `Provider` (`"deepseek"`).** It is attributed only for the `deepseek/` prefix and bare `deepseek-*` ids. The same weights served by another host stay `"other"`, because they authenticate with that host's keys, not `DEEPSEEK_API_KEY`. Examples: `azure_ai/deepseek-v4-pro`, `bedrock/…deepseek…`, `hosted_vllm/deepseek-ai/…`, `openrouter/deepseek/…`. -2. **Tool parsing treats DeepSeek as OpenAI-shaped, wherever it is hosted.** `detect_provider` returns `"openai"` for any id containing `deepseek`. Its `Provider` literal names a *response shape*, not a vendor. LiteLLM normalises DeepSeek tool calls to OpenAI's `tool_calls`, which is exactly what `_parse_gemini` already relies on for Gemini. -3. **EvalShift never switches thinking off.** A capture cannot record `thinking` (it is not in the SDK's `GENERATION_KEYS`), so the app's own setting is unknown. The API default is thinking on. Instead: - - `honors_temperature` returns `False` for thinking-by-default DeepSeek models, so the report's existing non-determinism banner fires (fixes F7). - - The client backfills `reasoning_content: " "` on assistant turns that lack it, for those models only (fixes F8). This is the same placeholder LiteLLM uses, done in EvalShift so it doesn't depend on the LiteLLM version or on an explicit `thinking` flag. -4. **"Thinking by default" means provider `deepseek` and `litellm.supports_reasoning(model=)` is `True`.** Any exception or `False` reads as "not thinking". That matches `capabilities.py`'s rule that uncertainty reads as honoured (no false banner). Caveat (accepted): LiteLLM's reasoning flag is also `True` for DeepSeek models whose thinking is opt-in (e.g. `deepseek/deepseek-v3.2`), so those get a false non-determinism banner and a harmless `reasoning_content` placeholder. The current API ids `deepseek-flash` / `deepseek-v4-pro` think by default. -5. **Pricing lookup tries the canonical (provider-prefixed) id first.** This fixes F6. Measured effect on existing ids (accepted, ruling R10): identical for every text model — `openai/gpt-4o-mini` and `anthropic/claude-*` are not table keys, so they still fall through to the bare/stripped form. A few niche Gemini keys do change: `gemini-exp-1206` now prices at $0 because its `gemini/` entry has zero prices, and image models' cache-read price and the image-preview >200k-token tiers differ between the prefixed and bare entries. -6. **`init --provider deepseek`** scaffolds: - - `deepseek-flash` as the source model - - `deepseek-v4-pro` as the target hint and judge - - the semantic block commented out, because DeepSeek has no embedding endpoint (same as Anthropic) - - `gemini` stays the default. -7. **Docs name the curated providers alphabetically** ("Anthropic, DeepSeek, Google and OpenAI"). This keeps `test_provider_scope_is_not_capped_at_three`'s regex guarding the old three-brand phrasing without editing it. A new docs-currency test makes every registry key env var and every `init --provider` choice appear in the reference docs, so the next provider can't be half-documented. - -### Maintainer decisions to confirm - -The plan proceeds with the default shown; change any of these before execution if you disagree. - -- **D1 – doctor row.** `DEEPSEEK_API_KEY` gets an always-shown doctor row, like the other three. A non-DeepSeek user sees one more yellow `✗`. *Alternative:* show the row only when the key is set or the config names a DeepSeek model (more code, and it breaks the "one row per provider" rule). -- **D2 – landing copy.** `Works with Anthropic, OpenAI, Google and DeepSeek models.` This reverses the 2026-09-21 "do not touch Hero.tsx:113" note, because you asked for it. *Alternative:* `Works with Anthropic, OpenAI, Google, DeepSeek and any model LiteLLM supports.` -- **D3 – blog posts stay as published.** `llm-regression-testing-in-ci.md:40-42` lists three keys. Blog posts are dated articles, the same policy used for `all --push` on 2026-09-21. -- **D4 – release.** The CLI ships this as **1.2.0** (a new provider and a new `init` choice is a minor bump). The SDK change is docs-only and needs no PyPI release; the PyPI README catches up at the next SDK release. The client copy ships only after CLI 1.2.0 is on PyPI. - -## Global Constraints - -- Work happens in per-repo worktrees already cut from fresh `origin/main`: `/home/lukas/repos/evalshift/evalshift-{cli,sdk,action,client}-wt-deepseek` on branches `feat/deepseek-provider`, `docs/deepseek`, `docs/deepseek-key`, `feat/deepseek-copy`. Never touch the original checkouts. Always run git as `git -C `. The plan file is gitignored-but-tracked: commit it with `git add -f`. -- CLI gates (all must pass before each commit): `uv run ruff check .`, `uv run ruff format --check .`, `uv run mypy --strict src/evalshift_cli`, `uv run pytest --cov-fail-under=90`, which together are `make ci`. -- Conventional Commits. Commit trailer: `Co-Authored-By: Claude Opus 5.5 (1M context) `. -- Every user-visible CLI change updates `DOCS.md`, `llms-full.txt`, the matching `docs/` page and `CHANGELOG.md` under `## [Unreleased]` (per `evalshift-cli/CLAUDE.md:77`). The SDK has the same CHANGELOG rule. The action, server and client have no CHANGELOG. -- No version bumps inside feature PRs; they happen in the release commit (Task E1). -- `llms-full.txt` is hand-maintained in each repo. `evalshift-client/public/{cli,sdk,ci}-llms-full.txt` are **never** hand-edited; they are refreshed by `npm run sync:llms` (Task E2). -- Enumerate doc sites with a repo-wide `git grep` before editing, not only from the lists in this plan. Two past plans miscounted hand-written site lists. -- Model ids used everywhere: `deepseek-flash`, `deepseek-v4-pro`. Canonical: `deepseek/deepseek-flash`, `deepseek/deepseek-v4-pro`. Env var: `DEEPSEEK_API_KEY`. - -## Review Focus - -These are the five inputs most likely to hurt a DeepSeek user that no single task's happy path covers. Each has a pinning test in the task named after the arrow. - -1. **A bare id recorded by a capture** (`deepseek-flash` with no prefix) must replay, price, pre-check its key and parse tool calls. → A1 (resolve), A2 (price), A3 (tool parse via the resolved id) -2. **A multi-round tool replay whose assistant turns came from another model** must not 400 on DeepSeek. → A4 client test with an assistant `tool_calls` turn and no `reasoning_content`; confirmed live in A7 -3. **A DeepSeek arm must not be reported as deterministic.** → A4 capabilities test: `honors_temperature` is `False` even though LiteLLM lists `temperature` -4. **DeepSeek hosted elsewhere** (`azure_ai/deepseek-v4-pro`, `hosted_vllm/deepseek-ai/…`) must parse tool calls, but must not demand `DEEPSEEK_API_KEY`. → A1 (provider stays `"other"`) + A3 (parses as OpenAI shape) -5. **Non-DeepSeek providers must be byte-for-byte unaffected.** No `reasoning_content` on Gemini/OpenAI messages; unchanged pricing keys for `gpt-4o-mini` / `gemini-2.5-flash`. → A4 negative client test; A2 regression test - ---- - -## File map - -**evalshift-cli** (branch `feat/deepseek-provider`) -- Modify `src/evalshift_cli/models/registry.py`: `Provider`, `PROVIDER_ENV_VARS`, `_MODELS`, prefix inference -- Modify `src/evalshift_cli/utils/cost.py`: `_price_table_key` order -- Modify `src/evalshift_cli/evaluators/tool_parser.py`: `detect_provider` -- Create `src/evalshift_cli/models/deepseek.py`: thinking-mode quirks (one responsibility: DeepSeek API behaviour) -- Modify `src/evalshift_cli/models/capabilities.py`: `honors_temperature` -- Modify `src/evalshift_cli/models/client.py`: backfill in `_dispatch_with_retry` -- Modify `src/evalshift_cli/cli/commands/init.py`, `_scaffold.py`, `_agents.py` -- Modify `scripts/smoke_live_tools.py` -- Docs: `README.md`, `DOCS.md`, `llms-full.txt`, `docs/getting-started.md`, `docs/faq.md`, `docs/configuration.md`, `CHANGELOG.md`, `pyproject.toml` (keywords) -- Tests: - - `tests/unit/test_model_registry.py`, `test_model_family.py`, `test_doctor.py`, `test_run_command.py`, `test_cost.py` - - `test_tool_parser.py` (+ fixture `tests/unit/fixtures/tool_responses/deepseek/single_tool_call.json`) - - new `test_deepseek.py`, plus `test_model_capabilities.py`, `test_model_client.py`, `test_init.py`, `test_docs_currency.py` - -**evalshift-sdk** (branch `docs/deepseek`): `README.md`, `DOCS.md`, `llms-full.txt`, `CHANGELOG.md` - -**evalshift-action** (branch `docs/deepseek-key`): `README.md`, `DOCS.md`, `llms-full.txt`, `tests/test_evalshift_action.py` - -**evalshift-client** (branch `feat/deepseek-copy`): -- `src/pages/landing/sections/Hero.tsx`, `src/pages/landing/Landing.test.tsx` -- `src/pages/docs/pages/{CliCommands,Cli,Faq,Action,Configuration,GettingStarted,SdkAdapters}.tsx` - -**evalshift-server:** no change. `source_model`/`target_model` are free strings (`app/runs/bundle.py:321-322`), and `judge_family_overlap` has no server-side enum. - ---- - -## Phase A: evalshift-cli - -### Task A0: Branch - -- [ ] **Step 1: Cut the branch from fresh main** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek fetch origin -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek switch -c feat/deepseek-provider origin/main -cd /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek && uv sync -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add -f docs/superpowers/plans/2026-09-30-deepseek-provider.md -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "docs(plan): DeepSeek as a supported provider" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A1: Registry knows DeepSeek (fixes F1, F2, F4, F5) - -**Files:** -- Modify: `src/evalshift_cli/models/registry.py:36-44` (`Provider`, `PROVIDER_ENV_VARS`), `:88-131` (`_MODELS`), `:237-263` (`_infer_provider_and_canonical` + docstrings at `:18-22`, `:66`) -- Test: `tests/unit/test_model_registry.py`, `tests/unit/test_model_family.py`, `tests/unit/test_doctor.py`, `tests/unit/test_run_command.py` - -**Interfaces:** -- Produces: `Provider = Literal["anthropic", "openai", "google", "deepseek", "other"]`; `PROVIDER_ENV_VARS["deepseek"] == ("DEEPSEEK_API_KEY",)`; `resolve_model("deepseek-flash").id == "deepseek/deepseek-flash"`; `resolve_model()` → `("deepseek/", "deepseek")`. Doctor, `run`/`compare` pre-checks, `insights/stage.py` and `models/family.py` pick this up with no code change. - -- [ ] **Step 1: Write the failing registry tests** - -In `tests/unit/test_model_registry.py`, replace the body of `test_every_provider_represented`: - -```python - def test_every_provider_represented(self) -> None: - providers = {m.provider for m in list_supported()} - assert providers == {"anthropic", "openai", "google", "deepseek"} -``` - -Append to `class TestResolveModel`: - -```python - def test_bare_deepseek_alias_uses_registry(self) -> None: - meta = resolve_model("deepseek-flash") - assert meta.id == "deepseek/deepseek-flash" - assert meta.provider == "deepseek" - assert "(passthrough)" not in meta.display_name - - def test_unknown_deepseek_prefix_inferred(self) -> None: - # What a capture records when the app called api.deepseek.com through - # the OpenAI client: the bare id, which LiteLLM cannot route alone. - meta = resolve_model("deepseek-v5-preview") - assert meta.id == "deepseek/deepseek-v5-preview" - assert meta.provider == "deepseek" - assert meta.display_name.endswith("(passthrough)") - - def test_prefixed_deepseek_id_passes_through(self) -> None: - meta = resolve_model("deepseek/deepseek-v5-preview") - assert meta.id == "deepseek/deepseek-v5-preview" - assert meta.provider == "deepseek" - - @pytest.mark.parametrize( - "model_id", - [ - "azure_ai/deepseek-v4-pro", - "hosted_vllm/deepseek-ai/DeepSeek-V4-Flash", - "openrouter/deepseek/deepseek-v4-pro", - ], - ) - def test_deepseek_on_another_host_is_not_the_deepseek_provider(self, model_id: str) -> None: - # Another host authenticates with its own keys, never DEEPSEEK_API_KEY, - # so it must not be attributed to the deepseek provider's key check. - assert resolve_model(model_id).provider == "other" - - -class TestProviderEnvVars: - def test_deepseek_key(self) -> None: - assert PROVIDER_ENV_VARS["deepseek"] == ("DEEPSEEK_API_KEY",) -``` - -Add `PROVIDER_ENV_VARS` to the file's `from evalshift_cli.models.registry import (...)` block, and add `import pytest` if it is not already imported. - -- [ ] **Step 2: Write the failing family, doctor and pre-check tests** - -Append to `class TestSharedJudgeFamily` in `tests/unit/test_model_family.py`: - -```python - def test_deepseek_judge_on_a_deepseek_arm_is_one_family(self) -> None: - roles = shared_judge_family( - judge_model="deepseek-v4-pro", - source_model="gpt-5.4-mini", - target_model="deepseek-flash", - ) - assert roles == ["target"] -``` - -In `tests/unit/test_doctor.py`, extend `TestRunChecksAPIKeys.test_partial_keys` with one more line at the end: - -```python - assert _by_name(results, "DEEPSEEK_API_KEY").status == "warn" -``` - -In `tests/unit/test_run_command.py`, add `"DEEPSEEK_API_KEY",` to the tuple in `TestRunApiKeyPrecheck._clear_keys`. Then append this method to the class: - -```python - def test_deepseek_arm_without_key_is_caught_before_the_run( - self, monkeypatch: pytest.MonkeyPatch, tmp_path: Path - ) -> None: - _scaffold(tmp_path) - monkeypatch.chdir(tmp_path) - self._clear_keys(monkeypatch) - - result = runner.invoke( - app, ["run", "--from", "deepseek-flash", "--to", "deepseek-v4-pro", "--yes"] - ) - assert result.exit_code == 1 - assert "missing API key" in result.stdout - assert "DEEPSEEK_API_KEY" in result.stdout -``` - -- [ ] **Step 3: Run the tests to verify they fail** - -Run: `uv run pytest tests/unit/test_model_registry.py tests/unit/test_model_family.py tests/unit/test_doctor.py::TestRunChecksAPIKeys tests/unit/test_run_command.py::TestRunApiKeyPrecheck -q` -Expected: FAIL. `providers` is missing `"deepseek"`, `resolve_model("deepseek-flash").provider == "other"`, `KeyError: 'deepseek'`, the family roles are `[]`, there is no `DEEPSEEK_API_KEY` row, and the run pre-check passes the DeepSeek arms. - -- [ ] **Step 4: Implement** - -In `registry.py`: - -```python -Provider = Literal["anthropic", "openai", "google", "deepseek", "other"] - -# Env vars LiteLLM reads to authenticate each provider, in preference -# order (primary first; the second entry is an accepted alias). -PROVIDER_ENV_VARS: Final[dict[Provider, tuple[str, ...]]] = { - "anthropic": ("ANTHROPIC_API_KEY",), - "openai": ("OPENAI_API_KEY",), - "google": ("GEMINI_API_KEY", "GOOGLE_API_KEY"), - "deepseek": ("DEEPSEEK_API_KEY",), -} -``` - -Update the `ModelMetadata.provider` docstring line to ``"anthropic"`` | ``"openai"`` | ``"google"`` | ``"deepseek"``. - -Append to `_MODELS`, after the Google block: - -```python - # ---- DeepSeek -------------------------------------------------------- - # The two ids DeepSeek's API serves as of 2026-09. Both run in thinking - # mode by default, which ignores temperature — see models/deepseek.py. - ModelMetadata( - id="deepseek/deepseek-flash", - provider="deepseek", - display_name="DeepSeek V4.1 Flash", - aliases=("deepseek-flash",), - ), - ModelMetadata( - id="deepseek/deepseek-v4-pro", - provider="deepseek", - display_name="DeepSeek V4 Pro", - aliases=("deepseek-v4-pro",), - ), -``` - -In `_infer_provider_and_canonical`, add `"deepseek": "deepseek",` to `prefix_to_provider`. Add this branch after the `gpt-`/`o1-`/`o3-` branch: - -```python - if id_or_alias.startswith("deepseek-"): - return f"deepseek/{id_or_alias}", "deepseek" -``` - -Add a bullet to its docstring decision tree ("If it starts with ``deepseek-`` → deepseek, prefix ``deepseek/``."). Add the same rule to the module docstring's prefix list (`:18-22`). - -- [ ] **Step 5: Run the tests to verify they pass, then the full gate** - -Run: `uv run pytest tests/unit/test_model_registry.py tests/unit/test_model_family.py tests/unit/test_doctor.py tests/unit/test_run_command.py -q`, then `make ci` -Expected: PASS. mypy may flag a `match`/`dict` over `Provider` that is now non-exhaustive. Fix any such site by adding the `"deepseek"` case, not by casting. - -- [ ] **Step 6: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add src/evalshift_cli/models/registry.py tests/unit/test_model_registry.py tests/unit/test_model_family.py tests/unit/test_doctor.py tests/unit/test_run_command.py -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "feat(models): register DeepSeek as a provider" -m "Bare deepseek-* ids now resolve to deepseek/…, DEEPSEEK_API_KEY is pre-checked and shown by doctor, and a DeepSeek judge on a DeepSeek arm is flagged as one family." -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A2: Price a bare id under its provider-prefixed key (fixes F6) - -**Files:** -- Modify: `src/evalshift_cli/utils/cost.py:138-153` (`_price_table_key`) -- Test: `tests/unit/test_cost.py` (`class TestEstimateCallCost`) - -**Interfaces:** -- Consumes: `resolve_model("deepseek-flash").id == "deepseek/deepseek-flash"` (A1) -- Produces: `estimate_call_cost("deepseek-flash", …) > 0` with the real LiteLLM table - -- [ ] **Step 1: Write the failing test** - -Append to `class TestEstimateCallCost`: - -```python - def test_bare_id_is_priced_under_its_provider_prefixed_key( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - # litellm's table holds DeepSeek under both the bare and the prefixed - # key, but cost_per_token can only infer a provider from the prefixed - # one — asked about the bare id it raises, which used to price a - # DeepSeek capture at $0. - monkeypatch.setattr( - cost_module.litellm, - "model_cost", - {"deepseek-flash": {}, "deepseek/deepseek-flash": {}}, - ) - - def priced(*, model: str, prompt_tokens: int, completion_tokens: int) -> tuple[float, float]: - if model != "deepseek/deepseek-flash": - raise ValueError(f"LLM Provider NOT provided: {model}") - return prompt_tokens * 0.001, completion_tokens * 0.002 - - monkeypatch.setattr(cost_module.litellm, "cost_per_token", priced) - - assert cost_module.estimate_call_cost("deepseek-flash", 1000, 100) == pytest.approx(1.2) -``` - -- [ ] **Step 2: Run it to verify it fails** - -Run: `uv run pytest tests/unit/test_cost.py::TestEstimateCallCost -q` -Expected: the new test FAILS (`0.0 != 1.2`). The other five pass. - -- [ ] **Step 3: Implement** - -In `_price_table_key`, change the loop to try the canonical form first, and update the docstring: - -```python - """The key litellm's price table holds ``model_id`` under, or ``None``. - - Tries the registry's canonical (provider-prefixed) form first, then the id - as recorded, then the canonical form with the prefix stripped. The order - matters: litellm keys some providers under both spellings but can only - price the prefixed one (``deepseek-flash`` is a key, yet - ``cost_per_token`` cannot infer its provider), while most first-party - entries exist only bare (``gpt-4o-mini``) and fall through to the second - or third candidate. A pure dict lookup: ``litellm.model_cost`` is the - bundled table, so a miss costs nothing and touches nothing. - """ - canonical = resolve_model(model_id).id - _, _, stripped = canonical.partition("/") - table = litellm.model_cost - for candidate in (canonical, model_id, stripped): -``` - -- [ ] **Step 4: Run the tests** - -Run: `uv run pytest tests/unit/test_cost.py tests/unit/test_captures_promote.py -q` -Expected: PASS. `test_resolves_registry_aliases_and_provider_prefixes` still passes because `gemini/gemini-2.5-flash` is not in its stub table. - -- [ ] **Step 5: Confirm against the real table (no network)** - -Run: `uv run python -c "from evalshift_cli.utils.cost import estimate_call_cost as c; print([c(m,1000,1000) for m in ('deepseek-flash','gpt-4o-mini','gemini-2.5-flash','claude-sonnet-4-5')])"` -Expected: four non-zero numbers. - -- [ ] **Step 6: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add src/evalshift_cli/utils/cost.py tests/unit/test_cost.py -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "fix(cost): price recorded calls under the provider-prefixed key first" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A3: Parse DeepSeek tool calls as OpenAI-shaped (fixes F3) - -**Files:** -- Modify: `src/evalshift_cli/evaluators/tool_parser.py:1-11` (module docstring), `:39-64` (`detect_provider`) -- Create: `tests/unit/fixtures/tool_responses/deepseek/single_tool_call.json` -- Test: `tests/unit/test_tool_parser.py` (`TestDetectProvider`, new `TestParseDeepSeek`), `tests/unit/test_model_client.py` (`TestCompleteWithTools`) - -**Interfaces:** -- Consumes: A1's canonical ids -- Produces: `detect_provider() == "openai"`. `ModelClient.complete_with_tools` serialises tools with `to_openai()` for DeepSeek and parses with `_parse_openai` - -- [ ] **Step 1: Create the fixture** - -`tests/unit/fixtures/tool_responses/deepseek/single_tool_call.json` is a LiteLLM-normalised DeepSeek thinking-mode response. It carries `reasoning_content`, which the parser must ignore: - -```json -{ - "id": "chatcmpl_synthetic_deepseek_single", - "object": "chat.completion", - "model": "deepseek-flash", - "choices": [ - { - "index": 0, - "finish_reason": "tool_calls", - "message": { - "role": "assistant", - "content": "", - "reasoning_content": "The user wants ACME's Q3 records; search the database first.", - "tool_calls": [ - { - "id": "call_00_synthetic", - "type": "function", - "function": { - "name": "search_db", - "arguments": "{\"query\": \"ACME Q3\"}" - } - } - ] - } - } - ], - "usage": {"prompt_tokens": 60, "completion_tokens": 40, "total_tokens": 100} -} -``` - -- [ ] **Step 2: Write the failing tests** - -In `TestDetectProvider.test_known_models`'s parametrize list, add: - -```python - ("deepseek/deepseek-flash", "openai"), - ("deepseek-v4-pro", "openai"), - # Same weights, other hosts: LiteLLM returns OpenAI's shape for all. - ("azure_ai/deepseek-v4-pro", "openai"), - ("hosted_vllm/deepseek-ai/DeepSeek-V4-Flash", "openai"), -``` - -Add after the OpenAI parser tests: - -```python -class TestParseDeepSeek: - def test_single_tool_call_ignores_reasoning_content(self) -> None: - raw = _load("deepseek", "single_tool_call") - model_id = "deepseek/deepseek-flash" - trace = parse_response_to_trace( - raw, provider=detect_provider(model_id), model_id=model_id - ) - assert trace.call_count == 1 - assert trace.calls[0].tool_name == "search_db" - assert trace.calls[0].arguments == {"query": "ACME Q3"} - assert trace.calls[0].call_id == "call_00_synthetic" - # The reasoning chain is not the answer. - assert not trace.final_text -``` - -Append to `TestCompleteWithTools` in `tests/unit/test_model_client.py`: - -```python - async def test_deepseek_bare_id_dispatches_openai_shaped_tools( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - captured = _patch_tools_acompletion(monkeypatch, _OPENAI_SINGLE_RESPONSE) - result = await ModelClient().complete_with_tools( - model="deepseek-flash", - prompt="hi", - tools=[_DEMO_TOOL], - ) - assert result.model_id == "deepseek/deepseek-flash" - assert result.trace.calls[0].tool_name == "search_db" - kwargs = captured["kwargs"] - assert kwargs["model"] == "deepseek/deepseek-flash" - assert kwargs["tools"][0]["type"] == "function" -``` - -- [ ] **Step 3: Run the tests to verify they fail** - -Run: `uv run pytest tests/unit/test_tool_parser.py tests/unit/test_model_client.py::TestCompleteWithTools -q` -Expected: FAIL with `ToolParseError: unknown: cannot detect provider for model id: 'deepseek/deepseek-flash'`. - -- [ ] **Step 4: Implement** - -In `detect_provider`, insert this before the final `raise`: - -```python - if "deepseek" in lowered: - # DeepSeek's API is OpenAI-compatible, and LiteLLM hands its tool calls - # back in OpenAI's ``tool_calls`` shape wherever the model is hosted - # (deepseek/, azure_ai/, bedrock/, hosted_vllm/, openrouter/ ...). - return "openai" -``` - -Update `detect_provider`'s docstring. Its `Raises:` paragraph should say the id "can't be mapped to one of the response shapes we parse (Anthropic, OpenAI, Gemini)". Add a sentence saying the return value names a **response shape**, so DeepSeek maps to `"openai"`. Add the same note to the module docstring. - -- [ ] **Step 5: Run the tests, then the gate** - -Run: `uv run pytest tests/unit/test_tool_parser.py tests/unit/test_model_client.py tests/unit/test_model_capabilities.py -q`, then `make ci` -Expected: PASS. - -- [ ] **Step 6: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add src/evalshift_cli/evaluators/tool_parser.py tests/unit/fixtures/tool_responses/deepseek tests/unit/test_tool_parser.py tests/unit/test_model_client.py -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "feat(tools): parse DeepSeek tool calls as OpenAI-shaped" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A4: DeepSeek thinking mode: banner the ignored temperature, backfill reasoning (fixes F7, F8) - -**Files:** -- Create: `src/evalshift_cli/models/deepseek.py` -- Modify: `src/evalshift_cli/models/capabilities.py:164-185` (`honors_temperature`) and its module docstring; `src/evalshift_cli/models/client.py:684-730` (`_dispatch_with_retry` entry) -- Test: create `tests/unit/test_deepseek.py`; extend `tests/unit/test_model_capabilities.py` and `tests/unit/test_model_client.py` - -**Interfaces:** -- Consumes: `resolve_model(...).provider == "deepseek"` (A1) -- Produces: - - `evalshift_cli.models.deepseek.REASONING_PLACEHOLDER: Final = " "` - - `thinking_by_default(model_id: str) -> bool` - - `backfill_reasoning_content(messages: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]` - - `honors_temperature("deepseek-flash") is False` whenever LiteLLM reports reasoning support - -- [ ] **Step 1: Write the failing module tests** - -`tests/unit/test_deepseek.py`: - -```python -"""Tests for :mod:`evalshift_cli.models.deepseek`. - -LiteLLM's reasoning flag is stubbed in every test: the bundled table is -upstream data, and a LiteLLM upgrade must not flip these tests. -""" - -from __future__ import annotations - -from typing import Any - -import pytest - -from evalshift_cli.models import deepseek as deepseek_module -from evalshift_cli.models.deepseek import ( - REASONING_PLACEHOLDER, - backfill_reasoning_content, - thinking_by_default, -) - - -def _stub_reasoning(monkeypatch: pytest.MonkeyPatch, result: bool | Exception) -> list[str]: - seen: list[str] = [] - - def fake(*, model: str, **_: Any) -> bool: - seen.append(model) - if isinstance(result, Exception): - raise result - return result - - monkeypatch.setattr(deepseek_module.litellm, "supports_reasoning", fake) - return seen - - -class TestThinkingByDefault: - def test_reasoning_deepseek_model_thinks_by_default( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - seen = _stub_reasoning(monkeypatch, True) - assert thinking_by_default("deepseek-flash") is True - # Asked about the canonical id, which is what LiteLLM keys on. - assert seen == ["deepseek/deepseek-flash"] - - def test_non_reasoning_deepseek_model_does_not(self, monkeypatch: pytest.MonkeyPatch) -> None: - _stub_reasoning(monkeypatch, False) - assert thinking_by_default("deepseek/deepseek-chat") is False - - def test_uncertain_answer_reads_as_not_thinking(self, monkeypatch: pytest.MonkeyPatch) -> None: - _stub_reasoning(monkeypatch, RuntimeError("no model info")) - assert thinking_by_default("deepseek-v5-preview") is False - - @pytest.mark.parametrize( - "model_id", ["gemini/gemini-2.5-flash", "gpt-4o", "azure_ai/deepseek-v4-pro"] - ) - def test_other_providers_never_ask( - self, monkeypatch: pytest.MonkeyPatch, model_id: str - ) -> None: - seen = _stub_reasoning(monkeypatch, True) - assert thinking_by_default(model_id) is False - assert seen == [] - - -class TestBackfillReasoningContent: - def test_assistant_turn_without_reasoning_gets_the_placeholder(self) -> None: - messages = [ - {"role": "user", "content": "find ACME"}, - {"role": "assistant", "content": "", "tool_calls": [{"id": "call_r0_0"}]}, - {"role": "tool", "tool_call_id": "call_r0_0", "content": "{}"}, - ] - out = backfill_reasoning_content(messages) - assert out[1]["reasoning_content"] == REASONING_PLACEHOLDER == " " - assert "reasoning_content" not in out[0] - assert "reasoning_content" not in out[2] - - def test_recorded_reasoning_is_kept(self) -> None: - messages = [{"role": "assistant", "content": "x", "reasoning_content": "because"}] - assert backfill_reasoning_content(messages)[0]["reasoning_content"] == "because" - - def test_input_is_not_mutated(self) -> None: - turn = {"role": "assistant", "content": ""} - backfill_reasoning_content([turn]) - assert "reasoning_content" not in turn -``` - -- [ ] **Step 2: Write the failing capability and client tests** - -Append to `tests/unit/test_model_capabilities.py`: - -```python -class TestDeepSeekThinkingMode: - """Thinking mode accepts ``temperature`` and ignores it; LiteLLM still lists it.""" - - def test_thinking_deepseek_model_does_not_honour_temperature( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - _stub_params(monkeypatch, _WITH_TEMPERATURE) - monkeypatch.setattr(litellm, "supports_reasoning", lambda **_: True) - assert honors_temperature("deepseek-flash") is False - - def test_non_thinking_deepseek_model_falls_back_to_the_probe( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - _stub_params(monkeypatch, _WITH_TEMPERATURE) - monkeypatch.setattr(litellm, "supports_reasoning", lambda **_: False) - assert honors_temperature("deepseek/deepseek-chat") is True -``` - -Append to `tests/unit/test_model_client.py` (import `litellm` at the top if it isn't already): - -```python -# --------------------------------------------------------------------------- -# DeepSeek thinking mode -# --------------------------------------------------------------------------- - -_REPLAYED_ROUND: list[dict[str, Any]] = [ - {"role": "user", "content": "find ACME"}, - { - "role": "assistant", - "content": "", - "tool_calls": [ - { - "id": "call_r0_0", - "type": "function", - "function": {"name": "search_db", "arguments": '{"query": "ACME"}'}, - } - ], - }, - {"role": "tool", "tool_call_id": "call_r0_0", "content": '{"rows": []}'}, -] - - -class TestDeepSeekReasoningBackfill: - async def test_replayed_assistant_turn_carries_placeholder_reasoning( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - # DeepSeek 400s a tools request whose earlier assistant turns lack - # reasoning_content; a teacher-forced round never has DeepSeek's own. - monkeypatch.setattr(litellm, "supports_reasoning", lambda **_: True) - captured = _patch_tools_acompletion(monkeypatch, _OPENAI_SINGLE_RESPONSE) - await ModelClient().complete_messages_with_tools( - model="deepseek-flash", - messages=[dict(m) for m in _REPLAYED_ROUND], - tools=[_DEMO_TOOL], - ) - sent = captured["kwargs"]["messages"] - assert sent[1]["reasoning_content"] == " " - assert "reasoning_content" not in sent[0] - - async def test_other_providers_are_sent_unchanged( - self, monkeypatch: pytest.MonkeyPatch - ) -> None: - monkeypatch.setattr(litellm, "supports_reasoning", lambda **_: True) - captured = _patch_tools_acompletion(monkeypatch, _OPENAI_SINGLE_RESPONSE) - await ModelClient().complete_messages_with_tools( - model="gpt-4o", - messages=[dict(m) for m in _REPLAYED_ROUND], - tools=[_DEMO_TOOL], - ) - assert captured["kwargs"]["messages"] == _REPLAYED_ROUND -``` - -- [ ] **Step 3: Run the tests to verify they fail** - -Run: `uv run pytest tests/unit/test_deepseek.py tests/unit/test_model_capabilities.py tests/unit/test_model_client.py -q` -Expected: `ModuleNotFoundError: evalshift_cli.models.deepseek`. The capability test returns `True`, and the client sends no `reasoning_content`. - -- [ ] **Step 4: Create `src/evalshift_cli/models/deepseek.py`** - -```python -"""DeepSeek API behaviour EvalShift has to account for, in one place. - -DeepSeek's current API models (``deepseek-flash``, ``deepseek-v4-pro``) run in -*thinking mode* unless the request turns it off -(https://api-docs.deepseek.com/guides/thinking_mode/). Thinking mode changes -two things the replay path depends on: - -* ``temperature`` is accepted and ignored — "setting them will not trigger an - error but will also have no effect". LiteLLM still lists ``temperature`` as - supported, so :func:`~evalshift_cli.models.capabilities.honors_temperature` - asks :func:`thinking_by_default` before trusting LiteLLM's answer, and the - report's non-determinism banner covers DeepSeek arms. -* A request carrying ``tools`` must pass back the ``reasoning_content`` of - every earlier assistant turn, or the API answers 400. A teacher-forced round - is rebuilt from the *recording* — often another model's — so there is no - DeepSeek reasoning to pass. :func:`backfill_reasoning_content` supplies the - single-space placeholder the API accepts. LiteLLM does the same, but only - when the caller passes ``thinking`` explicitly, which EvalShift never does - (a capture cannot record it); doing it here also keeps the fix independent - of the installed LiteLLM version. - -EvalShift never switches thinking off: the application under test runs with -DeepSeek's default, and replaying a different configuration would measure -the wrong thing. -""" - -from __future__ import annotations - -from collections.abc import Mapping, Sequence -from typing import Any, Final - -import litellm - -from evalshift_cli.models.registry import resolve_model - -#: The minimum ``reasoning_content`` DeepSeek accepts on an assistant turn — -#: an empty reasoning chain, the same value LiteLLM injects. -REASONING_PLACEHOLDER: Final = " " - - -def thinking_by_default(model_id: str) -> bool: - """Report whether ``model_id`` is a DeepSeek API model that thinks by default. - - Args: - model_id: Any user-supplied model id or alias; resolved first, so a - bare id recorded by a capture works. - - Returns: - ``True`` only for the ``deepseek`` provider when LiteLLM positively - reports reasoning support for the canonical id. Every other provider, - DeepSeek weights on another host, and every uncertain LiteLLM answer - return ``False`` — the same "uncertainty reads as honoured" rule as - :mod:`evalshift_cli.models.capabilities`. - """ - meta = resolve_model(model_id) - if meta.provider != "deepseek": - return False - try: - return bool(litellm.supports_reasoning(model=meta.id)) - except Exception: - return False - - -def backfill_reasoning_content(messages: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]: - """Return ``messages`` with :data:`REASONING_PLACEHOLDER` on bare assistant turns. - - Assistant turns that already carry a non-empty ``reasoning_content`` keep - it; every other role passes through. The input is not mutated. - """ - out: list[dict[str, Any]] = [] - for msg in messages: - if msg.get("role") == "assistant" and not msg.get("reasoning_content"): - out.append({**msg, "reasoning_content": REASONING_PLACEHOLDER}) - else: - out.append(dict(msg)) - return out - - -__all__ = ["REASONING_PLACEHOLDER", "backfill_reasoning_content", "thinking_by_default"] -``` - -- [ ] **Step 5: Wire it into capabilities and the client** - -In `capabilities.py`, import `from evalshift_cli.models.deepseek import thinking_by_default` and change `honors_temperature`'s body: - -```python - if thinking_by_default(model_id): - # Accepted and ignored by DeepSeek thinking mode; LiteLLM cannot see - # that. See evalshift_cli.models.deepseek. - return False - return not unsupported_params(model_id, ["temperature"]) -``` - -Extend its docstring's `Returns:` paragraph with one sentence naming this exception. Also add a bullet to the module docstring's list of exceptions to "LiteLLM's answer is the authority". - -In `client.py`, import `from evalshift_cli.models.deepseek import backfill_reasoning_content, thinking_by_default`. Insert this at the top of `_dispatch_with_retry`, before the existing `if canonical in self._temperature_rejected:` line: - -```python - if "messages" in kwargs and thinking_by_default(canonical): - # DeepSeek thinking mode 400s a tools request whose earlier - # assistant turns lack reasoning_content; see models/deepseek.py. - kwargs["messages"] = backfill_reasoning_content(kwargs["messages"]) -``` - -Add one sentence about this to the `_dispatch_with_retry` docstring's "adaptation" paragraph. - -- [ ] **Step 6: Run the tests, then the gate** - -Run: `uv run pytest tests/unit/test_deepseek.py tests/unit/test_model_capabilities.py tests/unit/test_model_client.py tests/unit/test_orchestrator.py tests/unit/test_tool_rounds.py -q`, then `make ci` -Expected: PASS. If an orchestrator test asserts on exact dispatched `messages` for a DeepSeek id, it doesn't exist today, so no fixture should change. Any diff means a non-DeepSeek path was touched, which is a bug. - -- [ ] **Step 7: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add src/evalshift_cli/models/deepseek.py src/evalshift_cli/models/capabilities.py src/evalshift_cli/models/client.py tests/unit/test_deepseek.py tests/unit/test_model_capabilities.py tests/unit/test_model_client.py -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "feat(models): account for DeepSeek thinking mode" -m "Thinking mode ignores temperature, so DeepSeek arms now get the non-determinism banner; replayed assistant turns get the placeholder reasoning_content DeepSeek requires on tools requests." -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A5: `evalshift init --provider deepseek` - -**Files:** -- Modify: `src/evalshift_cli/cli/commands/init.py:45` (`PROVIDERS`), `:51-70` (`_PROVIDER_MODELS`), `:83-90` (`_SEMANTIC_BLOCK_DISABLED`) -- Modify: `src/evalshift_cli/cli/commands/_scaffold.py:109-114` (`PROVIDER_API_KEY_ENVS`) -- Modify: `src/evalshift_cli/cli/commands/_agents.py:86-88` -- Test: `tests/unit/test_init.py` (`TestInitProvider`, the workflow class holding `test_provider_key_matches_provider`) - -**Interfaces:** -- Consumes: `PROVIDER_ENV_VARS`, `resolve_model` (A1) -- Produces: `PROVIDERS == ("gemini", "openai", "anthropic", "deepseek")`, which A6's docs test reads as `"|".join(PROVIDERS)` - -- [ ] **Step 1: Write the failing tests** - -In `TestInitProvider`, change the `test_every_provider_config_round_trips` loop to `for provider in PROVIDERS:`. Import `PROVIDERS` from `evalshift_cli.cli.commands.init`, and import `PROVIDER_API_KEY_ENVS` from `evalshift_cli.cli.commands._scaffold` if it isn't imported already. Then append: - -```python - def test_deepseek_provider_writes_deepseek_ids_and_comments_out_semantic( - self, in_tmp: Path - ) -> None: - result = runner.invoke(app, ["init", "--provider", "deepseek"]) - assert result.exit_code == 0, result.stdout - cfg = load_config(in_tmp / CONFIG_FILENAME) - assert cfg.defaults.source_model == "deepseek-flash" - assert cfg.evaluators.llm_judge[0].judge_model == "deepseek-v4-pro" - # DeepSeek has no embedding endpoint — semantic ships commented out. - assert cfg.evaluators.semantic is None - body = (in_tmp / CONFIG_FILENAME).read_text(encoding="utf-8") - assert "# semantic:" in body - assert "Anthropic has no embedding" not in body - assert "DEEPSEEK_API_KEY" in result.stdout - - def test_every_init_provider_key_is_the_registry_key(self) -> None: - # init and the run pre-check must agree on which env var authenticates - # a scaffold's models, or `init` tells the user to export the wrong one. - for provider in PROVIDERS: - source = _PROVIDER_MODELS[provider]["source_model"] - registry_provider = resolve_model(source).provider - assert PROVIDER_API_KEY_ENVS[provider] == PROVIDER_ENV_VARS[registry_provider][0] -``` - -Imports for that test: `_PROVIDER_MODELS` from `evalshift_cli.cli.commands.init`; `PROVIDER_ENV_VARS, resolve_model` from `evalshift_cli.models.registry`. - -In the workflow test class, next to `test_provider_key_matches_provider`, add: - -```python - def test_deepseek_workflow_uses_the_deepseek_key(self, in_tmp: Path) -> None: - body, _ = self._workflow(in_tmp, "--provider", "deepseek") - assert "DEEPSEEK_API_KEY" in body - assert "GEMINI_API_KEY" not in body -``` - -Also change the comment at `test_init.py:269` to `# No Anthropic embedding endpoint — semantic ships commented out.`, because the scaffold wording is changing. - -- [ ] **Step 2: Run the tests to verify they fail** - -Run: `uv run pytest tests/unit/test_init.py -q` -Expected: FAIL. `--provider deepseek` exits 2 ("unknown provider"), and `PROVIDER_API_KEY_ENVS` has no `"deepseek"` entry. - -- [ ] **Step 3: Implement** - -`init.py`: - -```python -PROVIDERS: Final = ("gemini", "openai", "anthropic", "deepseek") -``` - -Add to `_PROVIDER_MODELS`: - -```python - "deepseek": { - "source_model": "deepseek-flash", - "target_hint": "deepseek-v4-pro", - "judge_model": "deepseek-v4-pro", - "embedding_model": "", # no DeepSeek embedding endpoint - }, -``` - -Change the first two comment lines of `_SEMANTIC_BLOCK_DISABLED` to: - -``` - # Embedding-based drift score (advisory). This provider has no embedding - # endpoint — uncomment and set an OpenAI or Gemini embedding model (and -``` - -`_scaffold.py`: add `"deepseek": "DEEPSEEK_API_KEY",` to `PROVIDER_API_KEY_ENVS`. - -`_agents.py:87`: change it to `` (e.g. `GEMINI_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `DEEPSEEK_API_KEY`) matching the``. Re-wrap the line if ruff's line length requires it. - -- [ ] **Step 4: Find tests that pin the changed text** - -Run: `git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek grep -n "Anthropic has no embedding\|ANTHROPIC_API_KEY\`, \`OPENAI_API_KEY\`)" -- tests src` -Expected: no hits outside the lines just edited. Update any hit to the new wording. - -- [ ] **Step 5: Run the tests, then the gate** - -Run: `uv run pytest tests/unit/test_init.py tests/unit/test_agents.py -q`, then `make ci` -Expected: PASS. - -- [ ] **Step 6: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add src/evalshift_cli/cli/commands/init.py src/evalshift_cli/cli/commands/_scaffold.py src/evalshift_cli/cli/commands/_agents.py tests/unit/test_init.py -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "feat(init): scaffold DeepSeek projects with --provider deepseek" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A6: CLI docs, a docs-currency guard, CHANGELOG - -**Files:** -- Modify: `tests/unit/test_docs_currency.py` -- Modify: - - `README.md:302-306` - - `DOCS.md:72-80, 136, 385, 803, 888-890` + a new FAQ answer after `:931` - - `llms-full.txt:120-121, 425-427, 615-618` + the same FAQ answer - - `docs/getting-started.md:43-48` - - `docs/faq.md:50-56` + new question - - `docs/configuration.md:595-597` - - `pyproject.toml:14` - - `CHANGELOG.md` - -**Interfaces:** -- Consumes: `PROVIDER_ENV_VARS` (A1), `PROVIDERS` (A5) - -- [ ] **Step 1: Write the failing guard** - -Append to `tests/unit/test_docs_currency.py` (with the imports at the top of the file): - -```python -from evalshift_cli.cli.commands.init import PROVIDERS -from evalshift_cli.models.registry import PROVIDER_ENV_VARS - -#: Files that tell a user which env var authenticates which provider. A -#: provider the registry can authenticate but these files never name is a -#: provider whose users are told nothing — the state DeepSeek support shipped -#: into on 2026-09-30. -KEY_TABLE_FILES: tuple[str, ...] = ("DOCS.md", "llms-full.txt", "docs/getting-started.md") - -#: Files that spell out `init --provider`'s choices. -INIT_PROVIDER_FILES: tuple[str, ...] = ("DOCS.md", "llms-full.txt") - - -@pytest.mark.parametrize("name", KEY_TABLE_FILES) -@pytest.mark.parametrize("env_var", sorted(aliases[0] for aliases in PROVIDER_ENV_VARS.values())) -def test_key_docs_name_every_registry_provider(name: str, env_var: str) -> None: - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert env_var in text, f"{name} never tells users about {env_var}" - - -@pytest.mark.parametrize("name", INIT_PROVIDER_FILES) -def test_docs_list_every_init_provider(name: str) -> None: - text = (REPO_ROOT / name).read_text(encoding="utf-8") - - assert f"--provider {'|'.join(PROVIDERS)}" in text, ( - f"{name} lists `init --provider` choices that differ from init.PROVIDERS" - ) -``` - -- [ ] **Step 2: Run it to verify it fails** - -Run: `uv run pytest tests/unit/test_docs_currency.py -q` -Expected: FAIL on `DEEPSEEK_API_KEY` in all three files, and on `--provider gemini|openai|anthropic|deepseek` in both. - -- [ ] **Step 3: Enumerate every site first** - -Run: `git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek grep -nE "GEMINI_API_KEY|gemini\|openai\|anthropic|Claude, GPT|\(\`anthropic\`|gemini-\*. → Google|Anthropic, OpenAI and Google" -- README.md DOCS.md llms-full.txt docs AGENTS.md` -Expected: the sites listed in this task's **Files** block. Treat any extra hit as in scope unless it is a Gemini-only example (`examples/`, the `init --provider gemini` sample output) or tool/wire-shape prose. - -- [ ] **Step 4: Edit the key and choice sites** - -Use these exact edits: - -- `DOCS.md:72-80`: add `export DEEPSEEK_API_KEY=...` as a fourth line in the export block. -- `DOCS.md:888-890`: add the row ``| `DEEPSEEK_API_KEY` | — | DeepSeek auth |``. -- `llms-full.txt:425-427`: add the row `| DEEPSEEK_API_KEY | — | DeepSeek auth |`. -- `docs/getting-started.md:46-48`: add `export DEEPSEEK_API_KEY=`. -- `DOCS.md:136`: replace the `--provider` bullet with: - - ``- `--provider gemini|openai|anthropic|deepseek` — which provider's model ids the scaffold uses (prompted on a TTY; defaults to `gemini` otherwise). Gemini and OpenAI scaffolds include an embedding-based semantic evaluator; the Anthropic and DeepSeek scaffolds comment it out (no embedding endpoint).`` -- `DOCS.md:803`: `--provider gemini|openai|anthropic` → `--provider gemini|openai|anthropic|deepseek`. -- `llms-full.txt:121`: `[--provider gemini|openai|anthropic]` → `[--provider gemini|openai|anthropic|deepseek]`. - -- [ ] **Step 5: Edit the provider-scope prose (alphabetical, see Design decision 7)** - -- `README.md:304`: `Anthropic, OpenAI and Google ids additionally get a curated pricing and` → `Anthropic, DeepSeek, Google and OpenAI ids additionally get a curated pricing and`. -- `docs/faq.md:51-52`: `(Claude, GPT,` / `Gemini)` → `(Claude, DeepSeek,` / `Gemini, GPT)`. -- `DOCS.md:385` and `llms-full.txt:617`: in the prefix-inference list, add `` `deepseek-*` → DeepSeek`` in DOCS.md and `deepseek-* -> deepseek` in llms-full.txt, after the OpenAI entry. -- `docs/configuration.md:596`: `(`anthropic`, `openai`, `google`)` → `(`anthropic`, `deepseek`, `google`, `openai`)`. -- `pyproject.toml:14`: add `"deepseek"` to `keywords` after `"gemini"`. - -- [ ] **Step 6: Add the DeepSeek answer** - -Add this as a new question in `docs/faq.md`, placed after "what models does EvalShift support?". Add the same text to `DOCS.md` after line 931's answer, as a `###`-level entry matching its neighbours. Add it to `llms-full.txt` after line 618, re-wrapped in that file's terse style but keeping every fact: - -```markdown -### Does EvalShift work with DeepSeek? - -Yes. Export `DEEPSEEK_API_KEY` and use DeepSeek's API ids, `deepseek-flash` -or `deepseek-v4-pro`. A bare `deepseek-*` id (what a capture records when your -app calls `api.deepseek.com` through the OpenAI client) gets the `deepseek/` -prefix automatically. `evalshift init --provider deepseek` scaffolds a -DeepSeek project. Three things differ from other providers: - -- **Sampling is not controlled.** Both models run in thinking mode by - default, which accepts `temperature` and ignores it. EvalShift keeps thinking - on, because that is what your application runs, so DeepSeek arms are marked - non-deterministic in the report. Raise `defaults.samples_per_example` when - the verdict matters. -- **Replayed tool rounds carry an empty reasoning chain.** DeepSeek requires - the `reasoning_content` of earlier assistant turns on any request with - tools. A teacher-forced round comes from the recording, not from DeepSeek, - so EvalShift sends the single-space placeholder the API accepts. -- **No embeddings.** DeepSeek has no embedding endpoint. The `semantic` - evaluator needs an OpenAI or Gemini embedding model and its key, which is - why the DeepSeek scaffold ships it commented out. - -DeepSeek served by another host (self-hosted open weights, or a cloud region -of your choice) goes through that host's LiteLLM prefix (`hosted_vllm/`, -`azure_ai/`, `bedrock/`, ...) and its environment variables. Tool calls parse -the same way, but the key pre-check and the notes above apply to the -`deepseek/` API only. LiteLLM also reads `DEEPSEEK_API_BASE` to point the -`deepseek/` provider at a DeepSeek-compatible endpoint. -``` - -- [ ] **Step 7: Add the CHANGELOG entry** - -Under `## [Unreleased]`, add an `### Added` section above the existing `### Fixed` if there is none: - -```markdown -### Added - -- DeepSeek is a supported provider. `deepseek-flash` and `deepseek-v4-pro` - are in the model registry; bare `deepseek-*` ids (as recorded by a capture - of an app calling `api.deepseek.com` through the OpenAI client) resolve to - `deepseek/…`; `DEEPSEEK_API_KEY` is checked before a run and shown by - `evalshift doctor`; tool-call evals parse DeepSeek responses; a DeepSeek - judge grading a DeepSeek arm gets the judge-family warning; and - `evalshift init --provider deepseek` scaffolds a DeepSeek project. DeepSeek's - default thinking mode ignores `temperature`, so DeepSeek arms carry the - report's non-determinism banner, and replayed assistant turns are sent with - the placeholder `reasoning_content` DeepSeek requires on tool requests. - -### Fixed - -- `capture sync` priced calls recorded under a bare id at $0 when LiteLLM - keys the model under both spellings but can only price the provider-prefixed - one (DeepSeek). The price lookup now tries the provider-prefixed id first. -``` - -Merge the Fixed bullet into the existing `### Fixed` list rather than adding a second heading. - -- [ ] **Step 8: Run the docs tests, then the gate** - -Run: `uv run pytest tests/unit/test_docs_currency.py tests/unit/test_init.py -q`, then `make ci` -Expected: PASS, including `test_provider_scope_is_not_capped_at_three`, since the alphabetical lists don't match its regex. - -- [ ] **Step 9: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add tests/unit/test_docs_currency.py README.md DOCS.md llms-full.txt docs pyproject.toml CHANGELOG.md -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "docs: document DeepSeek and guard key/provider lists against the registry" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A7: Live verification with a real DeepSeek key (gate for the PR) - -This task needs `DEEPSEEK_API_KEY`, which the maintainer supplies. It costs a few cents. **Do not open the PR until every check below passes.** If one fails, stop and report the exact error; do not work around it. - -**Files:** -- Modify: `scripts/smoke_live_tools.py:74-77` (`MODELS`), `:105-113` (`_provider_or_skip`), plus a new multi-round check and a text-only chat-history check - -- [ ] **Step 1: Make the smoke script provider-generic and add a multi-round check** - -Replace `_provider_or_skip` so it reads keys from the registry and keeps DeepSeek fixtures out of `openai/`: - -```python -def _provider_or_skip(model: str) -> str | None: - """Return the fixture directory for ``model`` iff its provider's key is set.""" - meta = resolve_model(model) - keys = PROVIDER_ENV_VARS.get(meta.provider, ()) - if not any(os.environ.get(k) for k in keys): - return None - # detect_provider names a response shape; DeepSeek shares OpenAI's, but its - # live captures must not overwrite the OpenAI ones. - return "deepseek" if meta.provider == "deepseek" else detect_provider(model) -``` - -Import `from evalshift_cli.models.registry import PROVIDER_ENV_VARS, resolve_model # noqa: E402`. Set: - -```python -MODELS = [ - "gemini/gemini-2.5-flash", - "gemini/gemini-3.1-flash-lite-preview", - "deepseek-flash", - "deepseek-v4-pro", -] -``` - -Add this multi-round check, called once per model at the end of the model's loop body inside `main()`. It counts as a failure on exception: - -```python -async def _multi_round(client: ModelClient, model: str) -> None: - """Round 1 of a teacher-forced replay: an assistant turn DeepSeek never wrote.""" - messages: list[dict[str, Any]] = [ - {"role": "user", "content": "Look up ACME's Q3 revenue."}, - { - "role": "assistant", - "content": "", - "tool_calls": [ - { - "id": "call_r0_0", - "type": "function", - "function": {"name": "search_db", "arguments": '{"query": "ACME Q3"}'}, - } - ], - }, - {"role": "tool", "tool_call_id": "call_r0_0", "content": '{"revenue_musd": 12.4}'}, - ] - result = await client.complete_messages_with_tools(model=model, messages=messages, tools=TOOLS) - print(f" round 1: calls={result.trace.tool_names} text={bool(result.trace.final_text)}") -``` - -After it, also once per model and counted the same way (` history FAILED: ...` on exception), add a text-only chat-history check. The backfill puts the placeholder on this assistant turn too, and there are no tools: - -```python -async def _chat_history(client: ModelClient, model: str) -> None: - """A text-only replayed history: an assistant turn with no reasoning_content.""" - messages: list[dict[str, Any]] = [ - {"role": "user", "content": "Hi, I ordered a kettle last week."}, - {"role": "assistant", "content": "Thanks! How can I help with your kettle order?"}, - {"role": "user", "content": "What's your standard refund policy, in one sentence?"}, - ] - result = await client.complete_messages(model=model, messages=messages) - print(f" history: text={bool(result.text)}") -``` - -- [ ] **Step 2: Run the smoke script** - -Run: `DEEPSEEK_API_KEY=... uv run python scripts/smoke_live_tools.py` (the Gemini models skip when their key is unset) -Expected, for both `deepseek-flash` and `deepseek-v4-pro`: -- `single_tool` / `parallel` prompts print non-empty `calls:` and a cost above `$0.000000`. -- `text_only` prints no calls. -- `round 1:` prints with **no 400**. This is the live proof of F8's fix. -- `history: text=True` prints with no error. This proves the placeholder is harmless on a text-only request. -- New files appear under `tests/unit/fixtures/tool_responses/deepseek/*_live.json`. - -- [ ] **Step 3: Check the 400 really is what the backfill prevents** - -Run this once with the backfill disabled. Temporarily comment out the three inserted lines in `_dispatch_with_retry`, rerun only `deepseek-flash`, and confirm `round 1` FAILS with DeepSeek's "reasoning_content … must be passed back" 400. Then restore the lines and confirm it passes again. -Expected: FAIL without the backfill, PASS with it. If it passes without the backfill too, DeepSeek has relaxed the rule. Keep the backfill anyway (it's harmless), but record the observation in the PR description. - -- [ ] **Step 4: End-to-end through the CLI** - -Run: -```bash -cd /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek -DEEPSEEK_API_KEY=... uv run evalshift test-call --model deepseek-flash --max-tokens 2048 -cd examples/agent && DEEPSEEK_API_KEY=... GEMINI_API_KEY=... uv run evalshift compare --from deepseek-flash --to deepseek-v4-pro --yes -``` -Expected: -- `test-call` prints a reply and a non-zero cost. -- `compare` finishes and writes a report. -- The report shows the non-determinism banner naming both DeepSeek arms. -- `report.json`'s tool-call scores are populated, not errored. -- `evalshift doctor` in the same shell shows `DEEPSEEK_API_KEY set`. - -Then check a DeepSeek judge end to end. Copy `examples/agent` to a scratch directory, add an `llm_judge` entry with `judge_model: deepseek-v4-pro` (the judge `init --provider deepseek` scaffolds), and rerun the same `compare`. -Expected: -- The judge's verdicts are populated in `report.json`, not errored or unmeasured. -- The non-determinism banner also names `deepseek/deepseek-v4-pro` as the judge. It must be listed once, even though it is also an arm. - -- [ ] **Step 5: Commit the script and the live fixtures** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek add scripts/smoke_live_tools.py tests/unit/fixtures/tool_responses/deepseek -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek commit -m "test(smoke): cover DeepSeek and a replayed tool round in the live smoke script" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task A8: Open the CLI PR - -- [ ] **Step 1: Push and open the PR** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek push -u origin feat/deepseek-provider -cd /home/lukas/repos/evalshift/evalshift-cli-wt-deepseek && gh pr create --title "feat: DeepSeek as a supported provider" --body "$(cat <<'EOF' -Adds DeepSeek (`deepseek-flash`, `deepseek-v4-pro`) as a first-class provider: registry + key pre-check + doctor row, tool-call parsing, judge-family detection, `init --provider deepseek`, correct pricing of bare recorded ids, and handling for DeepSeek's default thinking mode (non-determinism banner; placeholder `reasoning_content` on replayed tool rounds, which DeepSeek otherwise rejects with a 400). - -Plan: `docs/superpowers/plans/2026-09-30-deepseek-provider.md`. Live-verified against the DeepSeek API (Task A7): . - -🤖 Generated with [Claude Code](https://claude.com/claude-code) -EOF -)" -``` - ---- - -## Phase B: evalshift-sdk (docs only) - -The SDK needs no code change. `wrap_openai` records whatever `model` the app sent, and Phase A makes that bare id replayable. - -### Task B1: Document DeepSeek capture - -**Files:** `README.md:66-68`, `DOCS.md:794`, `llms-full.txt:307-308, 501`, `CHANGELOG.md` - -- [ ] **Step 1: Branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek fetch origin -git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek switch -c docs/deepseek origin/main -``` - -- [ ] **Step 2: Enumerate the sites** - -Run: `git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek grep -n "Groq" -- README.md DOCS.md llms-full.txt llms.txt docs` -Expected: `README.md:68`, `DOCS.md:794`, `llms-full.txt:307-308, 501`, and `docs/DECISIONS.md:506`, which is a historical decision record: leave it. - -- [ ] **Step 3: Edit** - -- `README.md:68` comment: `# OpenAI(base_url=...) covers DeepSeek, Ollama, vLLM, Groq, ...` -- `DOCS.md:794`: insert DeepSeek at the start of the list (`DeepSeek, Ollama, vLLM, llama.cpp server, …`). Append this sentence to the paragraph: ``For DeepSeek, `wrap_openai(OpenAI(base_url="https://api.deepseek.com", api_key=os.environ["DEEPSEEK_API_KEY"]))` records `model_id` as the bare id you pass (`deepseek-flash`); the CLI (1.2.0+) resolves it to `deepseek/deepseek-flash` on replay.`` -- `llms-full.txt:307-308`: add `DeepSeek,` to the parenthesised list. `:501`: `# OpenAI(base_url=...) for DeepSeek / Ollama / vLLM / Groq ...`. - -- [ ] **Step 4: CHANGELOG** - -Add under `## [Unreleased]` → `### Changed` (create the heading if needed): - -```markdown -- Docs: DeepSeek is listed among the OpenAI-compatible APIs `wrap_openai` - captures unchanged, with the exact client construction. Replaying those - captures needs EvalShift CLI 1.2.0 or later. -``` - -- [ ] **Step 5: Run the SDK gate, commit, push, open a PR** - -Run (from `evalshift-sdk/`, mirroring `.github/workflows/ci.yml`): `uv run ruff check && uv run ruff format --check && uv run mypy && uv run pytest` -Expected: all green (docs-only change; this is a sanity run). - -```bash -git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek add README.md DOCS.md llms-full.txt CHANGELOG.md -git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek commit -m "docs: capture DeepSeek through wrap_openai" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -git -C /home/lukas/repos/evalshift/evalshift-sdk-wt-deepseek push -u origin docs/deepseek -``` - -Open the PR with `gh pr create` (body ends with the Claude Code line). **Merge it after the CLI 1.2.0 release**, because it names that version. - ---- - -## Phase C: evalshift-action (docs + one pinning test) - -`action.yml` has no provider inputs. Keys flow through the job `env:`, and `_secret_values` (`scripts/evalshift_action.py:826-834`) already redacts anything ending in `API_KEY`. - -### Task C1: Provider key tables and a redaction test - -**Files:** `README.md:137-160`, `DOCS.md:204-225`, `llms-full.txt:515-526`, `tests/test_evalshift_action.py` - -- [ ] **Step 1: Branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek fetch origin -git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek switch -c docs/deepseek-key origin/main -``` - -- [ ] **Step 2: Write the pinning test** - -Append to `tests/test_evalshift_action.py`: - -```python -def test_provider_keys_are_redacted_including_deepseek() -> None: - # Keys reach the CLI through the job env untouched; the log redactor - # matches on the `_API_KEY` suffix, so a new provider needs no code change. - env = { - "ANTHROPIC_API_KEY": "sk-ant-secret", - "DEEPSEEK_API_KEY": "sk-deepseek-secret", - "HOME": "/home/runner", - } - assert sorted(action._secret_values(env)) == ["sk-ant-secret", "sk-deepseek-secret"] -``` - -Run (from `evalshift-action/`): `uv run pytest tests/test_evalshift_action.py -q -k deepseek` -Expected: PASS immediately. This test pins existing behaviour; it does not drive a change. - -- [ ] **Step 3: Edit the three tables** - -After the Google row in each file, add: -- `README.md` (`## Model provider API keys` table): ``| DeepSeek | `DEEPSEEK_API_KEY` |`` -- `DOCS.md:211-215`: the same row. -- `llms-full.txt:517-521`: `| DeepSeek | DEEPSEEK_API_KEY |` - -Keep the column padding consistent with neighbouring rows. Then run `git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek grep -n "GEMINI_API_KEY or GOOGLE_API_KEY"` and confirm all three tables were covered. - -- [ ] **Step 4: Run the action's gate, commit, push, open the PR** - -Run (mirroring `.github/workflows/ci.yml`): `uv run pytest && uv run ruff check .` - -```bash -git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek add README.md DOCS.md llms-full.txt tests/test_evalshift_action.py -git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek commit -m "docs: list DEEPSEEK_API_KEY among provider keys" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -git -C /home/lukas/repos/evalshift/evalshift-action-wt-deepseek push -u origin docs/deepseek-key -``` - -**Merge it after** the automatic CLI-pin PR has moved the action to CLI ≥ 1.2.0. Before that, the pinned CLI cannot run DeepSeek tool evals. - ---- - -## Phase D: evalshift-client (landing + site docs) - -Ship **after CLI 1.2.0 is on PyPI** (D4). Stack: Vite + React 19 + TS; tests are Vitest next to the source. - -### Task D1: Landing copy - -**Files:** `src/pages/landing/sections/Hero.tsx:112-114`, `src/pages/landing/Landing.test.tsx:253-255` - -- [ ] **Step 1: Branch** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek fetch origin -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek switch -c feat/deepseek-copy origin/main -``` - -- [ ] **Step 2: Update the test first** - -`Landing.test.tsx:253-255`: - -```tsx - expect( - screen.getByText(/Works with Anthropic, OpenAI, Google and DeepSeek models\./), - ).toBeInTheDocument() -``` - -Run: `npx vitest run src/pages/landing/Landing.test.tsx` -Expected: FAIL, because the text is not found. - -- [ ] **Step 3: Update the copy (D2)** - -`Hero.tsx:113`: `Works with Anthropic, OpenAI, Google and DeepSeek models.` - -Run: `npx vitest run src/pages/landing/Landing.test.tsx` -Expected: PASS. - -- [ ] **Step 4: Commit** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek add src/pages/landing/sections/Hero.tsx src/pages/landing/Landing.test.tsx -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek commit -m "feat(landing): list DeepSeek among supported providers" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - -### Task D2: Site docs pages - -These are hand-written TSX mirrors of the CLI/SDK/action docs. No guard exists, so enumerate first. - -**Files:** `src/pages/docs/pages/{CliCommands,Cli,Faq,Action,Configuration,GettingStarted,SdkAdapters}.tsx` - -- [ ] **Step 1: Enumerate** - -Run: `git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek grep -nE "GEMINI_API_KEY|gemini\|openai\|anthropic|Claude, GPT|gpt-\*|\(anthropic|Ollama, vLLM" -- src/pages/docs` -Expected: the sites below. Leave `Changelog.tsx` alone (Task E1 handles it), and leave the tool/wire-shape pages (`SdkCapture.tsx`, `GoldenSuite.tsx`, `Agents.tsx`). - -- [ ] **Step 2: Edit** - -- `CliCommands.tsx:11`: `--provider gemini|openai|anthropic` → `--provider gemini|openai|anthropic|deepseek` -- `CliCommands.tsx:127-130` (`ENV_VARS`): add `{ name: "DEEPSEEK_API_KEY", detail: "DeepSeek auth" },` after the Anthropic entry. -- `Cli.tsx:221-222`: after `gpt-*/o1-*/o3-* → OpenAI`, add `, deepseek-* → DeepSeek` inside the parenthesis. -- `Faq.tsx:112`: `(Claude, GPT, Gemini)` → `(Claude, DeepSeek, Gemini, GPT)`. Also add a "Does EvalShift work with DeepSeek?" entry mirroring the CLI FAQ answer from Task A6 Step 6, in this file's existing Q/A component shape. -- `Action.tsx:183-185`: add `, DEEPSEEK_API_KEY` after `GEMINI_API_KEY/GOOGLE_API_KEY`. -- `Configuration.tsx:690-691`: `(anthropic, openai, google)` → `(anthropic, deepseek, google, openai)`. -- `GettingStarted.tsx:25`: `# or OPENAI_API_KEY / ANTHROPIC_API_KEY` → `# or OPENAI_API_KEY / ANTHROPIC_API_KEY / DEEPSEEK_API_KEY`. Keep the column alignment of the code block. -- `SdkAdapters.tsx:24`: `(Ollama, vLLM, Groq, OpenRouter, ...)` → `(DeepSeek, Ollama, vLLM, Groq, OpenRouter, ...)`. At `:205-206`, add DeepSeek at the start of the same list. - -- [ ] **Step 3: Run the client gate** - -Run: `npm run lint && npm run typecheck && npx vitest run && npm run build` -Expected: all green. Prerender succeeds for the docs routes. - -- [ ] **Step 4: Commit, push, PR** - -```bash -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek add src/pages/docs/pages -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek commit -m "docs(site): document DeepSeek keys, ids and init choice" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek push -u origin feat/deepseek-copy -``` - -Open the PR with `gh pr create`. The body ends with the Claude Code line. Vercel deploys from `main` on merge. - ---- - -## Phase E: Release and sync (maintainer-driven) - -### Task E1: Release - -Order: SDK → CLI → action pin PR → client llms sync. - -- [ ] **Step 1: CLI 1.2.0.** After the A8 PR merges, cut the release commit the usual way (`chore(release): 1.2.0`). It bumps `pyproject.toml` and moves `## [Unreleased]` to `## [1.2.0]`. Tag it; the trusted-publishing workflow ships it. -- [ ] **Step 2: Action.** Wait for `bump-cli-pin.yml`'s daily PyPI poll (04:23 UTC) to open the pin PR, then merge it. Then merge the C1 PR. -- [ ] **Step 3: SDK.** Merge the B1 PR. No PyPI release is needed (D4). -- [ ] **Step 4: Client changelog.** In the D PR (or a follow-up), add the 1.2.0 headline to `src/pages/docs/pages/Changelog.tsx` and bump `src/lib/version.ts`. This follows the precedent of client commit `b1330d9` ("advertise CLI 1.1.0"). - -### Task E2: Refresh the site's llms mirrors - -- [ ] **Step 1: Pull all three source repos to their merged `main`, then sync** - -```bash -for r in evalshift-cli evalshift-sdk evalshift-action; do git -C /home/lukas/repos/evalshift/$r switch main && git -C /home/lukas/repos/evalshift/$r reset --hard origin/main; done -cd /home/lukas/repos/evalshift/evalshift-client-wt-deepseek && npm run sync:llms -grep -c DEEPSEEK_API_KEY public/cli-llms-full.txt public/ci-llms-full.txt -grep -c DeepSeek public/sdk-llms-full.txt -``` - -Expected: every count is ≥ 1. - -- [ ] **Step 2: Commit and push to `main`** (the established flow for the llms sync) - -```bash -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek add public/cli-llms-full.txt public/sdk-llms-full.txt public/ci-llms-full.txt -git -C /home/lukas/repos/evalshift/evalshift-client-wt-deepseek commit -m "docs: sync llms-full mirrors for DeepSeek support" -m "Co-Authored-By: Claude Opus 5.5 (1M context) " -``` - ---- - -## Out of scope (follow-ups worth filing) - -- **Keeping a capture's real `reasoning_content`.** Captures of DeepSeek apps carry the model's own reasoning on assistant turns. The CLI's `ChatMessage` drops it at promotion, so replayed *history* gets the placeholder too. Preserving it would change the suite schema. -- **`thinking` / `reasoning_effort` in the SDK's `GENERATION_KEYS`.** This would let a replay honour an app that turned thinking off. The allow-list is frozen public API ("widening this is a deliberate decision"). -- **`top_p` below 0.95 is clamped in thinking mode.** This is a value constraint the dropped-params probe cannot see, like `temperature`'s reasoning-tier rejections. -- **A per-model `api_base` in `evalshift.yaml`** for EU/self-hosted endpoints. Today it works through LiteLLM's env vars (`DEEPSEEK_API_BASE`, `HOSTED_VLLM_API_BASE`, ...), which Task A6 documents. -- **Stale server README** (`evalshift-server/README.md:163-164` references a nonexistent `app/evaluations/runner.py`). Unrelated to providers; noted by the sweep. diff --git a/docs/superpowers/specs/2026-06-09-cli-agent-traces-design.md b/docs/superpowers/specs/2026-06-09-cli-agent-traces-design.md deleted file mode 100644 index 33e0a17..0000000 --- a/docs/superpowers/specs/2026-06-09-cli-agent-traces-design.md +++ /dev/null @@ -1,332 +0,0 @@ -# CLI Agent Traces Design - -## Summary - -Phase 2 adds CLI-only support for bring-your-own-agent trace evaluation. EvalShift should accept externally recorded source and target agent traces, normalize them into local run artifacts, compare the traces with the same evaluator/analyze/report pipeline used today, and surface actionable trace diffs in debug commands and the offline HTML report. - -This phase does not add SDKs, hosted features, web dashboards, production observability, or an agent runtime. EvalShift remains the local migration safety layer. - -## Goals - -- Define a first-class agent trace schema for local CLI artifacts. -- Import and validate external trace JSONL files for completed runs. -- Compare source and target agent timelines per `(prompt_id, example_id)`. -- Emit normal `EvalRecord` rows so existing analysis, migration policy, and reports continue to work. -- Extend `diff case`, `inspect`, `report.json`, and `report.html` to explain trace regressions. -- Preserve backward compatibility for existing prompt/output and current `ToolTrace` workflows. - -## Non-Goals - -- No Python SDK. -- No TypeScript SDK. -- No hosted/web implementation. -- No OpenTelemetry exporter implementation. -- No agent builder, workflow runner, or tool execution runtime. -- No generic observability ingestion outside migration test runs. - -## Current Repo Context - -The repo already has provider-level tool-call traces: - -- `src/evalshift/evaluators/tool_models.py` defines `ToolCall` and `ToolTrace`. -- `src/evalshift/runner/models.py` stores `Call.trace` in `raw.jsonl` for tool-aware prompts. -- `tool_selection`, `tool_arguments`, and `tool_trace_structure` compare provider-parsed tool traces. -- `reports/json.py` and `report.html.j2` can render side-by-side tool trace diffs. -- `inspect`, `diff case`, and `replay case` provide recorded-artifact debugging. -- Migration policy and failure taxonomy work already exists in `analysis/policy.py` and evaluator metadata. - -Phase 2 should extend these capabilities to multi-event agent traces without replacing them. - -## Trace Artifact Model - -Add a new package: - -```text -src/evalshift/traces/ - __init__.py - models.py - loader.py - diff.py -``` - -`models.py` defines strict Pydantic models: - -- `AgentTrace` -- `TraceEvent` -- `ModelCallEvent` -- `ToolCallEvent` -- `ToolResultEvent` -- `RetrievalEvent` -- `GuardrailEvent` -- `FinalOutputEvent` -- `ErrorEvent` - -Each imported trace belongs to one side of one example: - -```json -{ - "run_id": "r_20260609_ab12cd", - "prompt_id": "support_agent", - "example_id": "refund_017", - "role": "target", - "events": [] -} -``` - -Each event uses a discriminated `type` field and stable ordering: - -```json -{ - "type": "tool_call", - "name": "issue_refund", - "arguments": {"ticket_id": "T-1032", "amount": 42.0}, - "sequence_index": 3, - "timestamp": "2026-06-09T12:00:00Z", - "metadata": {"step": "refund"} -} -``` - -Required event fields: - -- `type` -- `sequence_index` -- `timestamp` -- `metadata` - -Event-specific fields: - -- `model_call`: `model_id`, `input`, `output`, `input_tokens`, `output_tokens`, `cost_usd`, `latency_ms` -- `tool_call`: `name`, `arguments`, `call_id`, `parent_call_id` -- `tool_result`: `name`, `call_id`, `result`, `error` -- `retrieval`: `source`, `query`, `documents` -- `guardrail`: `name`, `verdict`, `reason` -- `final_output`: `text` -- `error`: `message`, `category` - -Validation rules: - -- `sequence_index` values are unique and non-negative within a trace. -- Events are normalized into ascending `sequence_index` order. -- `tool_result.call_id`, when present, should reference a previous `tool_call.call_id`. -- `role` must be `source` or `target`. -- `prompt_id` and `example_id` must match examples in the run being imported. -- Unknown fields are rejected so trace contract drift is visible. - -## Import Command - -Add a `traces` Typer sub-app: - -```bash -evalshift traces import --source source-traces.jsonl --target target-traces.jsonl -``` - -Command behavior: - -1. Load `.evalshift/runs//state.json` and `raw.jsonl`. -2. Validate both JSONL files line by line. -3. Ensure every imported trace references known `prompt_id` and `example_id` values from the run. -4. Ensure imported `role` values match the side implied by `--source` or `--target`. -5. Write a normalized append-free artifact: - -```text -.evalshift/runs//traces.jsonl -``` - -6. Print an import summary: - -```text -Imported traces for run r_... -source: 40 traces -target: 40 traces -missing pairs: 0 -artifact: .evalshift/runs//traces.jsonl -``` - -Failure behavior: - -- Invalid JSONL reports file path and line number. -- Schema validation errors are grouped and rendered similarly to config/suite errors. -- Missing source/target pairs warn by default and fail only when `--strict` is passed. -- Import never mutates `raw.jsonl`, `scores.jsonl`, or `analysis.json`. - -## Trace Evaluation - -Add an evaluator family: - -```yaml -evaluators: - agent_trace: - - name: trace_safety - check_tool_order: true - check_arguments: true - check_missing_verification: true - verification_tools: ["check_refund_policy", "verify_order_status"] - dangerous_tools: ["issue_refund", "delete_record", "send_email"] -``` - -Add config models in `src/evalshift/config/models.py`: - -- `AgentTraceEvaluatorConfig` -- `evaluators.agent_trace: list[AgentTraceEvaluatorConfig] | None` - -Evaluation behavior: - -- `run_evaluate` loads `traces.jsonl` when `agent_trace` evaluators are configured. -- Each source/target trace pair produces an `EvalRecord`. -- Records are written to existing `scores.jsonl`; no new analysis stage is needed. -- Existing `analyze`, migration policy, and report severity classification remain the statistical layer. - -Scoring dimensions: - -- Tool sequence similarity. -- Missing expected verification before dangerous tool calls. -- Extra dangerous tool calls. -- Argument value drift for same-name matched tool calls. -- Tool result error drift. -- Final output presence and refusal-like error changes. - -Failure categories: - -- `TOOL_SELECTION_DRIFT` -- `TOOL_ORDER_DRIFT` -- `ARGUMENT_VALUE_DRIFT` -- `DANGEROUS_ACTION_DRIFT` -- `MISSING_VERIFICATION_STEP` -- `UNNECESSARY_TOOL_CALL` -- `TOOL_RESULT_DRIFT` -- `TRACE_SCHEMA_FAILURE` - -These should integrate with the existing failure taxonomy metadata used by reports. - -## Trace Diffing - -Add reusable trace diff logic in `src/evalshift/traces/diff.py`. - -Output model: - -- Matched events. -- Missing source events. -- Extra target events. -- Reordered tool calls. -- Argument field deltas. -- Verification gaps. -- Dangerous action flags. - -Use this model in: - -- `evalshift diff case ` -- report payload assembly -- HTML rendering -- tests - -The CLI diff should render a compact timeline: - -```text -source target -1 model_call claude... 1 model_call gpt... -2 tool_call check_policy - missing -3 tool_call issue_refund 2 tool_call issue_refund ARGUMENT_VALUE_DRIFT -4 final_output 3 final_output -``` - -## Report Changes - -Extend `reports/json.py`: - -- Load `traces.jsonl` if present. -- Attach `source_agent_trace` and `target_agent_trace` to top regression rows. -- Attach computed `trace_diff` to rows from `agent_trace` evaluators. -- Include aggregate top regression causes from agent trace metadata. - -Extend `report.html.j2`: - -- Render trace timelines for agent trace regressions. -- Show missing/extra/reordered events with clear labels. -- Show argument drift inline with field names and source/target values. -- Keep existing `ToolTrace` rendering for current tool-call evaluators. - -HTML remains single-file, no external assets, no JavaScript. - -## Debug Command Changes - -`evalshift inspect`: - -- `evalshift inspect --failed` should include agent trace failures. -- `evalshift inspect case ` should show trace summary when available. - -`evalshift diff case`: - -- Prefer agent trace diff when `traces.jsonl` has a pair for the case. -- Fall back to current text diff when no agent trace exists. - -`evalshift replay case`: - -- Keep current recorded-output behavior. -- Add `--trace` to print the imported target/source trace as normalized JSON. - -## Compatibility - -Existing users without `traces.jsonl` are unaffected. - -Current provider-level `ToolTrace` stays in `Call.trace`. Agent traces are broader user-provided timelines. Where useful, the current `ToolTrace` can be adapted into an `AgentTrace` internally for diff rendering, but that adapter should not change the persisted `raw.jsonl` format. - -`agent_trace` evaluators require imported traces. If configured and `traces.jsonl` is missing, `evaluate` should fail with a clear message: - -```text -agent_trace evaluators require imported traces. -Run: evalshift traces import --source ... --target ... -``` - -## Testing Strategy - -Unit tests: - -- Trace model validation and normalization. -- JSONL loader line-numbered errors. -- Import command success, missing pairs, strict failure, and unknown example failure. -- Trace diff matching, missing, extra, reorder, argument drift, and verification gap cases. -- Agent trace evaluator scoring and failure metadata. -- Config model parsing for `evaluators.agent_trace`. -- Debug command rendering/fallback behavior. -- Report payload and HTML rendering with agent traces. - -Integration tests: - -- Fixture run with imported source/target traces. -- `traces import -> evaluate -> analyze -> report`. -- Existing simple and tool pipeline tests stay green. - -Verification commands: - -```bash -pytest tests/unit/test_trace_models.py -pytest tests/unit/test_trace_import_command.py -pytest tests/unit/test_agent_trace_evaluator.py -pytest tests/unit/test_reports.py -pytest tests/integration/test_agent_trace_pipeline.py -ruff check . -ruff format --check . -mypy --strict src/evalshift -``` - -## Documentation - -Add: - -- `docs/traces.md` -- Configuration reference section for `evaluators.agent_trace` -- Getting-started note for CLI-only bring-your-own-agent traces -- Example trace JSONL files under `examples/agent-traces/` - -Docs should make the product boundary explicit: EvalShift evaluates traces supplied by the user's agent; it does not run the agent. - -## Rollout - -Implement in four focused increments: - -1. Trace schema and loader. -2. Trace import command and artifact storage. -3. Agent trace evaluator and config. -4. Debug/report rendering and example docs. - -Each increment should preserve existing CLI behavior and keep the full test suite green. diff --git a/docs/superpowers/specs/2026-07-21-strong-default-init-design.md b/docs/superpowers/specs/2026-07-21-strong-default-init-design.md deleted file mode 100644 index 30a59d0..0000000 --- a/docs/superpowers/specs/2026-07-21-strong-default-init-design.md +++ /dev/null @@ -1,179 +0,0 @@ -# Strong-default `evalshift init` — design - -**Date:** 2026-07-21 -**Status:** approved (approach + sections 1–2 approved explicitly; remainder -approved by delegation — "use your own judgement") - -## Problem - -`evalshift init` scaffolds a config that fails real migrations for reasons -unrelated to model quality. Observed on a live personalButler migration -(gemini-3.1-flash-lite-preview → gpt-5.4-mini, runs `r_20260720_*_3a56ff` -and `r_20260720_*_8bd2ba`): - -1. **Directional judge criterion vs anonymized A/B.** The scaffold's - `criterion_prompt` asks about "TARGET vs SOURCE", but - `PairwiseJudgeEvaluator` shuffles outputs into anonymous slots A/B. - The judge cannot orient the question; in run `3a56ff` it answered "A" - 12/12 times and the win/loss split was 100% explained by the slot - shuffle, 0% by content. -2. **Judge failures score as ties.** Judge/semantic API failures return - neutral 0.5/0.5 (or 0/0) scores with `error=None`; a run where 100% - of judge calls failed reads as a clean pass. -3. **Reasoning-model judges are unusable.** `temperature=0.0` is - hardcoded; gpt-5.6-class models reject any temperature ≠ 1 → - every call fails (then silently ties, per #2). -4. **`semantic.cosine` always ends "critical".** Source is pinned at 1.0 - (compared to itself), so the paired Wilcoxon tests "is target - byte-identical to source", which is always significant. The evaluator - can only ever produce a blocking regression. -5. **Policy budgets assume large n.** Profile defaults - (`max_overall_regression_rate: 0.03`, `min_equivalence_rate: 0.95`) - are unreachable at capture-scale n≈8–20 where one flipped example is - a 5–12% swing. Users loosen budgets to meaningless values to cope. -6. **`capture sync` promotes duplicates.** Re-exercising an agent on the - same data produces captures with identical `input_hash`; sync - promotes both. Observed twice: 12 rows/6 unique, 16 rows/8 unique — - every n, p-value, and effect size doubled/inflated. - -## Goals - -Config written by `init` should give honest verdicts with zero edits for -the common case. Deterministic signals gate; noisy signals inform. Small -n yields `inconclusive`, not false `fail`. Broken evaluator calls are -excluded, never neutral-scored. - -## Design - -### 1. `blocking` flag per evaluator (approved) - -* Every evaluator config model in `config/models.py` gains - `blocking: bool = True` (`SemanticEvaluatorConfig`, `LLMJudgeConfig`, - `StructuralEvaluatorConfig`, `ToolSelectionEvaluatorConfig`, - `ToolArgumentsEvaluatorConfig`, `ToolTraceStructureEvaluatorConfig`, - `AgentTraceEvaluatorConfig`). -* `EvalRecord` gains `blocking: bool = True`; stamped at evaluate time - (`_build_evaluators` attaches the config value to each evaluator - instance; `_score_one`/tool/trace paths copy it onto the record). - Old `scores.jsonl` files load as blocking (back-compat default). -* Scaffold ships `semantic` and `llm_judge` with `blocking: false` + - a comment explaining advisory semantics and when to flip. - -### 2. Decision engine: advisory-aware + CI-aware budgets (approved) - -In `analysis/policy.py`: - -* Records partition into blocking/advisory via `EvalRecord.blocking`. - Rate metrics (`regression_rate`, `equivalent_rate`, `improved_rate`, - `critical_regressions`, `tool_argument_drift_rate`) computed from - blocking records only. A parallel `advisory: PolicyMetricSummary | None` - field on `MigrationDecision` carries the same shape for advisory - records (None when none exist). -* Comparisons whose `evaluator_name` belongs to an advisory evaluator are - excluded from `_verdict_for` and `_blocking_regressions`; they are - collected into `advisory_regressions` (same `BlockingRegression` shape) - for the report. -* **Wilson CI on rate budgets.** `BudgetResult` gains - `ci_low: float | None`, `ci_high: float | None`, - `conclusive: bool = True`. For `max_overall_regression_rate` and - `min_equivalence_rate`, compute a 95% Wilson interval at blocking-n: - * observed within budget → **pass** (wide CI does not block a clean run) - * observed breaches and CI excludes the budget → **fail** - * observed breaches but CI straddles the budget → **inconclusive**, - with `reason` explaining n and the interval - Count budgets (`max_critical_regressions`) and cost/latency ratios - stay exact. -* Verdict precedence: conclusive budget fail → `fail`; blocking-severity - comparison → `fail`; non-conclusive breach → `inconclusive`; - regression-severity comparison → `conditional_pass`; else `pass`. -* Profile policy numbers stay strict (0.03/0.95 etc.) — small-n runs now - report `inconclusive` instead of false `fail`, and the numbers become - binding as suites grow. - -### 3. Judge fixes - -* **Symmetric scaffold criterion** (no TARGET/SOURCE, explicit tie): - - > Which output is more complete and correct? Prefer valid JSON over - > fenced or malformed JSON, no dropped or invented fields or entity - > ids, and conclusions grounded in the input. Answer "tie" when both - > are equivalent in substance and differ only in wording. - -* **`drop_params`.** `ModelClient` passes `drop_params=True` to - `litellm.acompletion` so unsupported params (temperature on - reasoning models) are dropped instead of erroring. Judge keeps - `temperature=0.0` for determinism where supported. -* **Failures raise.** `PairwiseJudgeEvaluator.score` re-raises judge - call/parse failures as `EvaluatorError`; `_score_one` already converts - any raise into an `error=`-stamped record, which `_metrics` and - slicing exclude. Same change in `CosineSimilarityEvaluator` for - embedding failures. The `Evaluator` protocol docstring drops - "should never raise". - -### 4. Semantic stays paired but advisory - -Scoring model unchanged (source ≡ 1.0 is by design a target-preservation -metric), but the scaffold marks it `blocking: false`, so its -guaranteed-significant Wilcoxon can no longer gate a verdict alone. Its -regressions surface in `advisory_regressions` and the report. The -`min_similarity` flag still drives `SEMANTIC_REGRESSION` failure -categories. - -### 5. Interactive provider selection in `init` (approved) - -* New `--provider [gemini|openai|anthropic]` option. When omitted and - stdin is a TTY, `init` prompts (numbered choice, default gemini). - Non-TTY without the flag → gemini + a hint that `--provider` exists. -* Provider fills the scaffold's model ids: - - | provider | source (edit-me baseline) | judge | embedding | - |-----------|------------------------------------|------------------------|--------------------------------| - | gemini | gemini-3.1-flash-lite-preview | gemini-3.1-pro-preview | gemini/gemini-embedding-001 | - | openai | gpt-5.4-mini | gpt-5.6-luna | openai/text-embedding-3-small | - | anthropic | claude-sonnet-5 | claude-opus-4-8 | *(semantic commented out — no Anthropic embedding endpoint; comment explains)* | - -* Scaffold comment on `judge_model`: prefer a judge from a different - family than source *and* target when the target is known. - -### 6. `capture sync` dedup - -* Before grouping, drop captures whose `(suite, input_hash)` was already - seen (first occurrence wins, iteration order = existing - `iter_captures` order). Summary line reports - `skipped N duplicate capture(s) (same input)`. -* `--keep-duplicates` opts out (variance measurement use case). - -## Out of scope - -* Entity-id set evaluator (deterministic JSON extraction scoring) — - separate feature, tracked for a follow-up. -* Report HTML redesign; only additive advisory labelling data in JSON - (template picks it up opportunistically). -* Registry refresh of stale model rows. - -## Testing - -* config: `blocking` parses on every evaluator model; default true; - `extra="forbid"` still rejects typos. -* policy: advisory records excluded from gating metrics; `advisory` - block populated; Wilson verdicts (small-n breach → inconclusive, - large-n breach → fail, clean small-n → pass); advisory comparisons - out of `blocking_regressions`, into `advisory_regressions`. -* judge/semantic: API failure raises → `_score_one` writes - `error=`-record; scores excluded from `_metrics`. -* client: `drop_params=True` present in completion kwargs. -* capture sync: duplicate `input_hash` skipped + reported; - `--keep-duplicates` keeps both; distinct hashes unaffected. -* init: `--provider openai` writes OpenAI ids; interactive prompt path; - non-TTY default; existing init tests updated for new template. - -## Compatibility - -* Old configs (no `blocking`) behave exactly as today (all blocking). -* Old `scores.jsonl` analyze fine (`blocking` defaults true). -* `migration_decision.json` gains additive fields - (`advisory`, `advisory_regressions`, per-budget `ci_low`/`ci_high`/ - `conclusive`); existing consumers unaffected. -* Judge/semantic failures that previously produced fake ties now produce - excluded error records — metrics change only for runs that were - already broken. diff --git a/docs/superpowers/specs/2026-07-23-ci-llms-full-link-design.md b/docs/superpowers/specs/2026-07-23-ci-llms-full-link-design.md deleted file mode 100644 index a67b356..0000000 --- a/docs/superpowers/specs/2026-07-23-ci-llms-full-link-design.md +++ /dev/null @@ -1,67 +0,0 @@ -# CI llms-full.txt documentation link — design - -Date: 2026-07-23 -Status: approved - -## Goal - -`evalshift init` currently advertises two hosted llms.txt references to coding agents -(`cli-llms-full.txt`, `sdk-llms-full.txt`). Add a third for the EvalShift GitHub Action: -`https://www.evalshift.dev/ci-llms-full.txt`, following the exact same pattern end to end — -source doc in the owning repo, hosted copy in the web app, link wired by `init`. - -## Changes by repo - -### 1. evalshift-action (source of truth) - -- Author `llms-full.txt` at the repo root: a dense single-file reference for AI tools. - - Header: name, canonical hosted copy URL (`https://evalshift.dev/ci-llms-full.txt`), - pinned CLI version, marketplace usage (`uses: babaliauskas/evalshift-action@v0`). - - What the action does, step by step: setup-python → pip install pinned `evalshift` → - `scripts/evalshift_action.py` (run pipeline, push hosted run, PR comment, gate). - - Full inputs table (from `action.yml`): `token`, `host`, `config`, `suite`, - `evalshift-version`, `python-version`, `fail-on`, `branch`, `base-branch`, - `create-project`, `comment`, `github-token`. - - Full outputs table: `run_url`, `diff_url`, `run_id`, `regression_count`, `conclusion`. - - `fail-on` gating semantics: `never` / `regression` / `any-slice-regression`. - - Branch auto-detection behavior and overrides. - - Copy-paste PR-gate workflow example with `EVALSHIFT_TOKEN` secret. - - Cross-links to `cli-llms-full.txt` and `sdk-llms-full.txt`. -- Content sourced from `action.yml`, `README.md`, `scripts/evalshift_action.py`. No behavior - claims not backed by those files. - -### 2. evalshift-client (hosting) - -- Copy the authored doc to `public/ci-llms-full.txt` (same mechanism as the cli/sdk copies). -- Add a third entry to `public/llms.txt` under "For AI coding tools": - `[Complete GitHub Action (CI) reference](https://evalshift.dev/ci-llms-full.txt)`. - -### 3. evalshift-cli (this repo) - -- `src/evalshift/cli/commands/_agents.py`: - - Add `CI_DOCS_URL: Final = "https://www.evalshift.dev/ci-llms-full.txt"`. - - Render as a third bullet ("EvalShift GitHub Action (CI)") in both documentation lists: - the `AGENT_INSTRUCTIONS` `## Documentation` section and `_render_pointer_block`. - - Extend the fetch-guard sentence in both places to cover CI / GitHub Action / - workflow tasks. - - Export `CI_DOCS_URL` in `__all__`. -- Tests (`tests/unit/test_agents.py`): TDD — assert `CI_DOCS_URL` appears in - `AGENT_INSTRUCTIONS` and in the pointer block, alongside the existing cli/sdk assertions. -- Docs travel with behavior: - - `DOCS.md`: `--wire-agents` bullet mentions the three hosted references. - - `llms-full.txt` (repo root): mention CI reference URL where sdk reference is mentioned. - - No `docs/` page changes: no per-topic page documents the agent wiring or the hosted - llms references today (verified by grep), and `docs/github-action.md` already covers - the action itself. - - `CHANGELOG.md` under `## [Unreleased]`. - -## Non-goals - -- No changes to action behavior, no new init flags, no renaming of existing URLs. -- No automation for syncing `evalshift-action/llms-full.txt` → `evalshift-client/public/` - (matches current manual pattern for cli/sdk). - -## Risks - -- Link 404s until the client repo deploys — acceptable, all three repos updated together. -- Doc drift between action repo and hosted copy — same accepted risk as existing pattern. diff --git a/docs/superpowers/specs/2026-08-23-temperature-value-rejection-design.md b/docs/superpowers/specs/2026-08-23-temperature-value-rejection-design.md deleted file mode 100644 index 8bc0f94..0000000 --- a/docs/superpowers/specs/2026-08-23-temperature-value-rejection-design.md +++ /dev/null @@ -1,126 +0,0 @@ -# Temperature value rejection — runtime adaptation design - -Date: 2026-08-23 -Status: approved, not yet implemented - -Companion to `2026-08-14-temperature-determinism-design.md`. That spec covers -*withdrawal* — a provider removing the `temperature` parameter, detected via the -`honors_temperature` probe in `models/capabilities.py`. This spec covers the failure -class that probe explicitly does not: a model that *advertises* `temperature` while -rejecting every value except its default. - -## Problem - -A CI run with `judge_model: gpt-5.6-terra` fails every judge call: - -> `litellm.BadRequestError: OpenAIException - Unsupported value: 'temperature' does -> not support 0.0 with this model. Only the default (1) value is supported.` - -EvalShift sends `temperature=0.0` on every call (`models/client.py`) and relies on -`drop_params=True` to survive models that reject non-default values. The comment at -`client.py:331-334` claims this works ("Reasoning-tier models reject temperature != 1; -drop_params tells LiteLLM to drop unsupported params instead of erroring"), and -`capabilities.py:20-23` repeats it ("that case is already handled by `drop_params`"). - -Both claims are false for the models that matter. Verified against litellm 1.98.0: - -- **o-series names** (`o3`, `o4-mini`, …): litellm's `OpenAIOSeriesConfig` special-cases - them and does drop `temperature != 1` under `drop_params`. The comment is true here. -- **gpt-5.x reasoning models**: litellm's `OpenAIGPT5Config` checks the model map's - `supports_none_reasoning_effort` flag. For `gpt-5.6-terra` the flag is `True`, so - litellm passes `temperature=0.0` through *untouched*, betting that the caller set - `reasoning_effort: "none"`. EvalShift never sets `reasoning_effort`, the live API - defaults to a reasoning tier, and the call 400s. Confirmed live: - `get_optional_params(model="gpt-5.6-terra", temperature=0.0, drop_params=True)` - returns `temperature: 0.0` in the outgoing params. -- **`honors_temperature`** returns `True` for these models — `temperature` *is* in - `supported_openai_params`. The probe detects withdrawal, not value constraints, and - says so in its docstring. Nothing detects value constraints. - -A second defect compounds it: the 400 maps to a generic `ModelError` in -`_map_exception`, and `_dispatch_with_retry` retries it with **identical kwargs** up to -`max_attempts`. A deterministic failure burns the whole retry budget per call — the CI -log shows `attempt 2 failed (ModelError); retrying in 0.99s` for every example. - -## Why not static knowledge - -litellm's own model map is the thing that is wrong here — its -`supports_none_reasoning_effort: True` entry is precisely what routes the bad value -through. Extending the EvalShift registry with reasoning-model prefixes repeats the -mistake the 2026-08-14 spec rejected: passthrough ids (this repo's registry carries -only `gpt-4o` / `gpt-4o-mini` for OpenAI) would miss every future reasoning id until a -release catches up. A run-start probe call per model was also rejected: it costs a real -API call on every run forever, where runtime adaptation costs one failed call per -affected model per process. - -## Decisions - -| Question | Decision | Why | -| --- | --- | --- | -| Where to fix | `_dispatch_with_retry` in `models/client.py` | Single choke point under `complete`, `complete_messages`, `complete_with_tools`, `complete_messages_with_tools` — judge, replay, tools, and insights paths all inherit it | -| How to detect | At call time, from the provider's own rejection | The provider is the only source of truth; every static source (litellm map, registry) has been shown wrong | -| Detection predicate | Underlying exception is `litellm.BadRequestError` AND its message contains `temperature` (case-insensitive) AND `"temperature" in kwargs` | All three or no intercept. A temperature-flavored 400 on a call that never sent the parameter is somebody else's bug and must surface | -| Adaptation | Pop `kwargs["temperature"]`, redispatch immediately; does **not** consume a retry attempt | The adapted call is a different request, not a retry of the failed one. Attempt counting, backoff, `AuthError` short-circuit, and exhaustion behavior are unchanged | -| Memoization | Per-client set of rejected canonical model ids; later calls to a listed model omit `temperature` before dispatch | One failed call per model per process. Matches the log-dedupe precedent at `client.py:54` | -| Reporting | Client exposes the set read-only; the orchestrator merges it into `state.non_deterministic_models` (dedup, stable order) before the final state write and report build | Reuses the existing surface end-to-end: `runner/models.py:159`, HTML banner (`reports/html.py`), JSON (`reports/json.py`), economics note (`reports/economics.py`). One concept — "sampling was not controlled for these models" — one banner | -| Cache keys | Unchanged — keyed on the *requested* temperature | Same precedent as the Gemini ignored-temperature case: identical requests produce identical keys; the banner carries the honesty. Changing key composition would orphan every existing cache entry | -| Logging | `log.warning` once per model on first rejection | States the model, that `temperature` was withdrawn from its calls, and that outputs are non-deterministic | -| False comments | Rewrite `client.py:331-334` and `capabilities.py:20-23` | Both must describe the real mechanism: litellm saves o-series names only; value constraints on other models are handled by this adaptation | - -## What this changes for users - -A judge (or source/target/insights) model that rejects `temperature=0.0` now works: -first call to it fails once, the client adapts, the run completes. The report and JSON -carry the model in `non_deterministic_models`, so the loss of the control variable is -visible in the same banner Gemini withdrawal uses. Previously every call to such a -model failed after the full retry budget and, for a `blocking: true` judge, poisoned -the gate. - -No config surface is added. No behavior changes for models that accept `temperature`. - -## Out of scope - -- **General deterministic-400 retry policy.** `BadRequestError`s that are not the - temperature case still burn the retry budget. Real, separate change — noting it here - so it is not silently forgotten. -- **Registry entries for gpt-5.x / o-series ids.** Complementary static data, not - needed for correctness once adaptation exists. -- **Setting `reasoning_effort: "none"` to keep temperature control on gpt-5.1+.** - Tempting, but it mutates the model's reasoning behavior under test — the same - asymmetric-confound argument that rejected prompt injection in the 2026-08-14 spec. - -## Tests (written first, per TDD) - -Unit, `tests/unit/test_model_client.py` (or the module's existing home), with a faked -`litellm.acompletion`: - -1. **Adapt and succeed** — first call raises `BadRequestError` naming temperature; - client redispatches without `temperature`, returns the second response; model id - lands in the rejected set; attempt counter unaffected (a subsequent transient error - still gets the full budget). -2. **Preemptive omission** — second call to the same model through the same client - never includes `temperature` in kwargs and makes exactly one dispatch. -3. **Non-temperature 400 unaffected** — `BadRequestError` with an unrelated message - follows the existing retry-then-raise path. -4. **No-temperature call not intercepted** — temperature-naming 400 on kwargs without - `temperature` raises; rejected set stays empty. -5. **AuthError short-circuit preserved.** -6. **Warning logged once** per model across repeated rejections. - -Orchestrator level: - -7. **Banner merge** — runtime-rejected model appears in `state.non_deterministic_models` - exactly once (dedup against probe-detected), and in the JSON report output. - -Integration: - -8. **Judge through client** — `llm_judge` against a faked value-rejecting model - produces a verdict instead of `EvaluatorError`. - -Gate: `make ci` (ruff + `mypy --strict` + pytest) green before any done-claim. - -## Documentation - -- `DOCS.md` determinism section: value-rejection case added next to withdrawal. -- `CHANGELOG.md`: Fixed entry under Unreleased. -- `llms-full.txt`: regenerate/update if it states the `drop_params` claim. diff --git a/docs/superpowers/specs/2026-09-09-namespace-collision-design.md b/docs/superpowers/specs/2026-09-09-namespace-collision-design.md deleted file mode 100644 index f700271..0000000 --- a/docs/superpowers/specs/2026-09-09-namespace-collision-design.md +++ /dev/null @@ -1,182 +0,0 @@ -# `evalshift` namespace collision — packaging design - -Date: 2026-09-09 -Status: approved; implemented in the same change -Plan: `docs/superpowers/plans/2026-09-08-external-review-response.md`, Phase 6 (review point #1) - -## Problem - -Two distributions ship a top-level `evalshift/` import package: - -| Distribution | Repo | Import package | Owns | -| --- | --- | --- | --- | -| `evalshift-sdk` | evalshift-sdk | `evalshift` | capture decorator, trace models, sinks. `evalshift/__init__.py` re-exports the public API. | -| `evalshift` | evalshift-cli | `evalshift` | Typer app, runner, evaluators, reports. | - -pip installs both into the same `site-packages/evalshift/` directory. The second -install overwrites the first one's `__init__.py`, the loser's public surface -disappears, and `pip uninstall` of either removes files the other still needs. -Every install page therefore repeats the same rule — separate virtual -environments — in `README.md`, `docs/getting-started.md`, `docs/sdk.md`, -`DOCS.md`, `AGENTS.md`, `llms-full.txt`, `examples/capture-first/`, and on the -SDK side in `README.md`, `DOCS.md`, and `examples/support_agent/README.md`. - -The external review (2026-09-08) named this the biggest setup friction. The SDK's -`docs/DECISIONS.md` had it tracked as **D1-followup** with two candidate fixes: -"CLI depends on SDK, or a `[cli]` extra". - -## Decision - -Option A from the plan, in its second form: **the CLI moves to the import -package `evalshift_cli` and declares `evalshift-sdk` as a runtime dependency.** -The SDK changes no code. - -Concretely: - -- `src/evalshift/` → `src/evalshift_cli/` in evalshift-cli; every internal import - `evalshift.` → `evalshift_cli.`. -- Distribution name stays `evalshift`. Console script stays `evalshift`, now - pointing at `evalshift_cli.cli.main:app`. `python -m evalshift` becomes - `python -m evalshift_cli`. -- `dependencies` gains `evalshift-sdk>=0.3.0`. -- `import evalshift` is the SDK in every environment that has the CLI. - `pip install evalshift` is enough for a project that both instruments an agent - and runs evaluations. - -### Why this form of A - -The plan's first form — CLI code under `evalshift.cli` inside the SDK's import -root — does not work across two distributions. The SDK's `evalshift/__init__.py` -re-exports the public API (`capture`, `record_model_call`, `configure`, the sinks, -`load_capture`, `SCHEMA_VERSION`, `__version__`). A subpackage contributed by a -second distribution requires `evalshift` to be a PEP 420 namespace package, which -forbids that `__init__.py`; the SDK would lose `import evalshift; evalshift.capture` -and its version attribute. Editable installs of the two repos side by side need -the namespace form as well. A separate top-level package has none of these -constraints and keeps each repo's import root, licence, and CI independent. - -### Why not B or C - -- **B — rename the SDK to `evalshift_sdk` with a deprecation shim.** Breaks - `from evalshift import capture` in every instrumented production agent once the - shim goes, to fix a problem those agents do not have. The SDK's import path is - its public API; the CLI's is not (see *Compatibility*). -- **C — one distribution with a `[cli]` extra.** A repo merge plus reconciling - AGPL-3.0-or-later (CLI) with MIT (SDK) per subpackage. Cost out of proportion. - -### The dependency - -`evalshift-sdk>=0.3.0`, no upper bound. - -- 0.3.0 is the first SDK that writes trace schema `2.0.0`, the schema - `captures/models.py` reads. -- No upper bound because the SDK is stdlib-only (the dependency adds nothing to - the CLI's environment) and because the CLI reads captures through its own - pydantic models, guarded by the SDK's vendored-mirror parity test. An SDK - release cannot break the CLI through the import. -- The dependency is a packaging guarantee, not a code path. The CLI still does not - import the SDK; files under `.evalshift/` remain the only interface, so the docs - keep "never call each other" and drop "separate environments". Phases 2 and 3 - may choose to import SDK readers later; nothing here requires it. - -## What changes for users - -| | Before | After | -| --- | --- | --- | -| Install the CLI | `pip install evalshift` | Same. Pulls `evalshift-sdk` with it. | -| Install the SDK only (production agents) | `pip install evalshift-sdk` | Same. | -| Both in one environment | Not supported. | Supported. | -| `evalshift` command | | Unchanged. | -| `evalshift.yaml`, `golden.jsonl`, `.evalshift/` | | Unchanged. | -| `import evalshift` | Whichever package was installed last. | Always the SDK. | -| `python -m evalshift` | Runs the CLI. | `python -m evalshift_cli`. | -| Scripts importing CLI internals | `from evalshift.models.client import …` | `from evalshift_cli.models.client import …` | -| GitHub Action | Installs `evalshift==`, calls the console script. | Unaffected. | - -## Compatibility and deprecation window - -There is no shim. The CLI cannot ship any file under `evalshift/` without -recreating the collision, so the old import path cannot be kept alive even for one -release. The CLI's Python import path has never been documented as public: the -README defines the surface as the command line and the file formats, and the only -internal-path mentions outside this repo's own code are `scripts/`, `docs/faq.md`, -and `llms-full.txt`. The window is therefore the CHANGELOG entry. The change ships -as **0.14.0** (minor, pre-1.0), marked *Breaking*, naming the two renames a user -could notice: `python -m evalshift` and the internal import path. 0.13.x stays on -PyPI, and the GitHub Action pins by version. - -Upgrade paths: - -1. CLI-only environment: `pip install -U evalshift`. pip removes 0.13's `evalshift/` - files by RECORD, then installs `evalshift_cli/` and the SDK. -2. Agent environment that already has the SDK: `pip install evalshift` now works. -3. Existing two-environment setups keep working. Nothing forces consolidation. -4. Development checkouts: `uv pip install -e ".[dev]"` again, so the editable - `.pth` moves from `evalshift` to `evalshift_cli`. - -## `evalshift doctor` check - -New row `evalshift-sdk`, second in the table after the Python row. Advisory only: -the CLI does not need the SDK to run, so the row never fails the command. - -| Condition | Status | Detail | -| --- | --- | --- | -| `evalshift-sdk` distribution installed and `import evalshift` exposes `capture` and `SCHEMA_VERSION` | ok | ` (import name evalshift)` | -| Distribution not installed (`--no-deps`, hand-built env) | warn | not installed; `from evalshift import capture` fails here; `pip install evalshift-sdk` | -| Module imports but is not the SDK | warn | `import evalshift` resolves to ``, not the SDK. An older evalshift CLI's leftover files or a local `evalshift/` directory shadow it. | -| Import raises | warn | `import evalshift` failed: `` | - -Detection imports the module (cheap, stdlib-only, inert without -`EVALSHIFT_CAPTURE=1`) and checks for the two attributes that neither a pre-0.14 -CLI package nor a stray directory has. The version comes from -`importlib.metadata`, falling back to `evalshift.__version__` for editable or -otherwise metadata-less installs. The importer and the version lookup are -injectable so tests exercise every row without touching the interpreter; one test -runs against the real environment and so also pins the dependency declaration — -CI's venv only gets the SDK through it. - -## Files - -evalshift-cli: - -- `src/evalshift/` → `src/evalshift_cli/` (git mv), imports rewritten in `src/`, - `tests/`, `scripts/`. -- `pyproject.toml`: scripts entry, wheel/sdist package paths, coverage source, - new dependency. -- `mypy.ini`, `ruff.toml` (`known-first-party`), `Makefile`, - `.github/workflows/ci.yml`, `.pre-commit-config.yaml`: path `src/evalshift_cli`. -- `cli/commands/all.py`: own-record detection compares the logger root to the new - package name. -- `cli/commands/doctor.py` + `tests/unit/test_doctor.py`: the row above. -- Docs: `README.md`, `docs/getting-started.md`, `docs/sdk.md`, `docs/faq.md`, - `DOCS.md`, `AGENTS.md`, `llms-full.txt`, `CLAUDE.md`, `CONTRIBUTING.md`, - `examples/capture-first/` (README and agent docstring), `CHANGELOG.md`. - -evalshift-sdk (docs only): - -- `docs/DECISIONS.md`: D-pkg and D1-followup marked resolved, pointing here. -- `README.md`, `DOCS.md`, `examples/support_agent/README.md`: drop the two-venv - rule; say the CLI installs the SDK. -- `tests/conformance/cli_models_vendored.py`: source path in the header. -- `CHANGELOG.md`. - -## Out of scope - -- The CLI importing SDK code. Phases 2 and 3 decide that on their own merits. -- A repo merge. -- evalshift-client's mirrored `public/cli-llms-full.txt`; it is synced from this - repo's `llms-full.txt` by that repo's process. -- A `__main__` in the SDK to redirect `python -m evalshift`. The SDK is a library; - Python's own message ("'evalshift' is a package and cannot be directly - executed") plus the CHANGELOG is enough. - -## Verification - -- `make ci` in evalshift-cli with `evalshift-sdk` installed from PyPI, not the - sibling checkout — the environment a user gets. -- `python -c "import evalshift, evalshift_cli; print(evalshift.__file__)"` prints - the SDK's path. -- `evalshift doctor` shows the `evalshift-sdk` row as ok. -- `uv build`; the wheel's file list contains no `evalshift/` entries. -- `examples/capture-first`: the agent and `evalshift capture sync` run from one - environment. diff --git a/docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md b/docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md deleted file mode 100644 index cf7f653..0000000 --- a/docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md +++ /dev/null @@ -1,340 +0,0 @@ -# Teacher-forced multi-round replay — design - -**Status:** implementing (Phase 2 of `docs/superpowers/plans/2026-09-08-external-review-response.md`). -**Scope:** `evalshift-cli` only. The SDK already records everything this needs -(`tool_call` / `tool_result` events keyed by `call_id`, `requested_tool_calls` on -`model_call` since schema 2.1.0). - -## Problem - -`evalshift run` issues one model call per example and never feeds a tool result -back. A captured agent turn is usually a loop — call tools, read results, call -more, answer — so every promoted case carries the whole loop in -`expected_tool_rounds` while replay can only ever reach round 1. `--rounds all` -today *flattens* every round into `expected_tools`, which the docs themselves -describe as a yardstick no single-shot replay can meet. Recorded tool results -(`ToolResultEvent.result`) are written by the SDK and consumed by nothing. - -## Decision in one paragraph - -Replay becomes **teacher-forced per round**. For round *k* (0-based) the -candidate is sent the rendered prompt followed by the *recorded* rounds -`0..k-1` — the recorded assistant tool calls and the recorded tool results, as -ordinary `assistant` / `tool` messages — and asked for round *k*. Its answer is -scored against `expected_tool_rounds[k]`, then discarded: the candidate's own -calls are **never** fed back. That is what makes round *k* comparable across -source, target and ground truth — all three saw byte-identical context — and it -is the same contract the multi-turn conversation replay already uses for -`history`. Promotion carries the recorded results on the example as -`tool_result_fixtures`; the runner replays as many rounds as the fixtures cover, -plus the answer round after the last covered round. `--rounds first` (the -default) writes no fixtures and behaves exactly as before. - -### What this rules out, and why - -* **Self-conditioned replay** (feed the candidate's *own* calls' results back, - looked up by name + argument hash). Different candidates would then see - different contexts from round 2 on, and a per-round comparison against - `expected_tool_rounds[k]` stops being fair. It also needs the halt-and-flag - policy for calls that have no fixture. Out of scope; a follow-up could add - it as a distinct mode. -* **Flattened `expected_tools`.** `--rounds all` no longer flattens. - `expected_tools` is always `expected_tool_rounds[0]`. The flattened list was - only ever right for comparing against an externally produced multi-round - trace, which is the `agent_trace` evaluator's job and reads imported traces, - not `expected_tools`. Recorded in CHANGELOG as a behaviour change. - -## Data model - -### Suite (`suite/models.py`) - -```python -class ToolResultFixture(_StrictModel): - tool_name: str = Field(min_length=1) - result: Any = None # ToolResultEvent.result, verbatim - error: str | None = None # ToolResultEvent.error, verbatim - -class SuiteExample(_StrictModel): - ... - expected_tool_rounds: list[list[ExpectedToolCall]] | None = None # unchanged - tool_result_fixtures: list[list[ToolResultFixture]] | None = None # NEW -``` - -Alignment is **positional**: `tool_result_fixtures[k][i]` is the result of -`expected_tool_rounds[k][i]`. No call ids on the suite: `ExpectedToolCall` -deliberately has none (a candidate invents its own), and the runner synthesises -ids per position when it builds the messages, the same way `_dispatch_message` -already does for `history` tool turns. - -Validation (model validator, load-time errors): - -* `tool_result_fixtures` requires `expected_tool_rounds`. -* `len(tool_result_fixtures) <= len(expected_tool_rounds)`. -* For every covered round `k`, - `len(tool_result_fixtures[k]) == len(expected_tool_rounds[k])` and - `tool_result_fixtures[k][i].tool_name == expected_tool_rounds[k][i].tool_name`. -* An empty inner list is only valid where the round itself is empty, which - `expected_tool_rounds` never contains, so effectively never. - -`None` means single-shot replay (every suite written before this field). The -field is omitted from nothing: `golden.jsonl` is written with -`model_dump_json()`, so the checked-in `examples/capture-first` rows gain -`"tool_result_fixtures": null` and are regenerated by the test that re-derives -them. - -**Rounds replayed** for an example, derived, not stored: - -``` -rounds_to_replay(example) = 1 if example.tool_result_fixtures is None - else len(example.tool_result_fixtures) + 1 -``` - -The `+1` is the round *after* the last covered round: when fixtures cover every -tool round, that is the answer round, where the recorded agent produced its -final text (`expected.final_output`) and called nothing. Replaying it is what -gives the text evaluators a real answer to compare, and it catches a candidate -that keeps calling tools when it should have answered. The expected tool list -for a round `k >= len(expected_tool_rounds)` is empty. - -### Tool trace (`evaluators/tool_models.py`) - -One `Call` row still holds the whole example (see *Runner*), so a trace must be -able to carry several responses: - -```python -class ToolCall(_StrictModel): - ... - round_index: int = Field(default=0, ge=0) # NEW - -class ToolTrace(_StrictModel): - ... - round_count: int = Field(default=1, ge=1) # NEW - def round(self, index: int) -> ToolTrace # NEW: calls of one round, re-indexed from 0 - def rounds(self) -> list[ToolTrace] # NEW: round(0) .. round(round_count - 1) -``` - -* `sequence_index` keeps counting across rounds (unique per trace, as now). -* Validator: every `round_index < round_count`. -* `final_text` is the **last** round's text — the answer a user would have seen. - `raised_refusal` / `refusal_text` are set if any round refused. -* Defaults make every existing `raw.jsonl` row load as a one-round trace. -* `ToolTrace.round(k)` returns a trace whose calls have `round_index == k`, - `sequence_index` renumbered from 0, `round_count == 1`, and `final_text` only - for the last round; that is what lets every existing single-round scoring - function run unchanged on one round. - -### Run state / calls (`runner/models.py`) - -`Call` is unchanged in shape. One row per `(prompt, example, role)`, as now: - -* `text` — last round's final text. -* `input_tokens` / `output_tokens` / `cost_usd` / `latency_ms` — summed over - rounds. -* `trace` — the merged multi-round trace above. -* `finish_reason` — the last round's, unless an earlier round was `"length"`, - which wins (a truncated round poisons every later comparison). -* `error` — a `ModelClientError` in round *k* sets - `error = f"round {k + 1}/{n}: {exc}"` and leaves `trace = None`, exactly as a - failed single-shot call does today. Rounds already completed are not written - separately; a partially replayed example is an unmeasured example. - -Resume granularity stays per example: an interruption mid-loop redoes the -whole example. Deliberate — `completed_call_keys` and every consumer of -`raw.jsonl` (evaluate pairing, reports, bundle) keep their `(prompt, example, -role)` identity. - -## Promotion (`captures/promote.py`) - -Under `rounds == "all"`: - -1. Build `tool_rounds` exactly as today (requested calls when the capture has - them, executed calls otherwise). -2. For each round *k*, pair every call with its `ToolResultEvent`: - * by `call_id` when both the call (`ToolCallEvent.call_id` / - `RequestedToolCall.call_id`) and a result carry one; - * else the first not-yet-consumed `tool_result` **with the same name** that - follows the round's `model_call` and precedes the next `model_call`. -3. Fixtures are written for rounds `0..m-1` where *m* is the first round with - any unpaired call (or every round when none is unpaired). Stop there, warn: - `"round {m+1} has {j} tool call(s) with no recorded result; replay will cover - rounds 1..{m}"`. -4. `expected_tools = tool_rounds[0]` (no flattening — behaviour change). - `expected_tool_count`, when `--tool-count` is set, is the total over - *replayed* tool rounds (rounds `0..m-1` plus round *m* if it exists), which - is what `tool_trace_structure` compares to the candidate's total call count. -5. `tool_result_fixtures = [[ToolResultFixture(tool_name, result, error), ...], ...]` - or `None` when *m* == 0 (round 1 itself has an unpaired call — replay is - single-shot and the warning says so). - -Under `rounds == "first"`: unchanged. No fixtures, `expected_tools` = round 1, -the existing "later calls were moved to expected_tool_rounds" warning stays, -reworded to point at `--rounds all` as the teacher-forced option rather than a -flattening one. - -`build_conversation_examples` (multi-turn) passes the field through untouched; -history and rounds compose (see *Messages*). - -`example_content_key` does **not** include fixtures: two captures with the same -inputs and history are the same case regardless of what their tools returned. - -## Runner (`runner/orchestrator.py`) - -`WorkItem` is unchanged: one per `(prompt, example, role)`. `_execute_with_tools` -grows an inner loop: - -``` -rounds = rounds_to_replay(item.example) -for k in range(rounds): - messages = build_round_messages(item.example, prompt_text, k) # see below - result = await client.complete_messages_with_tools(model, messages, tools, ...) - collect tokens / cost / latency; tag result.trace.calls with round_index = k -merge into one ToolTrace(round_count=rounds); build the Call -``` - -Round 0 keeps its current dispatch: `complete_with_tools(prompt)` for single-turn -examples, `complete_messages_with_tools(messages)` for `history` examples, so a -single-shot example makes byte-identical client calls to today (the mocked -clients in `tests/` and `ReplayClient` depend on that). Rounds `k >= 1` always go -through `complete_messages_with_tools`. - -### Messages for round *k* (`build_round_messages`) - -``` -[*history (as today, may be empty), {"role": "user", "content": prompt_text}, - # for each recorded round j < k: - {"role": "assistant", "content": "", "tool_calls": [ - {"id": f"call_r{j}_{i}", "type": "function", - "function": {"name": call.tool_name, "arguments": json.dumps(call.arguments or {})}} - for i, call in enumerate(expected_tool_rounds[j])]}, - {"role": "tool", "tool_call_id": f"call_r{j}_{i}", "content": render(fixture)} - for i, fixture in enumerate(tool_result_fixtures[j])] -``` - -`render(fixture)`: `fixture.result` verbatim when it is a `str`; otherwise -`json.dumps(result, ensure_ascii=False, default=str)`; when `error` is set, -`json.dumps({"error": error})`. Same OpenAI wire shape `_dispatch_message` -already emits; LiteLLM translates it per provider. `arguments is None` -(promoted with `--names-only`) renders as `{}`. - -### Cache key (`cache/store.py`) - -`cache_key(...)` gains `round_index: int | None = None`, hashed only when not -`None`, so every existing key is byte-identical. The tool path still bypasses -the cache (unchanged since v0.2), so today nothing passes it; it exists so that -when tool-call caching lands, the round dimension is already in the key and no -migration is needed. `_execute` (the text path) passes `None`. - -### Cost pre-flight (`utils/cost.py`) - -`estimate_run_cost` gains `per_example_calls: Sequence[int] | None = None` -(calls per example per model; default 1 each). `total_calls = -n_prompts * sum(per_example_calls) * len(models)`; `RunState.total_evaluations` -and the progress bar stay per `Call` row (one per example per role), so the -estimate's call count is a cost figure, not the progress denominator. The -orchestrator passes `rounds_to_replay(e)` per example and adds the rendered -fixture text length to `per_example_extra_chars`, so the growing context is -counted at least once. - -`preflight_cost` reports the same estimate. - -## Scoring (`evaluators/`) - -Evaluators keep their signature. Each one that reads tool traces splits both -traces with `ToolTrace.rounds()` and scores round *k* against -`expected_tool_rounds[k]` (or against "no tools" for `k >= -len(expected_tool_rounds)`), then reports **one record per example per axis** -whose `target_score` / `source_score` are the **mean over replayed rounds**. -Per-round detail travels in metadata: - -```json -"rounds": [ - {"round": 0, "expected_names": [...], "source_names": [...], "target_names": [...], - "source_score": 1.0, "target_score": 0.5}, - ... -] -``` - -Single-round traces (`round_count == 1`, no fixtures) score exactly as today and -carry no `rounds` key, so every existing `scores.jsonl` reader and every -existing test is unaffected. - -* **tool_selection.conformance** — per-round `_sequence_match` / - `_multiset_match` against `expected_tool_rounds[k]`; for `k` beyond the - recorded rounds, `1.0` iff the candidate called nothing. A round with no - ground truth and no calls on either side does not lower the mean. - `expected_no_tools` examples have no rounds and are unchanged. -* **tool_selection.divergence** — per-round `exact` / `set` / `first` of target - vs source, mean over rounds. Because the mean drops below 1.0 as soon as one - round differs, `max_tool_divergence` (which counts `delta < 0`) counts an - example as diverged if **any** replayed round diverged. No policy code change; - documented in `docs/configuration.md` and `docs/agents.md`. -* **tool_arguments** — `against: expected` pairs `expected_tool_rounds[k]` - with round *k*'s calls; `against: source` matches same-name calls within a - round. Mean of per-round means over rounds that had something to score. -* **tool_trace_structure** — `call_count` over the whole trace (all rounds); - `parallelism` compares per round (score 1.0 only if every round agrees); - `refusal_alignment` on the merged flags; `expected_count` against the total. - `details.rounds_replayed = {"source": n, "target": n}` added to metadata. -* **agent_trace** — imported traces, untouched. - -Top-level `source_names` / `target_names` in metadata stay the **flattened** -lists across rounds so `reports/json._tool_change` and every existing consumer -keep working. - -## Reports - -* `ToolChange` gains `rounds: list[RoundToolChange] | None` (`round`, - `source_names`, `target_names`, `expected_names`, `diverged`), read from the - new metadata. The per-example table and the top-regression card render one - line per round (`r1 a, b → a, b` / `r2 c → d`) when `rounds` is present, the - current single line otherwise. `report.json` carries the same list. -* `_build_tool_diffs` diffs per round (position resets each round) and prefixes - messages with `Round k:` when `round_count > 1`. -* `hosted/trace_events.from_tool_trace` emits `round: call.round_index`, and a - `final_output` event in the last round. The bundle format already allows it - (`BUNDLE_SPEC.md` §round). Its docstring's "every event is round 0" sentence - goes. - -## Docs and CHANGELOG (one pass, after the code lands) - -* `docs/agents.md` — rewrite "Agent rounds and what a replay can reproduce": - default single-shot, `--rounds all` = teacher-forced replay, what the - candidate sees per round, scoring is per round, the divergence budget counts - any-round divergence, and the answer round. Reword the `--rounds` help text - in `cli/commands/capture.py` (both `promote` and `sync`). -* `docs/evaluators.md`, `docs/configuration.md` (`max_tool_divergence` - semantics, capture lifecycle `--rounds`), `DOCS.md`, `llms-full.txt`, - `README.md` agent paragraph (replace the "how it sequences" wording with what - is tested). -* CHANGELOG `## [Unreleased]`: **Added** teacher-forced replay; **Changed** - `--rounds all` no longer flattens `expected_tools`. -* Plan Task 1.1 (the single-round caveat) is superseded: the caveat is written - once, as "default", in the same edit. - -## Test plan - -* `suite/models.py` — fixture alignment validators; `None` round-trips; v0.1–v0.3 - rows still load. -* `tool_models.py` — `round_index` default, `round_count` validator, - `rounds()` / `round(k)` renumbering, old JSON loads. -* `promote.py` — two-round capture with `rounds="all"` yields two - `expected_tool_rounds` and fixtures aligned by position, paired by `call_id`; - missing result stops coverage with the warning; `rounds="first"` writes no - fixtures; `expected_tools` is round 1 under both; requested-call captures pair - by `RequestedToolCall.call_id`; `examples/capture-first` re-derivation still - passes after regeneration. -* `orchestrator.py` — a fake client that records `messages` per call: a - fixture-bearing example produces `len(fixtures) + 1` calls, round *k*'s - messages contain exactly rounds `0..k-1` as assistant/tool turns with - positional ids and rendered results; tokens/cost sum; a round-2 error yields - one `Call` with `error` naming the round; single-shot examples make the same - calls as before (existing tests unchanged). -* `tests/integration/` — `ReplayClient` grows round-aware matching (fixtures may - carry `"round": k`; a `kind: "tools"` fixture without `round` matches every - round, as today); a capture → sync `--rounds all` → run → evaluate → report - pipeline shows per-round rows and a diverged round 2 counted under - `max_tool_divergence`. -* Evaluators — per-round conformance / divergence / arguments / structure with - hand-built multi-round traces; single-round records are byte-identical to - today's. -* Reports — `rounds` in `ToolChange`, per-round diff messages, JSON shape. diff --git a/pyproject.toml b/pyproject.toml index 0282394..0a407b4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -30,7 +30,7 @@ dependencies = [ # The capture SDK (import name `evalshift`). Stdlib-only, so it adds nothing to # the environment; declared so one `pip install evalshift` serves both # instrumenting an agent and running evaluations, and so `import evalshift` is - # always the SDK. See docs/superpowers/specs/2026-09-09-namespace-collision-design.md. + # always the SDK. See the co-install note in DOCS.md. "evalshift-sdk>=0.4.0", "typer>=0.12", "pydantic>=2.5", diff --git a/tests/unit/test_suite_models.py b/tests/unit/test_suite_models.py index f4be5be..f8039f0 100644 --- a/tests/unit/test_suite_models.py +++ b/tests/unit/test_suite_models.py @@ -587,8 +587,8 @@ def _agent_ex(**kw: Any) -> SuiteExample: class TestToolResultFixtures: """Recorded tool results ride on the example, aligned by position with - ``expected_tool_rounds`` — see - ``docs/superpowers/specs/2026-09-09-teacher-forced-replay-design.md``.""" + ``expected_tool_rounds`` — see "Agent rounds and what a replay can + reproduce" in ``docs/agents.md``.""" def test_absent_by_default_and_means_single_shot(self) -> None: ex = _agent_ex(id="e", expected_tool_rounds=[[_call("a")]])