From 6478d718bd5ff041ea3809d7d45428366cec5a5a Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Fri, 2 Oct 2026 20:57:53 +0200 Subject: [PATCH] chore: move to the evalshift GitHub org The repositories moved from the babaliauskas account to the evalshift organization. Point links, PyPI project URLs, mkdocs metadata, the release dispatch, the init --ci scaffold and the docs at evalshift/*. The CI pin check now matches evalshift/evalshift-action steps and still matches the old babaliauskas/evalshift-action name, which GitHub redirects, comparing owner and repo case-insensitively as GitHub does. Historical CHANGELOG entries and plan/spec docs keep the old name. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/release.yml | 6 ++-- AGENTS.md | 2 +- CHANGELOG.md | 13 +++++++ CONTRIBUTING.md | 8 ++--- DOCS.md | 16 ++++----- README.md | 14 ++++---- docs/agents.md | 4 +-- docs/changelog.md | 2 +- docs/configuration.md | 2 +- docs/getting-started.md | 2 +- docs/github-action.md | 4 +-- docs/sdk.md | 10 +++--- examples/capture-first/README.md | 2 +- examples/capture-first/evalshift.yaml | 2 +- llms-full.txt | 8 ++--- mkdocs.yml | 6 ++-- pyproject.toml | 10 +++--- src/evalshift_cli/cli/commands/_scaffold.py | 2 +- src/evalshift_cli/cli/commands/init.py | 2 +- .../reports/templates/report.html.j2 | 2 +- src/evalshift_cli/utils/ci_pin.py | 24 ++++++++++--- tests/integration/test_validate_command.py | 2 +- tests/unit/test_ci_pin.py | 35 +++++++++++++++++-- tests/unit/test_cli_capture.py | 2 +- tests/unit/test_doctor.py | 2 +- tests/unit/test_init.py | 6 ++-- 26 files changed, 123 insertions(+), 65 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 15192c1..2f9ae13 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -12,7 +12,7 @@ name: release # repo, this workflow file, and the `pypi` environment as a trusted # publisher. No token secret is stored anywhere. # - `EVALSHIFT_ACTION_DISPATCH_TOKEN` (optional): a fine-grained PAT with -# contents: write on babaliauskas/evalshift-action. Without it the +# contents: write on evalshift/evalshift-action. Without it the # dispatch step is skipped and the action repo's daily poll picks the # release up within a day. # @@ -84,7 +84,7 @@ jobs: needs: [build, publish] runs-on: ubuntu-latest steps: - - name: Dispatch evalshift-cli-release to babaliauskas/evalshift-action + - name: Dispatch evalshift-cli-release to evalshift/evalshift-action env: GH_TOKEN: ${{ secrets.EVALSHIFT_ACTION_DISPATCH_TOKEN }} VERSION: ${{ needs.build.outputs.version }} @@ -94,7 +94,7 @@ jobs: echo "::notice::EVALSHIFT_ACTION_DISPATCH_TOKEN is not set; skipping the dispatch. The action repo's daily PyPI poll will pick up ${VERSION} within a day." exit 0 fi - gh api "repos/babaliauskas/evalshift-action/dispatches" \ + gh api "repos/evalshift/evalshift-action/dispatches" \ -f event_type=evalshift-cli-release \ -f "client_payload[version]=${VERSION}" echo "Dispatched evalshift-cli-release for ${VERSION}; bump-cli-pin will open the pin PR." diff --git a/AGENTS.md b/AGENTS.md index 8465e4d..1305e05 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -38,7 +38,7 @@ repos (`evalshift-sdk`, `evalshift-action`); never edit them from here. two packages never call each other. The CLI imports as `evalshift_cli` and depends on the SDK, so both install into one environment. See [docs/sdk.md](docs/sdk.md). -- **GitHub Action** — `babaliauskas/evalshift-action@v0`. Runs the pipeline on +- **GitHub Action** — `evalshift/evalshift-action@v0`. Runs the pipeline on pull requests, pushes the run, keeps one PR comment updated, sets the `evalshift/regression` commit status. See [docs/github-action.md](docs/github-action.md). - **Hosted server** — API at `https://api.evalshift.dev`, web app at diff --git a/CHANGELOG.md b/CHANGELOG.md index a6c8522..90a6e33 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Changed + +- The EvalShift repositories moved from the `babaliauskas` GitHub account to + the `evalshift` organization: , + `evalshift-sdk` and `evalshift-action`. Links, package metadata and the + workflow `evalshift init --ci` scaffolds now use + `evalshift/evalshift-action@v0`. GitHub redirects the old names, so existing + workflows keep working. +- The CI pin check (`init`, `validate`, `doctor`, `capture sync`) recognizes + `evalshift/evalshift-action` steps as well as the old + `babaliauskas/evalshift-action` name, matching owner and repository names + case-insensitively as GitHub does. + ## [1.2.1] - 2026-10-01 ### Changed diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 4312d6c..c755e7e 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -10,7 +10,7 @@ We use [`uv`](https://docs.astral.sh/uv/) for environment and dependency managem ```bash # Clone and enter the repo -git clone https://github.com/babaliauskas/EvalShift.git +git clone https://github.com/evalshift/evalshift-cli.git cd evalshift # Create a virtualenv with Python 3.11 and install dev deps @@ -72,15 +72,15 @@ A release is a commit, a tag, and nothing else — CI does the publishing. asserts the tag matches `pyproject.toml`, builds with `uv build`, publishes to PyPI via [trusted publishing](https://docs.pypi.org/trusted-publishers/) (no token secret), and sends a `repository_dispatch` to - `babaliauskas/evalshift-action` so its `bump-cli-pin` workflow opens the + `evalshift/evalshift-action` so its `bump-cli-pin` workflow opens the pin-bump PR immediately. One-time setup, held outside the repo: - **PyPI trusted publisher** on the `evalshift` project: repository - `babaliauskas/evalshift-cli`, workflow `release.yml`, environment `pypi`. + `evalshift/evalshift-cli`, workflow `release.yml`, environment `pypi`. - **`EVALSHIFT_ACTION_DISPATCH_TOKEN`** (optional repo secret): fine-grained - PAT with contents: write on `babaliauskas/evalshift-action`. Without it the + PAT with contents: write on `evalshift/evalshift-action`. Without it the dispatch is skipped and the action repo's daily PyPI poll picks the release up within a day. diff --git a/DOCS.md b/DOCS.md index da3c322..9d5bc9f 100644 --- a/DOCS.md +++ b/DOCS.md @@ -58,7 +58,7 @@ uv pip install evalshift From source: ```bash -git clone https://github.com/babaliauskas/evalshift-cli +git clone https://github.com/evalshift/evalshift-cli cd evalshift-cli uv venv --python 3.11 source .venv/bin/activate @@ -89,8 +89,8 @@ EvalShift is four pieces. Each is released and documented independently; each ow | Piece | Distribution | What it does | Reference for humans | Reference for AI tools | | --- | --- | --- | --- | --- | | **CLI** | PyPI `evalshift` (import `evalshift_cli`) | Runs the suite on two models, scores, analyses, reports, bundles, pushes. | this document | | -| **SDK** | PyPI `evalshift-sdk` (import `evalshift`) | In-process capture: records your agent's model/tool calls to `.evalshift/captures/`. | [docs/sdk.md](docs/sdk.md), [SDK repo](https://github.com/babaliauskas/evalshift-sdk) | | -| **GitHub Action** | `babaliauskas/evalshift-action@v0` | Runs the pipeline on PRs, pushes the run, maintains one PR comment, sets the `evalshift/regression` status. | [docs/github-action.md](docs/github-action.md), [action repo](https://github.com/babaliauskas/evalshift-action) | | +| **SDK** | PyPI `evalshift-sdk` (import `evalshift`) | In-process capture: records your agent's model/tool calls to `.evalshift/captures/`. | [docs/sdk.md](docs/sdk.md), [SDK repo](https://github.com/evalshift/evalshift-sdk) | | +| **GitHub Action** | `evalshift/evalshift-action@v0` | Runs the pipeline on PRs, pushes the run, maintains one PR comment, sets the `evalshift/regression` status. | [docs/github-action.md](docs/github-action.md), [action repo](https://github.com/evalshift/evalshift-action) | | | **Hosted server** | service — API `https://api.evalshift.dev`, web app `https://evalshift.dev` | Stores pushed run bundles, diffs runs across branches, serves the web app, drives PR comments and gating. | [docs/hosted.md](docs/hosted.md) | covered by the CLI reference (`push`/`bundle` contract) | Data flow is one-directional: **SDK captures → CLI runs and bundles → server stores and diffs → web app displays.** The SDK and CLI never call each other — the interface is files under `.evalshift/captures/`. The CLI (import `evalshift_cli`) depends on the SDK (import `evalshift`), so one environment holds both. @@ -113,7 +113,7 @@ Doing it by hand, the mapping is: ## Quickstart -Point EvalShift at a real project. `evalshift init` writes a capture-first config, the [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk) records what your agent actually does, and `capture sync` turns those recordings into a golden suite: +Point EvalShift at a real project. `evalshift init` writes a capture-first config, the [evalshift-sdk](https://github.com/evalshift/evalshift-sdk) records what your agent actually does, and `capture sync` turns those recordings into a golden suite: ```bash evalshift init @@ -176,7 +176,7 @@ evalshift capture clean # delete promoted capture files + sweep o 7. Skips captures whose turn recorded an `error` event — a turn that died before the agent acted is not ground truth, and promoting it would assert `expected_no_tools: true` on a question that needed a tool. `--allow-errored` promotes it anyway (still never asserting `expected_no_tools`). `capture promote` exits non-zero on the same condition. Separately and unconditionally — `--allow-errored` does not help — a capture whose first `model_call` has no `toolset_ref` is refused: the SDK did not record what tools were offered, so there is nothing to carry, and re-capturing with a current `evalshift-sdk` is the only fix. 8. Warns when two captures claim the same `(conversation_id, turn_index)` (a retried turn), and when a promoted turn contains a failed tool result (`error`, or `{"success": false}`). Both stay warnings — see [Agent evals → What does not belong in a golden suite](docs/agents.md). 9. Writes `.evalshift/suites//golden.jsonl` and rewrites the managed `suites:` block in `evalshift.yaml` (between the `>>> evalshift suites` markers). -10. After the write (or after printing the block for you to paste), checks the CI pin: if a workflow under `.github/workflows/` uses `babaliauskas/evalshift-action` with an `evalshift-version` older than this CLI, with no pin at all, or with pins that are all newer than this CLI, it prints a warning naming the workflow and job plus the fix — the exact `evalshift-version: ""` line to set, or, for a newer pin, `pip install -U evalshift` locally. Advisory only — sync never edits a workflow and the exit code is unchanged. See [Pin drift](#pin-drift). +10. After the write (or after printing the block for you to paste), checks the CI pin: if a workflow under `.github/workflows/` uses `evalshift/evalshift-action` with an `evalshift-version` older than this CLI, with no pin at all, or with pins that are all newer than this CLI, it prints a warning naming the workflow and job plus the fix — the exact `evalshift-version: ""` line to set, or, for a newer pin, `pip install -U evalshift` locally. Advisory only — sync never edits a workflow and the exit code is unchanged. See [Pin drift](#pin-drift). Strictness knobs for the derived tool expectations: `--strict-args` (exact argument matches), `--names-only` (ignore arguments), `--tool-count` (also pin the call count, scoped the same way as `expected_tools`), `--rounds {first,all}` (which agent rounds become ground truth, default `first`). `--tag` attaches extra slice tags; `--print` previews the `suites:` block without writing. @@ -794,7 +794,7 @@ A suite wired under `suites:` is therefore selected by name — which is what `i `evalshift.yaml` is `extra="forbid"` everywhere, so the CLI that *reads* the config in CI must be at least as new as the CLI that *wrote* it locally — a newer `capture sync` or `init` can add keys an older release rejects outright. The action installs an exact version (`evalshift-version`, or its own default when the input is absent), which is where drift creeps in: you upgrade locally, re-sync, and CI still installs last month's release. -The CLI checks for this wherever it writes or validates config — `capture sync`, `init` (without `--ci`, next to a workflow it didn't write — `init --ci` pins the scaffolding CLI itself and does not warn about the file it just wrote), `doctor` (a `ci pin` row), and `validate` — by parsing every `.github/workflows/*.yml` for `babaliauskas/evalshift-action` steps and comparing their `evalshift-version` with its own: +The CLI checks for this wherever it writes or validates config — `capture sync`, `init` (without `--ci`, next to a workflow it didn't write — `init --ci` pins the scaffolding CLI itself and does not warn about the file it just wrote), `doctor` (a `ci pin` row), and `validate` — by parsing every `.github/workflows/*.yml` for `evalshift/evalshift-action` steps and comparing their `evalshift-version` with its own: - **stale** — a literal pin is older than the local CLI. Fix: set `evalshift-version: ""` on the step. - **unpinned** — a step has no `evalshift-version`, so the action default applies and may lag. Fix: add the pin. @@ -802,7 +802,7 @@ The CLI checks for this wherever it writes or validates config — `capture sync Equal pins, `${{ }}` expressions, unparseable versions, and an editable install without metadata (`0.0.0+unknown`) are silent. The check is advisory: it never edits a workflow and never changes an exit code, and in CI it is a no-op by construction (the running CLI *is* the pin). Config `version: 1` is not bumped for additive fields, nor for a removal that fails the load with a message naming the key — see [Configuration](docs/configuration.md#config-version-policy). -Secrets needed: a provider API key matching your config's models, and `EVALSHIFT_TOKEN` — a service account key from Settings → API tokens → Service accounts, scoped to `run:create` + `run:read` + `policy:read`, stored as an encrypted repository or environment secret. Not a personal token, never a literal in the workflow YAML, and never reachable from `pull_request_target`. Rotate by minting the successor first (24h grace), updating the secret, confirming a green run, then letting the old key expire. One thing a scoped key can't do, by design: auto-create the project (`project:create` is owner-only — pre-create it and set `create-project: false`). Full guidance: the action's [README](https://github.com/babaliauskas/evalshift-action#readme). +Secrets needed: a provider API key matching your config's models, and `EVALSHIFT_TOKEN` — a service account key from Settings → API tokens → Service accounts, scoped to `run:create` + `run:read` + `policy:read`, stored as an encrypted repository or environment secret. Not a personal token, never a literal in the workflow YAML, and never reachable from `pull_request_target`. Rotate by minting the successor first (24h grace), updating the secret, confirming a green run, then letting the old key expire. One thing a scoped key can't do, by design: auto-create the project (`project:create` is owner-only — pre-create it and set `create-project: false`). Full guidance: the action's [README](https://github.com/evalshift/evalshift-action#readme). --- @@ -993,6 +993,6 @@ Nowhere, by default. Model inputs/outputs go to the providers you configured (th - [docs/](docs/) — the mkdocs site: [getting-started](docs/getting-started.md), [configuration](docs/configuration.md), [evaluators](docs/evaluators.md), [methodology](docs/methodology.md), [agents](docs/agents.md), [conversations](docs/conversations.md), [traces](docs/traces.md), [sdk](docs/sdk.md), [hosted](docs/hosted.md), [github-action](docs/github-action.md), [faq](docs/faq.md) - [AGENTS.md](AGENTS.md) — repo orientation for AI coding agents; [CLAUDE.md](CLAUDE.md) — contributor workflow rules - [CHANGELOG.md](CHANGELOG.md) — release history -- [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk) — the in-process capture SDK +- [evalshift-sdk](https://github.com/evalshift/evalshift-sdk) — the in-process capture SDK - [llms-full.txt](llms-full.txt) — dense single-file reference for AI coding tools, hosted at - License: Apache-2.0 diff --git a/README.md b/README.md index 8a7e7f0..a8ffd3b 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ Open-source LLM migration and regression testing for AI agents. -[![CI](https://github.com/babaliauskas/evalshift-cli/actions/workflows/ci.yml/badge.svg)](https://github.com/babaliauskas/evalshift-cli/actions/workflows/ci.yml) +[![CI](https://github.com/evalshift/evalshift-cli/actions/workflows/ci.yml/badge.svg)](https://github.com/evalshift/evalshift-cli/actions/workflows/ci.yml) [![License: Apache 2.0](https://img.shields.io/badge/License-Apache_2.0-blue.svg)](LICENSE) [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://www.python.org/downloads/) [![PyPI](https://img.shields.io/pypi/v/evalshift.svg)](https://pypi.org/project/evalshift/) @@ -27,7 +27,7 @@ statistics**: paired tests, Cohen's d, 95% CIs, and Benjamini-Hochberg correction across every (prompt x evaluator x slice) comparison. An eval is only worth the examples in it. That is why the -[capture SDK](https://github.com/babaliauskas/evalshift-sdk) is part of the +[capture SDK](https://github.com/evalshift/evalshift-sdk) is part of the product rather than an add-on: it records real production runs — model calls, tool calls, final outputs — to disk, and `evalshift capture sync` promotes them into golden suites. Hand-written suites are fully supported too, but captured @@ -44,7 +44,7 @@ Four pieces, released and documented independently: | --- | --- | --- | | **SDK** — PyPI `evalshift-sdk` | Records what your agent actually did in production — model calls, tool calls, final output — as capture files on disk. Those captures become your golden suite. | [docs/sdk.md](docs/sdk.md) | | **CLI** — this repo, PyPI `evalshift` | Replays the suite on two models, scores, analyses, reports, bundles, pushes. | [DOCS.md](DOCS.md) | -| **GitHub Action** — `babaliauskas/evalshift-action@v0` | Runs the pipeline on pull requests, pushes the run, posts one PR comment, sets the `evalshift/regression` status. | [docs/github-action.md](docs/github-action.md) | +| **GitHub Action** — `evalshift/evalshift-action@v0` | Runs the pipeline on pull requests, pushes the run, posts one PR comment, sets the `evalshift/regression` status. | [docs/github-action.md](docs/github-action.md) | | **Hosted server** — `api.evalshift.dev`, web app at `evalshift.dev` | Optional. Stores pushed run bundles, diffs them across branches, drives PR comments and gating. | [docs/hosted.md](docs/hosted.md) | The SDK and the CLI never call each other — the interface is files under @@ -95,7 +95,7 @@ uv pip install evalshift-sdk # or: pip install evalshift-sdk From source (for contributors): ```bash -git clone https://github.com/babaliauskas/evalshift-cli.git +git clone https://github.com/evalshift/evalshift-cli.git cd evalshift-cli uv venv --python 3.11 source .venv/bin/activate @@ -113,7 +113,7 @@ candidate model to it. evalshift init # minimal capture-first evalshift.yaml ``` -Instrument the agent with [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk) +Instrument the agent with [evalshift-sdk](https://github.com/evalshift/evalshift-sdk) — stdlib-only, Python 3.10+, installed with the CLI or on its own: ```python @@ -242,7 +242,7 @@ The short version: `evalshift init --ci` scaffolds a production-shaped workflow: it discovers every committed suite under `.evalshift/suites/`, evaluates each on every -pull request via [`babaliauskas/evalshift-action@v0`](https://github.com/babaliauskas/evalshift-action) +pull request via [`evalshift/evalshift-action@v0`](https://github.com/evalshift/evalshift-action) (one matrix job per suite), pushes the runs to hosted EvalShift, compares against the latest compatible base-branch run, posts one PR comment, and gates merges on your `migration_policy` through a single required @@ -368,7 +368,7 @@ Runnable projects under [`examples/`](examples/): [Apache-2.0](LICENSE). Free for any use, commercial included — no share-back requirement, and an explicit patent grant. The capture SDK -([`evalshift-sdk`](https://github.com/babaliauskas/evalshift-sdk)), the piece +([`evalshift-sdk`](https://github.com/evalshift/evalshift-sdk)), the piece you import into your own application, is MIT. Earlier releases stay under the license they shipped with: `0.3.0` and earlier diff --git a/docs/agents.md b/docs/agents.md index e4e3cd2..8aa5cbe 100644 --- a/docs/agents.md +++ b/docs/agents.md @@ -60,7 +60,7 @@ parsed `ToolTrace`, and the configured `tool_*` evaluators score against the per-example `expected_tools` ground truth. For your own agent, the recommended path is capture-first: instrument it -with the [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk), +with the [evalshift-sdk](https://github.com/evalshift/evalshift-sdk), exercise it to record captures, then `evalshift capture sync` promotes them into a golden suite carrying the toolset your production agent actually offered — see [Getting started](getting-started.md). @@ -135,7 +135,7 @@ Nothing in `evalshift.yaml` wires a toolset to a prompt — dispatch reads it off each golden-suite *example* instead (`toolset_ref` or inline `tools`, see [Suite ground truth](#suite-ground-truth) below), so the same prompt can legitimately dispatch some examples with tools and others without, in one -run. [`examples/agent/tools.yaml`](https://github.com/babaliauskas/evalshift-cli/blob/main/examples/agent/tools.yaml) is just this project's human-readable record of what +run. [`examples/agent/tools.yaml`](https://github.com/evalshift/evalshift-cli/blob/main/examples/agent/tools.yaml) is just this project's human-readable record of what those tools are; it accepts either Anthropic-shape (`name` / `description` / `input_schema`) or OpenAI-shape (`{ "type": "function", "function": {...} }`) entries — `evalshift run` serialises whatever a toolset resolves to in diff --git a/docs/changelog.md b/docs/changelog.md index 8b70e7b..960298e 100644 --- a/docs/changelog.md +++ b/docs/changelog.md @@ -1,6 +1,6 @@ # Changelog -This page mirrors [`CHANGELOG.md`](https://github.com/babaliauskas/evalshift-cli/blob/main/CHANGELOG.md) +This page mirrors [`CHANGELOG.md`](https://github.com/evalshift/evalshift-cli/blob/main/CHANGELOG.md) in the repository. For the canonical, always-current version, see the file in the repo. diff --git a/docs/configuration.md b/docs/configuration.md index 31961a9..1844b2d 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -366,7 +366,7 @@ A list of prompt definitions. Each entry has: | `judge_model` | string | `gemini-3.1-flash-lite-preview` | Default LLM-as-judge model. | | `insights_model`| string | (none) | Model that writes the run-insights narrative rendered in `report.html` and uploaded with the bundle. Falls back to `judge_model` when unset — writing analytical prose is a harder task than a pairwise A/B verdict, so it is worth tuning separately. See [Run insights](#run-insights). | | `concurrency` | int | 10 (1 ≤ x ≤ 64) | Max in-flight LLM calls during `evalshift run` **and** `evalshift evaluate` (the embedding and judge calls made while scoring). | -| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions — for examples that offer tools, one entry per replayed round — plus `semantic` embeddings and `llm_judge` verdicts. See [Response cache](https://github.com/babaliauskas/evalshift-cli/blob/main/DOCS.md#response-cache) for the key. | +| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions — for examples that offer tools, one entry per replayed round — plus `semantic` embeddings and `llm_judge` verdicts. See [Response cache](https://github.com/evalshift/evalshift-cli/blob/main/DOCS.md#response-cache) for the key. | | `max_cost_usd` | float | 50.0 | Soft ceiling reserved for future enforcement. The pre-flight cost prompt currently triggers above $10 (skip with `--yes`). | | `max_tokens` | int | 4096 (`> 0`) | Completion length cap sent to every model call. Raise it if outputs are being truncated (the provider returns `finish_reason == "length"`); a `prompts[].max_tokens` entry overrides it per prompt. Truncated calls are detected, surfaced in the report, and **excluded from the regression statistics** so a cut-off output can't manufacture a false regression. | | `samples_per_example` | int | 1 (1 ≤ x ≤ 20) | How many times each `(prompt, example)` is sent to **each** model. Above 1, every sample is its own live call (the cache keys on the sample index), sample *i* of the source is scored against sample *i* of the target, and the example's row in `scores.jsonl` becomes the **mean over samples** with the per-sample scores and the within-example `delta_variance` under `metadata.samples`. The paired tests still run over examples, not samples, so this reduces noise without inflating `n`. Cost and the call count multiply by it; only worth turning on for a model that samples non-deterministically (see the report banner). See [Methodology](methodology.md#limitations-to-be-aware-of). | diff --git a/docs/getting-started.md b/docs/getting-started.md index f30e698..a7b5ebc 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -240,4 +240,4 @@ See [Hosted EvalShift](hosted.md) and [GitHub Action](github-action.md) for CI setup and privacy details. [uv]: https://docs.astral.sh/uv/ -[sdk]: https://github.com/babaliauskas/evalshift-sdk +[sdk]: https://github.com/evalshift/evalshift-sdk diff --git a/docs/github-action.md b/docs/github-action.md index 5b34ad5..7e7ff3f 100644 --- a/docs/github-action.md +++ b/docs/github-action.md @@ -182,12 +182,12 @@ The CLI checks for it wherever it writes or validates config — `capture sync`, `init` (without `--ci`, next to a workflow it did not write; `init --ci` pins the scaffolding CLI itself and does not warn about the file it just wrote), `doctor` (a `ci pin` row), and `validate`. It parses every `.github/workflows/*.yml` for -`babaliauskas/evalshift-action` steps and compares their `evalshift-version` +`evalshift/evalshift-action` steps and compares their `evalshift-version` with its own: ```text ⚠ CI installs evalshift 0.12.1 (.github/workflows/evalshift.yml, job evalshift) but the local CLI is 0.13.1 — an older CLI rejects config keys a newer one writes. - Fix: set `evalshift-version: "0.13.1"` on the babaliauskas/evalshift-action step. + Fix: set `evalshift-version: "0.13.1"` on the evalshift/evalshift-action step. ``` | Status | Trigger | Fix | diff --git a/docs/sdk.md b/docs/sdk.md index 7a1edf8..61c8467 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -1,13 +1,13 @@ # Capture SDK -The [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk) is a separate +The [evalshift-sdk](https://github.com/evalshift/evalshift-sdk) is a separate package (`pip install evalshift-sdk`, import name `evalshift`) that you install **inside your agent process**. It records what your agent actually did — model calls, tool calls, the final output — as JSON capture files on disk. The CLI promotes those captures into golden suites. This page covers the CLI side of that contract. The full SDK guide lives in the -SDK repo: [DOCS.md](https://github.com/babaliauskas/evalshift-sdk/blob/main/DOCS.md), +SDK repo: [DOCS.md](https://github.com/evalshift/evalshift-sdk/blob/main/DOCS.md), dense LLM reference at . ## The contract @@ -61,7 +61,7 @@ to state its masking policy in the call itself. `True` applies the SDK's `default_redactor` (emails, `sk-…`, `AKIA…`, `Bearer …`), `False` records verbatim, and a `(value) -> value` callable does something custom; any other value — `None` included — raises `TypeError`. Details: -[REDACTION.md](https://github.com/babaliauskas/evalshift-sdk/blob/main/docs/REDACTION.md). +[REDACTION.md](https://github.com/evalshift/evalshift-sdk/blob/main/docs/REDACTION.md). `tools` is required on the same entry points: the toolset the agent was offered, or `[]` if it never calls tools. Omitting it is a `TypeError` too. @@ -131,7 +131,7 @@ carries the recorded tool results so `run` replays later rounds teacher-forced `--keep-duplicates` disables dedup. Full behaviour: [Configuration](configuration.md) and the -[Capturing from production](https://github.com/babaliauskas/evalshift-cli/blob/main/DOCS.md#capturing-from-production) +[Capturing from production](https://github.com/evalshift/evalshift-cli/blob/main/DOCS.md#capturing-from-production) section of DOCS.md. ## 4. Evaluate a candidate against real behaviour @@ -145,7 +145,7 @@ from production traffic rather than hand-written examples. ## Related -- [`examples/capture-first/`](https://github.com/babaliauskas/evalshift-cli/tree/main/examples/capture-first) +- [`examples/capture-first/`](https://github.com/evalshift/evalshift-cli/tree/main/examples/capture-first) — every step above checked in: the instrumented agent, the captures and toolset sidecar it wrote, the promoted suite, and the `evalshift.yaml` whose managed `suites:` block `capture sync` filled in. diff --git a/examples/capture-first/README.md b/examples/capture-first/README.md index fdc6f1b..e6dccd1 100644 --- a/examples/capture-first/README.md +++ b/examples/capture-first/README.md @@ -1,7 +1,7 @@ # Capture-first example The flow the README and `evalshift init` recommend, checked in end to end: -instrument an agent with the [evalshift-sdk](https://github.com/babaliauskas/evalshift-sdk), +instrument an agent with the [evalshift-sdk](https://github.com/evalshift/evalshift-sdk), exercise it, then let `evalshift capture sync` turn what it recorded into a golden suite and wire that suite into `evalshift.yaml`. diff --git a/examples/capture-first/evalshift.yaml b/examples/capture-first/evalshift.yaml index 6f0b1bb..f57f838 100644 --- a/examples/capture-first/evalshift.yaml +++ b/examples/capture-first/evalshift.yaml @@ -1,5 +1,5 @@ # migration_profile: model-upgrade -# EvalShift configuration. See https://github.com/babaliauskas/EvalShift for docs. +# EvalShift configuration. See https://github.com/evalshift/evalshift-cli for docs. # # `init` writes only this file, set up for the capture-first flow: instrument # your agent with the evalshift-sdk, exercise it to record captures, then run diff --git a/llms-full.txt b/llms-full.txt index 18049cb..d5dfebb 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -23,7 +23,7 @@ Ecosystem — four pieces, one reference each. Fetch the one matching the task: | Piece | What it is | Reference for AI tools | | CLI (PyPI evalshift) | this document — run/score/analyze/report/bundle/push | https://www.evalshift.dev/cli-llms-full.txt | | SDK (PyPI evalshift-sdk) | in-process capture inside the user's agent | https://www.evalshift.dev/sdk-llms-full.txt | -| GitHub Action (babaliauskas/evalshift-action@v0) | PR run + comment + regression gate | https://www.evalshift.dev/ci-llms-full.txt | +| GitHub Action (evalshift/evalshift-action@v0) | PR run + comment + regression gate | https://www.evalshift.dev/ci-llms-full.txt | | Hosted server (api.evalshift.dev, web app evalshift.dev) | stores pushed bundles, diffs runs, drives PR comments/gating | this document (bundle/push contract) | Data flow: SDK captures -> CLI runs and bundles -> server stores/diffs -> web app displays. @@ -157,7 +157,7 @@ evalshift doctor Reports the toolset each configured suite carries (name, or the flat golden.jsonl); warns (warn-level, exit 0) when a suite's examples carry more than one distinct toolset -- legal (each example dispatches its own), but also the shape a wiring mistake takes. - Adds a `ci pin` row when any .github/workflows/*.yml uses babaliauskas/evalshift-action: + Adds a `ci pin` row when any .github/workflows/*.yml uses evalshift/evalshift-action: ok ("pinned to "; "not pinned (version unknown)" when every pin is a ${{ }} expression) when nothing drifts; warn (exit 0) on pin drift -- stale (pin < local), unpinned (input absent -> action default may lag), ahead (all pins > @@ -303,7 +303,7 @@ evalshift capture sync [--suite S] [--input-var NAME=input] [--tag T]... [--stri seeded from cases already promoted in the suite dir, so it spans sync runs. Reports "wired generation config for N case(s)" when promoted captures carried one. After writing (or printing the block to paste) warns on CI pin drift: a workflow step - `uses: babaliauskas/evalshift-action@...` whose evalshift-version is older than this CLI or + `uses: evalshift/evalshift-action@...` whose evalshift-version is older than this CLI or absent gets a warning naming workflow + job and the `evalshift-version: ""` to set; pins all newer than this CLI get one too, with the fix `pip install -U evalshift`. Advisory only: never edits the workflow, exit code unchanged. `init` (without --ci) warns @@ -1273,7 +1273,7 @@ evalshift compare --suite-name --to gemini-3.1-pro-preview --gate critic # baselines PRs diff against and are never cancelled. permissions: {contents: read, pull-requests: write, issues: write, statuses: write} env: {EVALSHIFT_NONINTERACTIVE: "1", GEMINI_API_KEY: "${{ secrets.GEMINI_API_KEY }}"} # key matches --provider -uses: babaliauskas/evalshift-action@v0 +uses: evalshift/evalshift-action@v0 with: {token: "${{ secrets.EVALSHIFT_TOKEN }}", config: evalshift.yaml, suite-name: "${{ matrix.suite }}", evalshift-version: "", fail-on: policy} diff --git a/mkdocs.yml b/mkdocs.yml index d2bd1c5..81f9b7c 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -1,8 +1,8 @@ site_name: EvalShift site_description: Open-source LLM migration and regression testing for AI agents. Compare two models, detect tool-call regressions, and gate model changes in CI. -site_url: https://babaliauskas.github.io/evalshift-cli/ -repo_url: https://github.com/babaliauskas/evalshift-cli -repo_name: babaliauskas/evalshift-cli +site_url: https://evalshift.github.io/evalshift-cli/ +repo_url: https://github.com/evalshift/evalshift-cli +repo_name: evalshift/evalshift-cli theme: name: material diff --git a/pyproject.toml b/pyproject.toml index 0282394..f37209d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -68,11 +68,11 @@ dev = [ ] [project.urls] -Homepage = "https://github.com/babaliauskas/evalshift-cli" -Documentation = "https://github.com/babaliauskas/evalshift-cli#readme" -Repository = "https://github.com/babaliauskas/evalshift-cli" -Issues = "https://github.com/babaliauskas/evalshift-cli/issues" -Changelog = "https://github.com/babaliauskas/evalshift-cli/blob/main/CHANGELOG.md" +Homepage = "https://github.com/evalshift/evalshift-cli" +Documentation = "https://github.com/evalshift/evalshift-cli#readme" +Repository = "https://github.com/evalshift/evalshift-cli" +Issues = "https://github.com/evalshift/evalshift-cli/issues" +Changelog = "https://github.com/evalshift/evalshift-cli/blob/main/CHANGELOG.md" [project.scripts] evalshift = "evalshift_cli.cli.main:app" diff --git a/src/evalshift_cli/cli/commands/_scaffold.py b/src/evalshift_cli/cli/commands/_scaffold.py index 408a340..9a62bec 100644 --- a/src/evalshift_cli/cli/commands/_scaffold.py +++ b/src/evalshift_cli/cli/commands/_scaffold.py @@ -251,7 +251,7 @@ - name: Run evalshift on ${{ matrix.suite }} if: ${{ env.HAS_EVALSHIFT_TOKEN == 'true' }} - uses: babaliauskas/evalshift-action@v0 + uses: evalshift/evalshift-action@v0 with: token: ${{ secrets.EVALSHIFT_TOKEN }} config: evalshift.yaml diff --git a/src/evalshift_cli/cli/commands/init.py b/src/evalshift_cli/cli/commands/init.py index 19787be..327e1c5 100644 --- a/src/evalshift_cli/cli/commands/init.py +++ b/src/evalshift_cli/cli/commands/init.py @@ -98,7 +98,7 @@ # Body of the minimal config, up to (but excluding) the suites region and the # migration_policy block. ``render_minimal_config`` appends those. _MINIMAL_YAML_BODY: Final = """\ -# EvalShift configuration. See https://github.com/babaliauskas/EvalShift for docs. +# EvalShift configuration. See https://github.com/evalshift/evalshift-cli for docs. # # `init` writes only this file, set up for the capture-first flow: instrument # your agent with the evalshift-sdk, exercise it to record captures, then run diff --git a/src/evalshift_cli/reports/templates/report.html.j2 b/src/evalshift_cli/reports/templates/report.html.j2 index d1ec7e1..4b50ebd 100644 --- a/src/evalshift_cli/reports/templates/report.html.j2 +++ b/src/evalshift_cli/reports/templates/report.html.j2 @@ -806,7 +806,7 @@ diff --git a/src/evalshift_cli/utils/ci_pin.py b/src/evalshift_cli/utils/ci_pin.py index 9594ba4..9f4ca15 100644 --- a/src/evalshift_cli/utils/ci_pin.py +++ b/src/evalshift_cli/utils/ci_pin.py @@ -5,7 +5,7 @@ GitHub Action installs an exact CLI version (``evalshift-version``, or its own default when the input is absent), which means the *reader* in CI must be at least as new as the *writer* on the developer's machine. This module finds -every ``babaliauskas/evalshift-action`` step under ``.github/workflows/`` and +every ``evalshift/evalshift-action`` step under ``.github/workflows/`` and compares its pin with the running CLI. Everything here is advisory: unreadable or invalid workflow files are skipped @@ -24,7 +24,10 @@ from packaging.version import InvalidVersion, Version #: ``uses:`` prefix that identifies an EvalShift action step. -ACTION_USES_PREFIX: Final = "babaliauskas/evalshift-action@" +ACTION_USES_PREFIX: Final = "evalshift/evalshift-action@" +#: Prefixes the action had before it moved from a personal account to the +#: `evalshift` org. GitHub redirects them, so existing workflows still use them. +LEGACY_ACTION_USES_PREFIXES: Final = ("babaliauskas/evalshift-action@",) #: The action input that pins the CLI version installed in CI. VERSION_INPUT: Final = "evalshift-version" #: Version reported by an editable install without package metadata. @@ -37,7 +40,7 @@ @dataclass(frozen=True, slots=True) class ActionPin: - """One ``babaliauskas/evalshift-action`` step and the CLI version it pins. + """One ``evalshift/evalshift-action`` step and the CLI version it pins. Attributes: workflow: Workflow file path relative to the project root. @@ -107,12 +110,24 @@ def find_action_pins(project_root: Path) -> list[ActionPin]: if not isinstance(step, dict): continue uses = step.get("uses") - if not isinstance(uses, str) or not uses.startswith(ACTION_USES_PREFIX): + if not isinstance(uses, str) or not _is_action_step(uses): continue pins.append(_pin_from_step(workflow, str(job_name), step)) return pins +def _is_action_step(uses: str) -> bool: + """Whether a ``uses:`` value names the EvalShift action under any of its owners. + + GitHub treats owner and repository names case-insensitively, so this does too. + """ + lowered = uses.lower() + return any( + lowered.startswith(prefix.lower()) + for prefix in (ACTION_USES_PREFIX, *LEGACY_ACTION_USES_PREFIXES) + ) + + def _load_jobs(path: Path) -> dict[Any, Any] | None: """Return the ``jobs:`` mapping of a workflow file, or ``None`` if unusable.""" try: @@ -253,6 +268,7 @@ def _ahead_finding(pins: list[ActionPin], cli_version: str) -> CiPinFinding: __all__ = [ "ACTION_USES_PREFIX", + "LEGACY_ACTION_USES_PREFIXES", "UNKNOWN_VERSION", "VERSION_INPUT", "WORKFLOWS_DIR", diff --git a/tests/integration/test_validate_command.py b/tests/integration/test_validate_command.py index 6fb147c..883e621 100644 --- a/tests/integration/test_validate_command.py +++ b/tests/integration/test_validate_command.py @@ -140,7 +140,7 @@ def test_warns_after_the_success_line_when_ci_pins_an_older_cli( workflow.parent.mkdir(parents=True) workflow.write_text( "on: push\njobs:\n evalshift:\n runs-on: ubuntu-latest\n steps:\n" - " - uses: babaliauskas/evalshift-action@v0\n" + " - uses: evalshift/evalshift-action@v0\n" ' with:\n evalshift-version: "0.0.1"\n', encoding="utf-8", ) diff --git a/tests/unit/test_ci_pin.py b/tests/unit/test_ci_pin.py index 90732cf..b76624a 100644 --- a/tests/unit/test_ci_pin.py +++ b/tests/unit/test_ci_pin.py @@ -27,13 +27,15 @@ def _workflow(root: Path, name: str, body: str) -> Path: return path -def _action_job(job: str, *, with_lines: str = "") -> str: +def _action_job( + job: str, *, with_lines: str = "", uses: str = "evalshift/evalshift-action@v0" +) -> str: return ( f" {job}:\n" " runs-on: ubuntu-latest\n" " steps:\n" " - uses: actions/checkout@v7\n" - " - uses: babaliauskas/evalshift-action@v0\n" + f" - uses: {uses}\n" " with:\n" ' token: "${{ secrets.EVALSHIFT_TOKEN }}"\n' + with_lines ) @@ -82,6 +84,33 @@ def test_literal_pin(self, tmp_path: Path) -> None: ) ] + @pytest.mark.parametrize( + "uses", + [ + # The action moved from a personal account to the `evalshift` org; + # GitHub redirects the old name, so existing workflows keep it. + "babaliauskas/evalshift-action@v0", + # GitHub owner and repository names are case-insensitive. + "Evalshift/evalshift-action@v0", + "EVALSHIFT/Evalshift-Action@v0", + ], + ) + def test_matches_moved_owner_and_any_case(self, tmp_path: Path, uses: str) -> None: + _workflow( + tmp_path, + "evalshift.yml", + "on: push\njobs:\n" + _action_job("evalshift", with_lines=_pinned("0.12.1"), uses=uses), + ) + assert [p.version for p in find_action_pins(tmp_path)] == ["0.12.1"] + + def test_other_owners_action_is_ignored(self, tmp_path: Path) -> None: + _workflow( + tmp_path, + "evalshift.yml", + "on: push\njobs:\n" + _action_job("evalshift", uses="someone-else/evalshift-action@v0"), + ) + assert find_action_pins(tmp_path) == [] + def test_absent_input_is_none(self, tmp_path: Path) -> None: _workflow(tmp_path, "evalshift.yml", "on: push\njobs:\n" + _action_job("evalshift")) [pin] = find_action_pins(tmp_path) @@ -130,7 +159,7 @@ def test_invalid_yaml_is_skipped_silently(self, tmp_path: Path) -> None: assert [p.job for p in find_action_pins(tmp_path)] == ["evalshift"] def test_non_yaml_files_are_ignored(self, tmp_path: Path) -> None: - _workflow(tmp_path, "README.md", "uses: babaliauskas/evalshift-action@v0\n") + _workflow(tmp_path, "README.md", "uses: evalshift/evalshift-action@v0\n") assert find_action_pins(tmp_path) == [] diff --git a/tests/unit/test_cli_capture.py b/tests/unit/test_cli_capture.py index 937dce4..288c0c6 100644 --- a/tests/unit/test_cli_capture.py +++ b/tests/unit/test_cli_capture.py @@ -1958,7 +1958,7 @@ def _write_stale_workflow(root: Path, version: str = "0.0.1") -> Path: path.parent.mkdir(parents=True) path.write_text( "on: push\njobs:\n evalshift:\n runs-on: ubuntu-latest\n steps:\n" - " - uses: babaliauskas/evalshift-action@v0\n" + " - uses: evalshift/evalshift-action@v0\n" f' with:\n evalshift-version: "{version}"\n', encoding="utf-8", ) diff --git a/tests/unit/test_doctor.py b/tests/unit/test_doctor.py index 032f00d..dc87fcc 100644 --- a/tests/unit/test_doctor.py +++ b/tests/unit/test_doctor.py @@ -571,7 +571,7 @@ def _workflow(root: Path, with_lines: str) -> None: path.parent.mkdir(parents=True) path.write_text( "on: push\njobs:\n evalshift:\n runs-on: ubuntu-latest\n steps:\n" - " - uses: babaliauskas/evalshift-action@v0\n" + " - uses: evalshift/evalshift-action@v0\n" " with:\n" + with_lines, encoding="utf-8", ) diff --git a/tests/unit/test_init.py b/tests/unit/test_init.py index 32bd98c..9ac113a 100644 --- a/tests/unit/test_init.py +++ b/tests/unit/test_init.py @@ -383,7 +383,7 @@ def test_no_ci_flag_skips_workflow(self, in_tmp: Path) -> None: def test_ci_flag_writes_valid_workflow(self, in_tmp: Path) -> None: body, wf = self._workflow(in_tmp) - assert "babaliauskas/evalshift-action@v0" in body + assert "evalshift/evalshift-action@v0" in body assert "EVALSHIFT_TOKEN" in body assert "EVALSHIFT_NONINTERACTIVE" in body jobs = wf["jobs"] @@ -442,7 +442,7 @@ def test_selects_each_suite_by_name_not_by_path(self, in_tmp: Path) -> None: assert isinstance(jobs, dict) steps = jobs["evalshift"]["steps"] (action_step,) = [ - step for step in steps if str(step.get("uses", "")).startswith("babaliauskas/") + step for step in steps if str(step.get("uses", "")).startswith("evalshift/") ] assert action_step["with"]["suite-name"] == "${{ matrix.suite }}" assert "suite" not in action_step["with"] @@ -514,7 +514,7 @@ def _stale_workflow(root: Path) -> None: path.parent.mkdir(parents=True) path.write_text( "on: push\njobs:\n evalshift:\n runs-on: ubuntu-latest\n steps:\n" - " - uses: babaliauskas/evalshift-action@v0\n" + " - uses: evalshift/evalshift-action@v0\n" ' with:\n evalshift-version: "0.0.1"\n', encoding="utf-8", )