From 9823d2c777dbfe5443b5848c94ab8f0d123a6401 Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 12:46:27 +0200 Subject: [PATCH 1/7] test(docs): pin the documented version to pyproject DOCS.md still advertised 1.0.1 after 1.1.0 shipped, while llms-full.txt had the right number. Both headers are now checked against evalshift_cli.__version__, the installed metadata of pyproject's version. Co-Authored-By: Claude Opus 5.5 (1M context) --- DOCS.md | 2 +- tests/unit/test_docs_currency.py | 23 +++++++++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/DOCS.md b/DOCS.md index 76e0dbd..3f75645 100644 --- a/DOCS.md +++ b/DOCS.md @@ -12,7 +12,7 @@ hosted (opt-in) run history, diffs, PR gates The suite is the crux, so the capture SDK is the recommended way to build one: it records real production runs to disk and `evalshift capture sync` promotes them into golden suites. Hand-written suites are fully supported — see [The golden suite](#the-golden-suite). -- **Package name:** `evalshift` · **CLI entry point:** `evalshift` · **version:** 1.0.1 +- **Package name:** `evalshift` · **CLI entry point:** `evalshift` · **version:** 1.1.0 - **Python:** >= 3.11 · **License:** Apache-2.0 · **Status:** stable - **Local-first.** Runs, scores, stats, and reports all happen on your machine under `.evalshift/`. The only network calls are the model API calls you asked for — and, if you opt in, pushes to the hosted service. - **Four pieces:** CLI (this doc), SDK, GitHub Action, hosted server — each with its own machine-readable reference for AI tools. See [Ecosystem and AI-tool references](#ecosystem-and-ai-tool-references). diff --git a/tests/unit/test_docs_currency.py b/tests/unit/test_docs_currency.py index c55da5e..879f8b9 100644 --- a/tests/unit/test_docs_currency.py +++ b/tests/unit/test_docs_currency.py @@ -18,6 +18,7 @@ import pytest +import evalshift_cli from evalshift_cli.cli.commands.init import PROVIDERS from evalshift_cli.models.registry import PROVIDER_ENV_VARS @@ -142,3 +143,25 @@ def test_docs_list_every_init_provider(name: str) -> None: assert f"--provider {'|'.join(PROVIDERS)}" in text, ( f"{name} lists `init --provider` choices that differ from init.PROVIDERS" ) + + +#: The two reference files that print the CLI version in their header. +#: `DOCS.md` writes `**version:** X`, `llms-full.txt` writes `version: X`. +VERSION_HEADER_FILES: tuple[str, ...] = ("DOCS.md", "llms-full.txt") + + +@pytest.mark.parametrize("name", VERSION_HEADER_FILES) +def test_reference_header_version_matches_the_package(name: str) -> None: + """The documented version is the released one, not the one before it. + + `evalshift_cli.__version__` is the installed metadata of `pyproject.toml`'s + `version`. The two headers drifted apart once already: `llms-full.txt` said + 1.1.0 while `DOCS.md` still said 1.0.1. + """ + text = (REPO_ROOT / name).read_text(encoding="utf-8") + match = re.search(r"version:(?:\*\*)? (\S+)", text) + + assert match is not None, f"{name} has no `version:` line in its header" + assert match.group(1) == evalshift_cli.__version__, ( + f"{name} says version {match.group(1)}; the package is {evalshift_cli.__version__}" + ) From 9fa58ab361b56d167abeba22e2f52104c7f690f6 Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 12:49:35 +0200 Subject: [PATCH 2/7] docs(config): SliceConfig docstring matches behaviour The docstring described filter as a Python expression evaluated against each example's inputs. It is a literal tag, and analysis does not read the top-level slices: block at all: slices come from example tags and per-slice budgets are keyed by tag under migration_policy.slices. Docstring only; no behaviour change. Co-Authored-By: Claude Opus 5.5 (1M context) --- src/evalshift_cli/config/models.py | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/src/evalshift_cli/config/models.py b/src/evalshift_cli/config/models.py index 2fa72c1..f898c6a 100644 --- a/src/evalshift_cli/config/models.py +++ b/src/evalshift_cli/config/models.py @@ -413,16 +413,18 @@ def tool_evaluator_names(self) -> frozenset[str]: class SliceConfig(_StrictModel): - """A named slice — a subset of suite examples to analyse separately. + """One entry of the top-level ``slices:`` block. + + The block is validated (the reserved ``overall`` name is rejected) and + recorded in the run bundle, but analysis does not read it today: slices + come from example ``tags`` -- one per distinct tag, plus ``all`` -- and + per-slice budgets are keyed by tag under ``migration_policy.slices``. + None of the fields below renames, filters, or scopes anything. Attributes: - name: Human-readable slice name surfaced in reports. - filter: A Python expression evaluated against each example's inputs. - Examples in the slice are those for which the expression is truthy. - (Evaluation safety is the responsibility of the runtime; for the - MVP we will document that ``evalshift.yaml`` is treated as - trusted, like any project config file.) - applies_to: Glob list of prompt IDs this slice applies to. + name: Slice name. + filter: A literal tag, not an expression. + applies_to: Glob list of prompt IDs. """ name: str = Field(min_length=1) From d9d2c173e4b572523f15f2b89ce65cf610437546 Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 13:03:23 +0200 Subject: [PATCH 3/7] docs: correct reference facts in DOCS.md and llms-full.txt Audit findings F1-F26 against the code, mirrored across both references: the init profile table (INIT_PROFILE_POLICIES numbers plus a tool-divergence column), a multi-turn example that now validates, the top-level slices: block documented as validated but not applied, resume hashing the suite path rather than its contents, the response cache serving tool-less examples only (with the full key), compare opening the report only with --open, the CLI import name and single-environment install, the complete failure-label set, optional config version, --no-insights skipping silently, --policy-gate failing with no policy, EVALSHIFT_DIR's full reach, the action's full input list and require-policy, exit code 2, and login reusing a still-valid token. Twins from the pages audit are fixed here too: imported traces stay local, push reuses an existing bundle, bundle needs a git SHA, and failed calls become errored rows excluded from the statistics. Co-Authored-By: Claude Opus 5.5 (1M context) --- DOCS.md | 89 +++++++++++++++------------- llms-full.txt | 159 ++++++++++++++++++++++++++++++-------------------- 2 files changed, 144 insertions(+), 104 deletions(-) diff --git a/DOCS.md b/DOCS.md index 3f75645..a66cbf3 100644 --- a/DOCS.md +++ b/DOCS.md @@ -88,12 +88,12 @@ EvalShift is four pieces. Each is released and documented independently; each ow | Piece | Distribution | What it does | Reference for humans | Reference for AI tools | | --- | --- | --- | --- | --- | -| **CLI** | PyPI `evalshift` (import `evalshift`) | Runs the suite on two models, scores, analyses, reports, bundles, pushes. | this document | | +| **CLI** | PyPI `evalshift` (import `evalshift_cli`) | Runs the suite on two models, scores, analyses, reports, bundles, pushes. | this document | | | **SDK** | PyPI `evalshift-sdk` (import `evalshift`) | In-process capture: records your agent's model/tool calls to `.evalshift/captures/`. | [docs/sdk.md](docs/sdk.md), [SDK repo](https://github.com/babaliauskas/evalshift-sdk) | | | **GitHub Action** | `babaliauskas/evalshift-action@v0` | Runs the pipeline on PRs, pushes the run, maintains one PR comment, sets the `evalshift/regression` status. | [docs/github-action.md](docs/github-action.md), [action repo](https://github.com/babaliauskas/evalshift-action) | | | **Hosted server** | service — API `https://api.evalshift.dev`, web app `https://evalshift.dev` | Stores pushed run bundles, diffs runs across branches, serves the web app, drives PR comments and gating. | [docs/hosted.md](docs/hosted.md) | covered by the CLI reference (`push`/`bundle` contract) | -Data flow is one-directional: **SDK captures → CLI runs and bundles → server stores and diffs → web app displays.** The SDK and CLI never call each other — the interface is files under `.evalshift/captures/`. Because both use the top-level import name `evalshift`, install them in **separate virtual environments**. +Data flow is one-directional: **SDK captures → CLI runs and bundles → server stores and diffs → web app displays.** The SDK and CLI never call each other — the interface is files under `.evalshift/captures/`. The CLI (import `evalshift_cli`) depends on the SDK (import `evalshift`), so one environment holds both. The CLI reference is generated from [llms-full.txt](llms-full.txt) at this repo's root — edit that file when CLI behaviour changes. The SDK and Action references are owned by their own repos; the copies served from `evalshift.dev` are synced from there. @@ -122,7 +122,7 @@ evalshift capture sync evalshift compare --suite-name --to ``` -`evalshift compare` drives the full pipeline — `doctor → run → evaluate → analyze → report` — over one suite under one live progress display, then opens `report.html`: a single-file, offline-capable HTML report with per-prompt/per-slice comparisons, severity badges, effect sizes with 95% CIs, and a migration-policy verdict panel. `run`/`compare` estimate worst-case cost up front and prompt for confirmation above $10 (skip with `--yes`). +`evalshift compare` drives the full pipeline — `doctor → run → evaluate → analyze → report` — over one suite under one live progress display, then writes `report.html` (add `--open` to open it in your browser): a single-file, offline-capable HTML report with per-prompt/per-slice comparisons, severity badges, effect sizes with 95% CIs, and a migration-policy verdict panel. `run`/`compare` estimate worst-case cost up front and prompt for confirmation above $10 (skip with `--yes`). See [Project setup](#project-setup) and [Capturing from production](#capturing-from-production). No captures to work from? Write `golden.jsonl` by hand — see [The golden suite](#the-golden-suite). @@ -130,20 +130,22 @@ See [Project setup](#project-setup) and [Capturing from production](#capturing-f ## Project setup -`evalshift init` is the real-project entry point. Writes **only** a minimal, capture-first `evalshift.yaml`: a passthrough `replay` prompt (`content: "{input}"`), advisory semantic + LLM-judge evaluators, an empty managed `suites:` block for `capture sync` to fill, and a migration policy. The intended flow: instrument your agent with the evalshift-sdk → record captures → `evalshift capture sync` → run against the promoted suite. +`evalshift init` is the real-project entry point. Writes **only** a minimal, capture-first `evalshift.yaml`: a passthrough `replay` prompt (`content: "{input}"`), advisory LLM-judge and (for Gemini and OpenAI) semantic evaluators, an empty managed `suites:` block for `capture sync` to fill, and a migration policy. The intended flow: instrument your agent with the evalshift-sdk → record captures → `evalshift capture sync` → run against the promoted suite. `init` options: - `--provider gemini|openai|anthropic|deepseek` — which provider's model ids the scaffold uses (prompted on a TTY; defaults to `gemini` otherwise). Gemini and OpenAI scaffolds include an embedding-based semantic evaluator; the Anthropic and DeepSeek scaffolds comment it out (no embedding endpoint). - `--profile` — pre-tuned migration-policy budgets: -| Profile | regression ≤ | critical ≤ | equivalence ≥ | arg drift ≤ | cost Δ ≤ | latency Δ ≤ | -|---|---|---|---|---|---|---| -| `model-upgrade` (default) | 3% | 0 | 95% | 1% | +20% | +30% | -| `cost-reduction` | 2% | 0 | 97% | 1% | +5% | +30% | -| `local-model` | 5% | 0 | 90% | 2% | +0% | +50% | -| `quantization` | 2% | 0 | 97% | 0.5% | +0% | +20% | -| `provider-switch` | 3% | 0 | 95% | 1% | +20% | +40% | +| Profile | regression ≤ | critical ≤ | equivalence ≥ | arg drift ≤ | tool divergence ≤ | cost Δ ≤ | latency Δ ≤ | +|---|---|---|---|---|---|---|---| +| `model-upgrade` (default) | 30% | 1 | 75% | 20% | 20% | +30% | +30% | +| `cost-reduction` | 2% | 0 | 97% | 1% | 2% | +5% | +30% | +| `local-model` | 5% | 0 | 90% | 2% | 5% | +0% | +50% | +| `quantization` | 2% | 0 | 97% | 0.5% | 2% | +0% | +20% | +| `provider-switch` | 3% | 0 | 95% | 1% | 3% | +20% | +40% | + +The `model-upgrade` row is the loose first-migration starting point — the `migration_policy` field defaults (see [Migration policy and CI gating](#migration-policy-and-ci-gating)); the other four are tighter presets to move to as the suite grows. - `--ci` — also scaffold `.github/workflows/evalshift.yml` (see [GitHub Action](#github-action)). - `--wire-agents` (default on) — write `EVALSHIFT.md`, a guide for AI coding agents, and point existing agent files (`AGENTS.md`, `CLAUDE.md`, `GEMINI.md`, `.cursorrules`, `.github/copilot-instructions.md`) at it, creating `AGENTS.md` if none exist. Both the guide and the pointer blocks link the three hosted llms.txt references — [cli-llms-full.txt](https://evalshift.dev/cli-llms-full.txt), [sdk-llms-full.txt](https://evalshift.dev/sdk-llms-full.txt), and [ci-llms-full.txt](https://evalshift.dev/ci-llms-full.txt) (GitHub Action). Idempotent; disable with `--no-wire-agents`. @@ -206,7 +208,7 @@ init → doctor → run → evaluate → analy ``` - **`doctor`** validates local config and shows which provider keys are visible. Exit 1 only when an existing `evalshift.yaml` fails validation; missing keys are soft warnings. Its second row, `evalshift-sdk`, reports the SDK version the `evalshift` import name resolves to in this environment (`warn` when the SDK is missing or shadowed by an older CLI's leftover files; never a failure). It also reports the toolset each configured suite carries (or the flat `golden.jsonl`) and flags a suite whose examples carry more than one distinct toolset — legal (each example dispatches its own), but also the shape a wiring mistake takes. When a workflow under `.github/workflows/` uses the GitHub Action it adds a `ci pin` row: `ok` (`pinned to `) when CI installs this CLI version, `warn` when the pin is older, absent, or newer than the local CLI (see [Pin drift](#pin-drift)). When the config wires an `llm_judge` evaluator and names both `defaults.source_model` and `target_model`, a `judge family` row warns for every `judge_model` that resolves to the same provider as an arm (self-preference bias; `ok` "from a third family" otherwise, no row when either arm is unset) — advisory, never a failure; `validate` prints the same line and the report repeats it above the verdict (see [`evaluators.llm_judge`](docs/configuration.md#evaluatorsllm_judge)). -- **`run`** parses prompts, validates every example against every prompt, estimates cost, then dispatches `(prompt × example × {source, target})` calls through an async orchestrator under a concurrency semaphore. Responses are cached; progress is checkpointed every 50 completions. +- **`run`** parses prompts, validates every example against every prompt, estimates cost, then dispatches `(prompt × example × {source, target})` calls through an async orchestrator under a concurrency semaphore. Responses to tool-less examples are cached (tool-calling examples are always dispatched live — see [Response cache](#response-cache)); progress is checkpointed every 50 completions. - **`evaluate`** scores each (source, target) pair with the configured evaluators, one `EvalRecord` per pair × evaluator. Scoring runs under the same `defaults.concurrency` semaphore as `run`, and the embedding/judge calls it makes go through the same response cache. - **`analyze`** runs paired statistics per `(prompt, evaluator, slice)`, applies Benjamini–Hochberg FDR correction, classifies severities, and — when a `migration_policy` is configured — computes a pass/fail verdict. - **`report`** renders the single-file HTML report (no external assets; works offline and attaches cleanly to a PR or email), and writes the machine-written [run insights](#run-insights) narrative unless `--no-insights` is passed. The page opens on a verdict / advisory-signal / economics panel row and a six-cell run strip (examples, calls, failed-or-truncated, spend, latency Δ, mean score Δ), then the executive summary, the narrative, one section per prompt, and the methodology. Every figure on it is derived from the run's own artefacts; the deltas in the header are the run-level rollup of the per-prompt economics. Top regressions are collapsed cards — expand one for the trace diff, the tool diffs and the conversation context. The report is dark-only. @@ -218,8 +220,8 @@ Run ids look like `r_20260722_golden_a1b2c3` (`r___`). Un | File | Written by | Contents | |---|---|---| -| `state.json` | run | Run status, models, config hash, progress counters, `non_deterministic_models`, `dropped_params`, `evaluator_coverage` — attempted vs recorded per axis, the pairs that produced no row, and the axis's `blocking` flag (atomic write) | -| `raw.jsonl` | run | One line per model call: rendered prompt, output, tokens, cost, latency, tool trace, error | +| `state.json` | run | Run status, models, config hash, progress counters, `non_deterministic_models`, `dropped_params`, and — added by `evaluate` — `evaluator_coverage`: attempted vs recorded per axis, the pairs that produced no row, and the axis's `blocking` flag (atomic write) | +| `raw.jsonl` | run | One line per (prompt, example, role, sample): rendered prompt, output, tokens, cost, latency, tool trace, error. A teacher-forced multi-round replay is one row (tokens, cost and latency summed over its rounds) | | `scores.jsonl` | evaluate | One line per (pair × evaluator): source/target scores, delta, explanation | | `analysis.json` | analyze | Per-comparison statistics, severities, notes | | `migration_decision.json` | analyze | Policy verdict + per-budget detail (only when `migration_policy` set) | @@ -231,11 +233,11 @@ Run ids look like `r_20260722_golden_a1b2c3` (`r___`). Un ### Checkpointing and resume -`state.json` records a `config_hash` (SHA-256 over the canonicalised config plus the suite path). `run --resume` picks up the most recent in-progress run, verifies the hash still matches (aborts if config or suite changed), and skips every `(prompt, example, role)` already present in `raw.jsonl`. Calls that errored are counted as done — they are not retried automatically. +`state.json` records a `config_hash` (SHA-256 over the canonicalised config plus the suite path). `run --resume` picks up the most recent in-progress run, verifies the hash still matches (aborts if the config or the suite **path** changed — suite *contents* are not hashed, so start a fresh run after editing examples), and skips every `(prompt, example, role, sample)` already present in `raw.jsonl`. Calls that errored are counted as done — they are not retried automatically. ### Response cache -Live responses are cached in SQLite at `~/.evalshift/cache.db`, keyed by SHA-256 over canonical JSON of `(model, prompt, inputs, temperature, max_tokens[, history])`, with a 7-day TTL. Re-running an identical evaluation is nearly free. Disable per-project with `defaults.cache: false`; wipe with `evalshift cache clear`. +Live responses to **tool-less** examples are cached in SQLite at `~/.evalshift/cache.db`, keyed by SHA-256 over canonical JSON of `(model, prompt, inputs, temperature, max_tokens[, history])` — plus `generation_config`, the toolset fingerprint, the round index and the sample index when each is set — with a 7-day TTL. Re-running an identical tool-less evaluation costs no run-stage calls. **Examples with a non-empty toolset are not cached:** every `run` of an agent suite dispatches them live, at full price. The evaluate-stage embedding and judge caches below still apply to them. Disable per-project with `defaults.cache: false`; wipe with `evalshift cache clear`. The cache covers the evaluate stage too: `semantic` embeddings are keyed by `(embedding model, text)`, and `llm_judge` verdicts by `(judge model, criterion, source output, target output)`. The judge key uses a canonical A/B ordering, so the per-call orientation randomization doesn't halve the hit rate — the orientation that was actually used is recorded with the verdict and replayed on a hit, leaving `metadata.target_was_a` faithful. @@ -292,7 +294,6 @@ migration_policy: min_equivalence_rate: 0.75 max_tool_argument_drift: 0.20 max_tool_divergence: 0.20 - tool_argument_drift_floor: 0.9 max_cost_increase: 0.30 max_latency_increase: 0.30 @@ -308,12 +309,12 @@ suites: {} | Field | Type / default | Meaning | |---|---|---| -| `version` | literal `1`, required | Config schema version | +| `version` | literal `1`, default `1` (optional) | Config schema version | | `project` | `str \| None` | Hosted project slug, `org/project` (regex `^[a-z0-9-]+/[a-z0-9-]+$`) | | `prompts` | list, required, ≥1 | Prompt definitions (unique ids enforced) | | `defaults` | block | Run defaults, below | | `evaluators` | block | Evaluator configs, see [Evaluators](#evaluators) | -| `slices` | list | Named suite subsets, below | +| `slices` | list | Validated and recorded in the bundle, but **not applied** — slices come from example tags. See below | | `migration_policy` | block \| absent | Regression budgets, see [Migration policy](#migration-policy-and-ci-gating) | | `suites` | map | Named suites (`{name: {source: captured\|jsonl, path: ..., evaluators: ..., managed: true}}`); the block between the `>>> evalshift suites` markers is managed by `capture sync`. See [Per-suite evaluators](#per-suite-evaluators) | | `retention` | block | `max_runs_per_suite` (default 20, `0` disables), `run_ttl_days` (default off) | @@ -342,14 +343,16 @@ Nothing replaced it. Delete the block; express any gate you meant by it as a `mi ### `slices` +Slices come from the suite, not from this block: every distinct example `tag` becomes a slice under its own name, alongside the implicit `all` slice. Every configured evaluator is analysed once overall and once per slice. Per-slice budgets go under [`migration_policy.slices`](#migration-policy-and-ci-gating), keyed by the tag. + ```yaml -slices: +slices: # validated and recorded in the run bundle; NOT applied by analysis - name: security - filter: security # matched against each example's tags list - applies_to: ["*"] # optional: restrict to prompt ids + filter: security # a literal tag + applies_to: ["*"] # glob list of prompt ids ``` -A slice collects the examples whose `tags` contain the `filter` string. Every configured evaluator is analysed once overall and once per slice, and migration-policy budgets can be tightened per slice. `overall` is reserved — it names the run-level scope in the run bundle — and is rejected as a slice `name`, as an example tag, and as a `migration_policy.slices` key. +The top-level `slices:` block still loads — it is validated and copied into the run bundle's evaluator config — but analysis does not read it today: `name`, `filter` and `applies_to` rename, filter and scope nothing, and a run reports the same slices with or without it. `overall` is reserved — it names the run-level scope in the run bundle — and is rejected as a slice `name`, as an example tag, and as a `migration_policy.slices` key. Slices holding exactly the same examples are collapsed to one before any test runs — duplicates restate the same numbers as if they were independent findings and skew the Benjamini–Hochberg correction anti-conservatively (extra copies of a p-value shrink every adjusted p-value in the family, so results look more significant than they are). `all` and any slice named under `migration_policy.slices` always survive; otherwise the provenance tag `captured` (written by `capture promote`) loses to an ordinary tag, then alphabetical order decides. Drops are reported on the terminal and as `collapsed_slices` in `analysis.json`. See [docs/methodology.md](docs/methodology.md). @@ -432,7 +435,7 @@ Each entry in `expected_tools`: - `match_strategy`: `exact` (arguments must match exactly), `subset` (default; expected keys must be present and equal, extras allowed), `contains_per_field` (per-field containment). - `provenance`: `captured` (default, what `capture promote`/`sync` write — transcribed from the source model's own call, unverified) or `reviewed` (a human has confirmed it). Scoring is identical; the flag only decides whether the run discloses that its ground truth is source-derived — see [Agent evaluation → Ground truth](#ground-truth). -Validators enforced at load: exactly one of `toolset_ref` / `tools` is required — neither, or both, fails to load; `expected_no_tools: true` is incompatible with non-empty `expected_tools`, a non-empty `expected_tool_rounds`, or a nonzero `expected_tool_count`; `tool_result_fixtures` requires `expected_tool_rounds`, cannot cover more rounds than it has, and every covered round needs one result per expected call with a matching `tool_name`; `history` may contain at most one `system` message and it must come first; duplicate ids across the suite are rejected. The loader collects **all** schema errors before failing, so you fix a broken suite in one pass. +Validators enforced at load: exactly one of `toolset_ref` / `tools` is required — neither, or both, fails to load; `expected_no_tools: true` is incompatible with non-empty `expected_tools`, a non-empty `expected_tool_rounds`, or a nonzero `expected_tool_count` — and so is an empty toolset, spelled `tools: []` or a `toolset_ref` naming the empty toolset (a call offered no tools cannot produce a tool call); `tool_result_fixtures` requires `expected_tool_rounds`, cannot cover more rounds than it has, and every covered round needs one result per expected call with a matching `tool_name`; `history` may contain at most one `system` message and it must come first; duplicate ids across the suite are rejected. The loader collects **all** schema errors before failing, so you fix a broken suite in one pass. --- @@ -476,7 +479,7 @@ Evaluators score each (source, target) output pair. Scores live in `[0, 1]`; the - `blocking: true|false` (default `true`) — **blocking** evaluators feed the migration-policy verdict and CI gates; **advisory** (`blocking: false`) evaluators are computed, reported, and summarised separately but can never fail a run on their own. The `init` scaffold ships semantic and judge as advisory deliberately: at small suite sizes their noise would gate the verdict. - `applies_to: ["*"]` — restrict to specific prompt ids (where supported). -When an evaluator's own measurement breaks (judge call fails, embedding call fails), the record is stored as **errored and excluded from the statistics** — not silently scored neutral. Upstream failures are different: if a *model call* failed or was truncated, the pair gets a neutral 0.5/0.5 record with the error attached, so the run always completes. +When an evaluator's own measurement breaks (judge call fails, embedding call fails), the record is stored as **errored and excluded from the statistics** — not silently scored neutral. Upstream failures are recorded the same way: if a *model call* failed or was truncated, the pair gets an errored row (a 0.5/0.5 placeholder with the error attached), kept in `scores.jsonl` and excluded from the statistics. The run always completes. ### Structural (`evaluators.structural`, list) — free, no API calls @@ -505,13 +508,13 @@ Pairwise A/B comparison per criterion: the judge sees the two outputs anonymised ### Failure categories -Regressions carry machine-readable labels that the report and hosted diff group by: `FORMAT_FAILURE`, `SEMANTIC_REGRESSION`, `TOOL_SELECTION_DRIFT`, `ARGUMENT_VALUE_DRIFT`, `TOOL_TRACE_STRUCTURE_DRIFT`, `TOOL_ORDER_DRIFT`, `DANGEROUS_ACTION_DRIFT`, `MISSING_VERIFICATION_STEP`, `UNNECESSARY_TOOL_CALL`, and `REFUSAL_REGRESSION`. That is the complete set — every label the evaluators emit is declared in `evaluators/failures.py`. The machine labels live in `scores.jsonl`, `report.json` and the bundle; every rendered surface (the HTML report, decision prose, the run narrative) shows the plain-language display name instead — `TOOL_SELECTION_DRIFT` renders as "Different tools chosen" — with the mapping declared beside the labels in `evaluators/failures.py`. +Regressions carry machine-readable labels that the report and hosted diff group by: `FORMAT_FAILURE`, `SEMANTIC_REGRESSION`, `TOOL_SELECTION_DRIFT`, `TOOL_GROUND_TRUTH_MISS` (both models missed the same recorded tool ground truth — a broken-harness signal, not a migration finding; see [Migration policy and CI gating](#migration-policy-and-ci-gating)), `ARGUMENT_VALUE_DRIFT`, `TOOL_TRACE_STRUCTURE_DRIFT`, `TOOL_ORDER_DRIFT`, `DANGEROUS_ACTION_DRIFT`, `MISSING_VERIFICATION_STEP`, `UNNECESSARY_TOOL_CALL`, and `REFUSAL_REGRESSION`. That is the complete set — every label the evaluators emit is declared in `evaluators/failures.py`. The machine labels live in `scores.jsonl`, `report.json` and the bundle; every rendered surface (the HTML report, decision prose, the run narrative) shows the plain-language display name instead — `TOOL_SELECTION_DRIFT` renders as "Different tools chosen" — with the mapping declared beside the labels in `evaluators/failures.py`. `ARGUMENT_VALUE_DRIFT` counts **regressions**: it is stamped only when the target scored below the source. Under `against: expected` both models can miss the same recorded expectation by the same margin — a zero delta, and a fact about your ground truth rather than a migration defect, already reported as such. Policy budgets are unaffected by the label: `max_tool_argument_drift` counts calls whose *target* score fell below `tool_argument_drift_floor`. ### Cost -Structural and tool-call evaluators are free (pure computation over recorded outputs). Semantic costs one embedding call per output (one per pair when the two outputs are identical); `llm_judge` costs one judge-model call per (pair × criterion) — usually the dominant evaluation cost. Both go through the same cache as everything else, so re-running `evaluate` over an unchanged run costs nothing. +Structural and tool-call evaluators are free (pure computation over recorded outputs) — except that when an `evaluators.semantic` block exists, `tool_arguments` embeds free-text argument values (under the default `auto` strategy) and `semantic`-strategy fields with its model. Semantic costs one embedding call per output (one per pair when the two outputs are identical); `llm_judge` costs one judge-model call per (pair × criterion) — usually the dominant evaluation cost. Both go through the same cache as everything else, so re-running `evaluate` over an unchanged run costs nothing. --- @@ -571,6 +574,7 @@ EvalShift evaluates multi-turn agents by **teacher-forced replay**: each turn is ```jsonl {"id": "conv1_t2", "inputs": {"input": "1pm works"}, "conversation_id": "conv_9f2", "turn_index": 2, + "tools": [{"name": "get_calendar", "description": "List free slots on a day.", "input_schema": {"type": "object", "properties": {"day": {"type": "string"}}, "required": ["day"]}}], "history": [ {"role": "system", "content": "You are a scheduling assistant."}, {"role": "user", "content": "Can we move my appointment?"}, @@ -666,7 +670,7 @@ Budgets (fractions, not percents): | `max_cost_increase` | 0.30 | Relative avg-cost increase, target vs source | | `max_latency_increase` | 0.30 | Relative avg-latency increase | | `fail_on_dropped_params` | `false` | Fail when `state.json` → `dropped_params` is non-empty — an arm could not honour a generation constraint the source capture recorded. Top level only: a model either accepts a parameter or does not, which no subset of examples can vary. | -| `slices` | `{}` | Per-slice overrides (unset fields inherit the top level) | +| `slices` | `{}` | Per-slice overrides keyed by example tag — each distinct tag is a slice (unset fields inherit the top level) | These defaults are a first-migration starting point, not a shipping gate: a fresh suite should *report* the regressions it found rather than fail on a couple of reworded tool arguments. `evalshift init` writes exactly these numbers (`--profile` picks a tighter set — `cost-reduction`, `quantization`, `provider-switch`, `local-model`); tighten them as the suite grows and the migration nears merge. @@ -679,14 +683,14 @@ How the verdict is computed: - **A cost/latency ratio measured only from zeroes says so.** When the calls exist but every `cost_usd` (or `latency_ms`) is `0` on both models, `analyze` adds a recommendations line beside the `conclusive: false`: `The cost increase budget could not be measured: all 4 error-free calls across both models recorded a cost of 0, so its observed 0.00 is a default, not a measurement.` A run with no calls gets no line — the empty `raw.jsonl` already explains itself. Emitted once per run, since every scope reads the same calls. - **Every budget reports its own `denominator`** — the sample `observed` was computed over. Scored records for the regression rate, the equivalence rate and the critical count — counted over *measurements*, so an evaluator scoring two axes contributes two rows per example; `tool_arguments` rows for tool drift; `tool_selection.divergence` rows for tool divergence; the error-free calls behind both averages for the cost and latency ratios. Slices report their own counts. `0` means "counted, and the sample was empty", so `observed` is a default; a *missing* `denominator` means "no sample size reported" and is **not** zero — only bundles written before the field say that, and the hosted gate falls back to `conclusive` for them. It is the same number the `1/n` granularity warning is judged on, and it is orthogonal to `conclusive`: an all-zero cost ratio counted every call it averaged and still measured nothing, so it reports a positive denominator beside `conclusive: false`. The hosted gate derives its own Wilson interval from these denominators, over the same three rate budgets and with the same confidence constant the CLI uses, so a local verdict and a hosted one now agree on whether a breach was confirmed — see [methodology](docs/methodology.md). - Any conclusive budget failure, or any blocking critical/high-severity comparison → **`fail`**. Any lower-severity blocking regression → **`conditional_pass`**. Otherwise → **`pass`**. -- **`migration_policy` is the single source of truth for these budgets.** `migration_decision.json`'s `policy` field is the resolved policy `analyze` computed the verdict under — every top-level budget with its default applied, plus `slices`. `bundle`/`push` do not read `migration_decision.json`; `bundle` re-resolves the verdict from `evalshift.yaml` at bundle time, so the bundle's `decision.policy` (see [Hosted EvalShift](#hosted-evalshift)) is the policy the config held when the bundle was built — editing `migration_policy` between `analyze` and `push` changes what the bundle carries. Either way, that is what lets the hosted gate check a pull request against exactly the budgets the bundle's own verdict used, instead of a separate, web-edited policy. `null` when no `migration_policy` is configured, and on a `migration_decision.json` written before this field existed. `evalshift.yaml` is the source of truth for that policy, and the web app's project policy view is becoming a read-only display of the snapshot each run pushed. +- **`migration_policy` is the single source of truth for these budgets.** `migration_decision.json`'s `policy` field is the resolved policy `analyze` computed the verdict under — every top-level budget with its default applied, plus `slices`. `bundle`/`push` do not read `migration_decision.json`; `bundle` re-resolves the verdict from `evalshift.yaml` at bundle time, so the bundle's `decision.policy` (see [Hosted EvalShift](#hosted-evalshift)) is the policy the config held when the bundle was built — editing `migration_policy` between `analyze` and `bundle` changes what the bundle carries. `push ` builds a bundle only when `run_bundle.json.gz` is missing — an existing bundle is uploaded as-is — so re-run `evalshift bundle ` after a config edit. Either way, that is what lets the hosted gate check a pull request against exactly the budgets the bundle's own verdict used, instead of a separate, web-edited policy. `null` when no `migration_policy` is configured, and on a `migration_decision.json` written before this field existed. `evalshift.yaml` is the source of truth for that policy, and the web app's project policy view is becoming a read-only display of the snapshot each run pushed. - **A slice budget gates the run exactly like an overall one.** The budgets under `migration_policy.slices` are evaluated on the same terms as the top-level ones: a conclusively breached slice budget **fails** the run, and an unconfirmed breach makes it `inconclusive` — the same Wilson rule, counted over that slice's own denominator. Since the overall rows can all be green in a run a slice budget fails, `recommendations` names the one that blocked (`The 'security' slice breached its overall regression rate budget (the share of scored comparisons where the target did worse): 20% over n=20 vs the 0% limit.`) and the `inconclusive` `reason` scope-qualifies it the same way. Per-slice verdicts under `slices[*].verdict` are unchanged, and a slice that fails on *comparison severity* rather than a budget still only downgrades an overall `pass` to `conditional_pass`. - Semantic drift that stays above `min_similarity` counts as equivalent, not regression. CI wiring (on `analyze` and `compare`): - `--gate critical,high` — exit 1 when any comparison at those severities exists (allowed values: `critical`, `high`, `medium`, `low`). -- `--policy-gate` — exit 1 when the policy verdict is `fail` **or** `conditional_pass`. +- `--policy-gate` — exit 1 when the policy verdict is `fail` **or** `conditional_pass`, or when no `migration_policy` is configured. `inconclusive` exits 0. - When `$GITHUB_STEP_SUMMARY` is set, `analyze` appends a markdown results table to the job summary. --- @@ -709,7 +713,7 @@ defaults: ``` - **Cost**: one model call per run (a second only when the first generation is rejected). The narrative is cached in `insights.json` and keyed on the run's `config_hash` plus the model id, so re-running `report` or `push` costs nothing; changing either invalidates the cache and regenerates. -- **Skipped** when `--no-insights` is passed, when no API key is set for the chosen model, and when the run directory has no usable `evalshift.yaml`. Every skip is a warning, never an error. +- **Skipped** silently when `--no-insights` is passed, and with a warning when no API key is set for the chosen model or the run directory has no usable `evalshift.yaml`. A skip is never an error. - **Never fatal.** Any failure inside generation is logged and leaves the narrative empty; a run that already has its statistics is not worth failing over a missing paragraph. - **What gets sent to the model**: the pre-rendered figures, plus the **worst 8 regressions'** inputs and both models' outputs (each truncated to 2000 characters). That is the same exposure `llm_judge` already has, but it is real — if your suite carries data you would not send to an LLM judge, run with `--no-insights`. - `insights.json` is a cache envelope (`{"config_hash": …, "insight": {…}}`); only the inner `insight` object is uploaded. Do not hand-edit it — an envelope the CLI does not recognise is treated as a cache miss. @@ -730,19 +734,20 @@ evalshift logout ``` - Credentials live in `~/.evalshift/credentials` (owner-only permissions). Precedence: CLI flags (`--host`/`--token`) > env (`EVALSHIFT_HOST`/`EVALSHIFT_TOKEN`) > credentials file. `--no-browser` prints the approval URL for remote shells. +- Re-running `login` (without `--token`) while the stored token for that host still works reuses it — no new token is minted — and prints `already logged in as `. To switch accounts, run `evalshift logout` first. - **`login` issues a personal token — don't use one in CI.** A personal token belongs to you and stops working when your membership does, which is correct on a workstation and fatal in a pipeline. For CI, mint a **service account key** in the web app (Settings → API tokens → Service accounts), scope it to the permissions the job needs, and pass it as `EVALSHIFT_TOKEN` from an encrypted secret instead of running `login` on the runner. Service accounts are org-owned, so the key survives the person who created it leaving. - Projects are `org/project` slugs — from `--project`, or the `project:` key in config. Missing projects are auto-created when permissions allow (`--no-create-project` disables; project-scoped tokens can't auto-create). -- `evalshift bundle ` builds the upload artefact without uploading; `push --bundle ` uploads a prebuilt one. `run_bundle.json.gz` carries the manifest, per-example rows (inputs, both outputs, per-evaluator scores and the cost/latency deltas), each example's `traces` — one stream per model side with the ordered tool calls, arguments, any final text, and round markers (`model_call` input/output payloads are deliberately excluded, and oversized tool results are shortened rather than dropped) — the aggregate, `analysis`, the policy `decision` (whose `policy` field is the resolved `migration_policy` this run's verdict was computed under, or `null` when none is configured — see [Migration policy and CI gating](#migration-policy-and-ci-gating)), a run-level `economics` rollup (per-role calls, tokens, cost, latency), `methodology_notes`, the [insights](#run-insights) narrative, the evaluator config and the dataset snapshot. **`report.html` is not uploaded** — it is still written to the run directory for local viewing, and the hosted app renders the run from the data instead. Bundle bytes are deterministic: the same run always compresses identically. Pushes are idempotent on run id. On GitHub Actions, git metadata (`GITHUB_SHA`, branch refs) is baked into the bundle so the server can pair PR runs with base-branch baselines. +- `evalshift bundle ` builds the upload artefact without uploading; `push --bundle ` uploads a prebuilt one. `run_bundle.json.gz` carries the manifest, per-example rows (inputs, both outputs, per-evaluator scores and the cost/latency deltas), each example's `traces` — the replay's own tool-call traces, one stream per model side with the ordered tool calls (names, arguments, call ids), round markers, any final text and refusal messages (no `model_call` events and no tool results; a stream over 256 KB keeps its leading events and is flagged `truncated`; imported agent traces from `traces import` stay local and are not uploaded) — the aggregate, `analysis`, the policy `decision` (whose `policy` field is the resolved `migration_policy` this run's verdict was computed under, or `null` when none is configured — see [Migration policy and CI gating](#migration-policy-and-ci-gating)), a run-level `economics` rollup (per-role calls, tokens, cost, latency), `methodology_notes`, the [insights](#run-insights) narrative, the evaluator config and the dataset snapshot. **`report.html` is not uploaded** — it is still written to the run directory for local viewing, and the hosted app renders the run from the data instead. Bundle bytes are deterministic: the same run always compresses identically. Pushes are idempotent on run id. On GitHub Actions, git metadata (`GITHUB_SHA`, branch refs) is baked into the bundle so the server can pair PR runs with base-branch baselines; elsewhere it comes from git, and `bundle` fails with `could not determine a valid 40-character git SHA; run inside git or set GITHUB_SHA` outside a git checkout. `push ` builds a bundle only when `run_bundle.json.gz` is missing and otherwise uploads the existing one as-is. - `push` validates the bundle against the server's own schema **before** it opens a connection, so a stale, hand-edited or foreign bundle fails locally (`✗ bundle failed schema validation: ...`, exit 1) instead of after a full upload. A bundle at or over **50 MB** compressed prints a warning naming the server's **100 MB** hard limit and uploads anyway — the hard limit is configurable server-side, so the CLI quotes it rather than enforcing a stale copy. -- Two more notices, both before `push` reports success. A bundle with no `decision.policy` — no `migration_policy` configured — still uploads and renders like any gated run, but the hosted gate then has nothing of this run's own to check: unless the project still has an old web-app policy for the server to fall back on, it reports `inconclusive` and never blocks the pull request. `push` says so first, before the network is touched — and so before it can tell the two apart, hence the hedge: `! this run carries no migration policy; unless this project still has an old web-app policy, the hosted gate reports inconclusive and never blocks — add migration_policy to evalshift.yaml`. And once the server's initiate response comes back — before the bundle is uploaded — if the project's only policy lives in the web app and `evalshift.yaml` has no `migration_policy` of its own, `push` prints that policy back as a ready-to-paste block: `! this project has a policy configured in the web app; move it into evalshift.yaml:` followed by a `migration_policy:` YAML block holding only the keys the web app actually set. See [docs/hosted.md](docs/hosted.md#bundle-and-push) for the full example. +- Two more notices, both before `push` reports success. A bundle with no `decision.policy` — no `migration_policy` configured — still uploads and renders like any gated run, but the hosted gate then has nothing of this run's own to check: unless the project still has an old web-app policy for the server to fall back on, it reports `inconclusive` and never blocks the pull request (unless the GitHub Action runs with `require-policy: true`, which fails the job for such a run). `push` says so first, before the network is touched — and so before it can tell the two apart, hence the hedge: `! this run carries no migration policy; unless this project still has an old web-app policy, the hosted gate reports inconclusive and never blocks — add migration_policy to evalshift.yaml`. And once the server's initiate response comes back — before the bundle is uploaded — if the project's only policy lives in the web app and `evalshift.yaml` has no `migration_policy` of its own, `push` prints that policy back as a ready-to-paste block: `! this project has a policy configured in the web app; move it into evalshift.yaml:` followed by a `migration_policy:` YAML block holding only the keys the web app actually set. See [docs/hosted.md](docs/hosted.md#bundle-and-push) for the full example. ### What uploads and what stays local The full field-by-field data contract lives in [docs/hosted.md — Privacy model](docs/hosted.md#privacy-model--exactly-what-uploads); this is the summary. The CLI has **no telemetry** — no analytics, no crash reporting. Its only network traffic is (1) your configured model providers, with your own keys, during `run`/`evaluate`/`report`, and (2) the hosted API on `login`, `whoami`, and `push`. -**A push uploads**, inside `run_bundle.json.gz`: the manifest (run id, `org/project` slug, model ids, suite name, git SHA/branch/PR number, the local suite file path string, content hashes, timestamp, CLI version); per-example rows — the example's template `inputs` and `expected` output **verbatim**, both models' **full output text**, tool-call traces (tool names and arguments; imported traces also carry capped tool results, retrieval queries/documents and guardrail verdicts), per-evaluator scores and error strings, per-side cost and latency, tags; aggregate/analysis/decision/economics (numbers, not content); methodology notes; the insights narrative (prose that can quote the regressions it summarizes); the evaluator config with every prompt body replaced by a `content_hash` (prompt names, file paths and variable names do ship, and so does each `llm_judge` `criterion_prompt`); and a dataset snapshot holding only metadata plus an `examples_hash`. Request metadata beside the bundle: the bearer token as an auth header to the configured host only, and the compressed size. +**A push uploads**, inside `run_bundle.json.gz`: the manifest (run id, `org/project` slug, model ids, suite name, git SHA/branch/PR number, the local suite file path string, content hashes, timestamp, CLI version); per-example rows — the example's template `inputs` and `expected` output **verbatim**, both models' **full output text**, the replay's tool-call traces (tool names, arguments and call ids, round markers, final text and refusal messages, capped at 256 KB per side; imported agent traces are not uploaded), per-evaluator scores and error strings, per-side cost and latency, tags; aggregate/analysis/decision/economics (numbers, not content); methodology notes; the insights narrative (prose that can quote the regressions it summarizes); the evaluator config with every prompt body replaced by a `content_hash` (prompt names, file paths and variable names do ship, and so does each `llm_judge` `criterion_prompt`); and a dataset snapshot holding only metadata plus an `examples_hash`. Request metadata beside the bundle: the bearer token as an auth header to the configured host only, and the compressed size. -**Never uploads**: provider API keys, the hosted token (never inside a bundle), prompt bodies and system prompts, suite conversation histories, tool definitions/schemas, `raw.jsonl`, the response cache, `.evalshift/captures/`, `state.json`, `report.json`, `report.html`. +**Never uploads**: provider API keys, the hosted token (never inside a bundle), prompt bodies and system prompts, suite conversation histories, tool definitions/schemas, `raw.jsonl`, imported agent traces (`traces.jsonl`), the response cache, `.evalshift/captures/`, `state.json`, `report.json`, `report.html`. **Still your responsibility**: `inputs`, `expected`, outputs and traces upload verbatim, so whatever customer data or secrets your suite or your models put in them uploads too. Redact at capture time (SDK redaction boundary) and inspect the exact bytes first: `evalshift bundle `, then `gunzip -c .evalshift/runs//run_bundle.json.gz | jq .` — `push --bundle` uploads exactly the file you inspected. @@ -770,13 +775,13 @@ Exit code is 1 and nothing is uploaded. The CLI never decides entitlements itsel Runs on pushes to main create the base-branch baselines PRs diff against, so the workflow cancels superseded runs on PRs only, never on main. -The action runs the pipeline, pushes the candidate run, finds the latest compatible base-branch run, fetches the hosted diff, maintains a single marked PR comment, and sets the `evalshift/regression` commit status. Inputs: `token` (required), `host`, `config` (default `evalshift.yaml`), `suite-name` (a `suites:` key; preferred) **or** `suite` (a path, default `golden.jsonl`) — mutually exclusive, `fail-on` (`policy` (default — hosted migration-policy verdict, falling back to regression gating when unreachable) | `never` | `regression` | `any-slice-regression`), `evalshift-version` (exact CLI version from PyPI), `create-project` (default `true`), `comment` (default `true`). With no baseline yet, the comment notes the push and gating passes. +The action runs the pipeline, pushes the candidate run, finds the latest compatible base-branch run, fetches the hosted diff, maintains a single marked PR comment, and sets the `evalshift/regression` commit status. Inputs: `token` (required), `host`, `config` (default `evalshift.yaml`), `suite-name` (a `suites:` key; preferred) **or** `suite` (a path, default `golden.jsonl`) — mutually exclusive, `fail-on` (`policy` (default — hosted migration-policy verdict, falling back to regression gating when unreachable) | `never` | `regression` | `any-slice-regression`), `evalshift-version` (exact CLI version from PyPI), `python-version` (default `3.12`), `require-policy` (default `false`; under `fail-on: policy`, `true` fails the job when the pushed run carries no migration policy — by default such a run is reported as ungated with a workflow warning and passes), `branch` / `base-branch` (candidate and baseline branch overrides, auto-detected when omitted), `create-project` (default `true`), `comment` (default `true`), `github-token` (token for PR comments and the commit status; defaults to the workflow's `github.token`), `repo-private` (defaults to the GitHub context; used for the private-repo CI entitlement check). Under `fail-on: policy`, `pass`, `conditional_pass` and `inconclusive` pass and `fail` fails. With no baseline yet, the comment notes the push; under `regression` / `any-slice-regression` gating passes, while under the default `policy` mode the job still follows the run's policy verdict. ### Selecting a suite: name, not path The action takes either `suite-name:` (a key under `suites:` in `evalshift.yaml`) or `suite:` (a path). They load the same rows, but only the **name** resolves that suite's own `evaluators:` block — the one `capture sync` writes for a tool-calling suite (`EvalShiftConfig.evaluators_for` maps a `None` name to the top-level block). Select a wired suite by path and it is silently scored with the top-level `evaluators:` instead; if those are `semantic` + `llm_judge` and the suite's rows are tool calls, nothing scores and the run fails at `analyze` with `scores.jsonl is empty`. There is no warning, because a bare path is a legitimate way to run a suite that has no entry under `suites:`. -A suite wired under `suites:` is therefore selected by name — which is what `init --ci` scaffolds (`suite-name: ${{ matrix.suite }}`, the matrix carrying directory names, which are the keys `capture sync` writes). `suite:` is for a one-off file outside the config. The name form needs an `evalshift-version` pin of `0.14.0` or newer — the release that added `--suite-name`. +A suite wired under `suites:` is therefore selected by name — which is what `init --ci` scaffolds (`suite-name: ${{ matrix.suite }}`, the matrix carrying directory names, which are the keys `capture sync` writes). `suite:` is for a one-off file outside the config. The name form needs an `evalshift-version` pin of `0.14.0` or newer. ### Pin drift @@ -865,7 +870,7 @@ EvalShift follows [Semantic Versioning](https://semver.org). From **1.0.0** onwa |---|---| | `evalshift.yaml` | Every documented field, its type, and its meaning. Unknown keys are rejected (`extra="forbid"`), so the schema is a contract in both directions — a typo fails loudly rather than being ignored. | | Command names and flags | Every command in the [Command reference](#command-reference) and its options, including the severities accepted by `--gate` and the verdicts that trip `--policy-gate`. | -| Exit codes | `0` success · `1` failure, or a gate breach under `--gate` / `--policy-gate`. | +| Exit codes | `0` success · `1` failure, or a gate breach under `--gate` / `--policy-gate` · `2` usage error (an unknown option or value, e.g. `init --provider `). | | Documented artifact fields | The `report.json`, `analysis.json`, `scores.jsonl` and `raw.jsonl` keys described in this document. | | The run bundle | Shared with the hosted server and versioned in its own right — see [Hosted EvalShift](#hosted-evalshift). | @@ -892,7 +897,7 @@ EvalShift follows [Semantic Versioning](https://semver.org). From **1.0.0** onwa | `DEEPSEEK_API_KEY` | — | DeepSeek auth | | `EVALSHIFT_NONINTERACTIVE` | unset | Non-empty → skip the cost-confirmation prompt (implied `--yes`); set in scaffolded CI | | `EVALSHIFT_MAX_RUNS` | unset | Override `retention.max_runs_per_suite`; `0`/`none`/`unlimited`/`off` disables count pruning | -| `EVALSHIFT_DIR` | `.evalshift` | Base dir for SDK captures the `capture` commands read | +| `EVALSHIFT_DIR` | `.evalshift` | Base dir for `/captures` (what the `capture` commands read), `/suites` (where promotion writes) and `/toolsets` (toolset sidecars); `run` looks there first when resolving a `toolset_ref`, then beside the suite file. Does not move `.evalshift/runs/` or the response cache | | `EVALSHIFT_HOST` | `https://api.evalshift.dev` | Hosted API base URL | | `EVALSHIFT_TOKEN` | unset | Hosted token (beats the credentials file, loses to `--token`) | | `EVALSHIFT_CREDENTIALS_PATH` | `~/.evalshift/credentials` | Credentials file override | @@ -906,15 +911,15 @@ Keys are consumed by LiteLLM at call time; EvalShift itself never stores or tran ### Will `run` cost me money? -Yes — every `run` calls a real model. Before dispatch you get a worst-case cost estimate (assumes every completion hits the registry `default_max_tokens`, 4096 — actual cost is usually much lower); above $10 it asks for confirmation. The cache makes repeat runs of unchanged calls free. Cheapest iteration loop: small suite first, cache on. +Yes — every `run` calls a real model. Before dispatch you get a worst-case cost estimate (assumes every completion hits the registry `default_max_tokens`, 4096 — actual cost is usually much lower); above $10 it asks for confirmation. The cache makes repeat runs of unchanged tool-less calls free; tool-calling examples are dispatched live (full price) on every run. Cheapest iteration loop: small suite first, cache on. ### A model call failed mid-run -The error is recorded on that call in `raw.jsonl`; the run completes. At evaluate time the affected pair is scored neutral (0.5/0.5) with the error attached, so it can't masquerade as a regression or an improvement. Re-running the same command re-uses cached successes and retries only the failures (errored calls in a *resumed* run are not retried — start a fresh run to retry them). +The error is recorded on that call in `raw.jsonl`; the run completes. At evaluate time the affected pair gets an errored row (a 0.5/0.5 placeholder with the error attached) that is excluded from the statistics, so it can't masquerade as a regression or an improvement. Re-running the same command re-uses cached successes (tool-less examples only — tool-calling examples are all dispatched again) and retries the failures (errored calls in a *resumed* run are not retried — start a fresh run to retry them). ### `--resume` aborts with a config-hash mismatch -Resume requires the config and suite to be byte-identical to the original run — a changed config would corrupt the pairing. Start a fresh run. +Resume requires the config and the suite path to match the original run — a changed config would corrupt the pairing. Start a fresh run. The suite's *contents* are not part of the hash, so an edited suite at the same path resumes without this error; start a fresh run after editing examples. ### Everything comes back severity `none` diff --git a/llms-full.txt b/llms-full.txt index 2cd8bcc..64fd8de 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -81,7 +81,7 @@ as one section under the pipeline block (also on stage failure); errors are neve | File | Stage | Contents | |---|---|---| | state.json | run | status (in_progress/completed/failed), models, config_hash, counters, non_deterministic_models, dropped_params, evaluator_coverage (written by evaluate: attempted vs recorded per axis, the pairs that produced no row, + the axis's blocking flag; absent flag reads as true) | -| raw.jsonl | run | one (prompt, example, role) per line: prompt, output, tokens, cost, latency, tool trace, error. A teacher-forced multi-round example is ONE row: tokens/cost/latency summed, text = last round's answer, trace.round_count = rounds, each call tagged round_index | +| raw.jsonl | run | one (prompt, example, role, sample) per line: prompt, output, tokens, cost, latency, tool trace, error. A teacher-forced multi-round example is ONE row: tokens/cost/latency summed, text = last round's answer, trace.round_count = rounds, each call tagged round_index | | scores.jsonl | evaluate | one EvalRecord per (pair x evaluator): source/target scores in [0,1], delta | | analysis.json | analyze | per-(prompt,evaluator,slice) statistics + severity | | migration_decision.json | analyze | policy verdict (only when migration_policy configured) | @@ -91,11 +91,15 @@ as one section under the pipeline block (also on stage failure); errors are neve | run_bundle.json.gz | bundle/push | optional hosted upload bundle | Cache: SQLite ~/.evalshift/cache.db, key = SHA-256 of canonical JSON -{model, prompt, inputs, temperature, max_tokens[, history]}, TTL 7 days. defaults.cache: false -disables; `evalshift cache clear` wipes. -Resume: `run --resume` continues newest in_progress run; requires config_hash match (config + -suite byte-identical), skips (prompt_id, example_id, role, sample_index) keys already in -raw.jsonl; errored calls count as done and are NOT retried. +{model, prompt, inputs, temperature, max_tokens[, history]} + generation_config, toolset +fingerprint, round index, sample index when set; TTL 7 days. TOOL-LESS examples only: an example +with a non-empty toolset is NOT cached -- every run of an agent suite is live and full price. +Evaluate-stage embedding/judge caches still apply. defaults.cache: false disables; `evalshift +cache clear` wipes. +Resume: `run --resume` continues newest in_progress run; requires config_hash match (canonical +config + suite PATH; suite contents are NOT hashed, so start fresh after editing examples), +skips (prompt_id, example_id, role, sample_index) keys already in raw.jsonl; errored calls +count as done and are NOT retried. Repeated sampling: defaults.samples_per_example (default 1, max 20) sends every (prompt, example) to each model N times (raw.jsonl rows carry sample_index; the cache keys on it, so each sample is a live call). evaluate scores source sample i against target sample i, then folds the samples @@ -121,7 +125,8 @@ evalshift init [-f/--force] [-d/--directory DIR] [--ci] [--wire-agents/--no-wire [--provider gemini|openai|anthropic|deepseek] [--profile PROFILE] Writes ONLY a minimal capture-first evalshift.yaml: passthrough prompt (id: replay, detection: manual, content: "{input}", variables: [input]), advisory - semantic + llm_judge evaluators (blocking: false), empty managed suites: block, migration + semantic (gemini/openai; commented out for anthropic/deepseek) + llm_judge evaluators + (blocking: false), empty managed suites: block, migration policy from --profile. --ci also writes .github/workflows/evalshift.yml. --wire-agents writes EVALSHIFT.md and points AGENTS.md/CLAUDE.md/GEMINI.md/.cursorrules/copilot-instructions at it; creates AGENTS.md when none of those files exist. Idempotent. @@ -129,10 +134,10 @@ evalshift init [-f/--force] [-d/--directory DIR] [--ci] [--wire-agents/--no-wire Without --ci, warns after writing when an existing .github/workflows/*.yml pins an older evalshift-version than this CLI, or none (advisory, exit 0; see CI pin drift). --ci writes the pin itself and does not warn about the file it just wrote. - PROFILE budgets (regression/critical/equivalence/arg-drift/cost/latency): - model-upgrade (default): .03/0/.95/.01/.20/.30 | cost-reduction: .02/0/.97/.01/.05/.30 - local-model: .05/0/.90/.02/.00/.50 | quantization: .02/0/.97/.005/.00/.20 - provider-switch: .03/0/.95/.01/.20/.40 + PROFILE budgets (regression/critical/equivalence/arg-drift/tool-divergence/cost/latency): + model-upgrade (default): .30/1/.75/.20/.20/.30/.30 (= MigrationPolicy defaults) + cost-reduction: .02/0/.97/.01/.02/.05/.30 | local-model: .05/0/.90/.02/.05/.00/.50 + quantization: .02/0/.97/.005/.02/.00/.20 | provider-switch: .03/0/.95/.01/.03/.20/.40 evalshift doctor Env/config check. Exit 1 ONLY when an existing evalshift.yaml fails validation; missing API keys are soft warnings (exit 0). @@ -166,7 +171,8 @@ evalshift evaluate RUN_ID [-c CONFIG] -> scores.jsonl directly above the verdict block. Also on EvaluateResult.harness_check. evalshift analyze RUN_ID [-c CONFIG] [--gate SEVS] [--policy-gate] --gate: comma-separated from {critical,high,medium,low}; any matching comparison -> exit 1. - --policy-gate: exit 1 when verdict is fail OR conditional_pass. + --policy-gate: exit 1 when verdict is fail OR conditional_pass, or when no migration_policy is + configured; inconclusive exits 0. If $GITHUB_STEP_SUMMARY set, appends a markdown results table. evalshift report RUN_ID [-c CONFIG] [--open] [--insights/--no-insights (default on)] -> report.html + report.json (+ insights.json). --no-insights skips the narrative and its @@ -191,7 +197,9 @@ evalshift compare [run flags] [--gate SEVS] [--policy-gate] [--open] [--push] Hosted: evalshift login [--token es_...] [--host URL] [--no-browser] [--timeout SECS=900] Device-code browser flow, or --token (verified via GET /me). Credentials stored at - ~/.evalshift/credentials (owner-only perms). + ~/.evalshift/credentials (owner-only perms). Without --token, a still-valid stored token for + the host is reused (no new token minted; "already logged in as "); a 401/403 falls + through to the browser flow. To switch accounts: `evalshift logout`, then `login`. Issues a PERSONAL token: tied to your membership, dies with it. Correct for a workstation, wrong for CI. For CI mint a SERVICE ACCOUNT KEY in the hosted web app (Settings -> API tokens -> Service accounts): org-owned machine identity, never owner-equivalent (role is @@ -203,12 +211,13 @@ evalshift login [--token es_...] [--host URL] [--no-browser] [--timeout SECS=900 evalshift logout | evalshift whoami [--host] [--token] evalshift bundle RUN_ID [-c] [-s/--suite] [--suite-name] [-o/--output PATH] [--project org/project] Builds run_bundle.json.gz locally, no upload. Payload: manifest, examples (inputs, both - outputs, per-evaluator scores, cost/latency deltas, traces[] — one stream per model side: - ordered tool calls w/ arguments, any final text, round markers; model_call input/output - excluded; oversized tool results shortened not dropped; same trace shown on the hosted - run-detail page), aggregate, analysis, decision (decision.policy = resolved migration_policy - w/ CLI defaults applied, or null when none is configured -- see "Migration policy verdict - algorithm" below), + outputs, per-evaluator scores, cost/latency deltas, traces[] — the replay's own tool-call + trace, one stream per model side: ordered tool calls w/ names, arguments, call ids; round + markers; any final text; refusal messages. NO model_call events, NO tool results; a stream + over 256KB keeps its leading events and is flagged truncated. Imported traces (traces import) + stay LOCAL, never bundled. Same trace shown on the hosted run-detail page), aggregate, + analysis, decision (decision.policy = resolved migration_policy w/ CLI defaults applied, or + null when none is configured -- see "Migration policy verdict algorithm" below), # examples[].passed is False when the pair scored NO rows: all() over an empty list is True, # and "nothing measured" reading as "passed" is the same silence-as-success bug as the # fabricated skip scores. Example.passed is a required non-nullable bool server-side, so @@ -235,7 +244,10 @@ evalshift push [RUN_ID] [--bundle PATH] [--project] [--host] [--token] [--create-project/--no-create-project (default on)] [-c] [-s] [--suite-name] Requires RUN_ID or --bundle. Idempotent on run id. Auto-creates missing projects when permissions allow (project-scoped tokens cannot). GitHub env (GITHUB_SHA, ref vars) is baked - into the bundle for base-branch pairing. + into the bundle for base-branch pairing; elsewhere the SHA comes from git, and bundle fails + ("could not determine a valid 40-character git SHA; run inside git or set GITHUB_SHA") + outside a git checkout. Branch: GITHUB_HEAD_REF > GITHUB_REF_NAME > git > "local". + push RUN_ID builds run_bundle.json.gz only when missing; an existing bundle is uploaded as-is. Credential precedence: flags > EVALSHIFT_HOST/EVALSHIFT_TOKEN env > ~/.evalshift/credentials. Local validation: the bundle is schema-checked against the vendored server export BEFORE any HTTP call, so a stale/hand-edited/foreign --bundle fails in <1s ("bundle failed schema @@ -250,7 +262,8 @@ evalshift push [RUN_ID] [--bundle PATH] [--project] [--host] [--token] migration policy; unless this project still has an old web-app policy, the hosted gate reports inconclusive and never blocks — add migration_policy to evalshift.yaml" -- without one the run still uploads and renders like any gated one, and the PR it belongs to is never - blocked unless the project still carries an old web-app policy the server falls back on; + blocked unless the project still carries an old web-app policy the server falls back on (or + the GitHub Action runs with require-policy: true, which fails the job for such a run); printing this early means it cannot know which case it is in, hence the hedge. (2) once the server's initiate response comes back -- before the bundle is uploaded -- if the project's only policy lives in the web app (response.legacy_project_policy) and the yaml @@ -398,7 +411,8 @@ PUBLIC -- a breaking change here requires a new major version: - evalshift.yaml: every documented field, its type and meaning. extra="forbid", so unknown keys are rejected in both directions. - Command names and their flags, including --gate severities and --policy-gate verdicts. -- Exit codes: 0 success; 1 failure or gate breach. +- Exit codes: 0 success; 1 failure or gate breach; 2 usage error (unknown option/value, e.g. + init --provider ). - Documented fields of report.json, analysis.json, scores.jsonl, raw.jsonl. - The run bundle (shared with the hosted server, versioned in its own right). @@ -428,7 +442,7 @@ major version. Deprecations warn on stderr, never stdout. | DEEPSEEK_API_KEY | — | DeepSeek auth | | EVALSHIFT_NONINTERACTIVE | unset | non-empty -> implied --yes (skip cost prompt); set in scaffolded CI | | EVALSHIFT_MAX_RUNS | unset | overrides retention.max_runs_per_suite; 0/none/unlimited/off disables | -| EVALSHIFT_DIR | .evalshift | base dir for SDK captures read by `capture` commands | +| EVALSHIFT_DIR | .evalshift | base for /captures (capture cmds read), /suites (promotion writes), /toolsets (sidecars; run tries it first for toolset_ref, then the suite's dir). Not runs/ or the cache | | EVALSHIFT_HOST | https://api.evalshift.dev | hosted API base URL | | EVALSHIFT_TOKEN | unset | hosted token (beats credentials file, loses to --token) | | EVALSHIFT_CREDENTIALS_PATH | ~/.evalshift/credentials | credentials file override | @@ -444,7 +458,7 @@ fails to load with "`thresholds` was removed: it was free-form and gated nothing evalshift.yaml; migration_policy is the single source of truth for gating." Nothing replaced it — delete the block, express any gate you meant by it as a migration_policy budget. -version: 1 # required literal +version: 1 # literal 1, default 1 (optional) project: str|null # hosted slug, regex ^[a-z0-9-]+/[a-z0-9-]+$ prompts: # required, >=1, unique ids # Template axis (suites = dataset axis); every prompt x every example. Init's @@ -462,7 +476,7 @@ defaults: judge_model: str = gemini-3.1-flash-lite-preview insights_model: str|null # run-narrative model; falls back to judge_model concurrency: int = 10 (1..64) # applies to run AND evaluate - cache: bool = true # covers completions, embeddings, judge verdicts + cache: bool = true # covers tool-less completions, embeddings, judge verdicts max_cost_usd: float = 50.0 # soft ceiling, reserved for future enforcement max_tokens: int = 4096 # truncated calls are EXCLUDED from stats samples_per_example: int = 1 (1..20) # repeats per (prompt, example) per model; scores averaged per example @@ -550,6 +564,7 @@ evaluators: # every evaluator config also takes blocking numeric_tolerance: float = 0.05 # relative error, linear decay to 0 at tolerance optional_fields_scored: lenient|strict = lenient # field on one side only -> 0.5 | 0.0 applies_to: ["*"] + use_llm_judge_fallback: bool = false # RESERVED: accepted, nothing reads it (no effect) # calls matched greedily by (tool_name, nearest sequence_index); score = mean over calls # against: expected only -- a ground-truth field NEITHER side produced is dropped from # that call's denominator on BOTH sides (stale expectation, not a model defect; scored @@ -570,10 +585,13 @@ evaluators: # every evaluator config also takes blocking check_missing_verification: bool = true verification_tools: [str] dangerous_tools: [str] -slices: - - name: str # "overall" is RESERVED (run-level scope in the bundle); - filter: str # tag literal matched against example.tags - applies_to: ["*"] # rejected as slice name, example tag, and policy slice key +slices: # VALIDATED + recorded in the bundle, NOT APPLIED: + - name: str # name/filter/applies_to rename, filter and scope + filter: str # nothing today (filter is a literal tag). Slices + applies_to: ["*"] # come from example tags instead: one per distinct + # tag, plus "all". "overall" is RESERVED (run-level + # scope in the bundle): rejected as slice name, + # example tag, and policy slice key. suites: # managed block; capture sync rewrites between : # ">>> evalshift suites" markers source: captured|jsonl = captured @@ -589,7 +607,7 @@ suites: # managed block; capture sync rewrites betwe agent_trace: [...]|null # (resolution reads model_fields_set) # Resolved by EvalShiftConfig.evaluators_for(suite_name) -- the single resolution point; # evaluate, report and the hosted bundle all route through it, so scored set == reported set. - # RunState.suite_name carries the name (set by run/all from --suite-name; None for a raw + # RunState.suite_name carries the name (set by run/compare from --suite-name; None for a raw # --suite ). None, or a name with no entry -> the top-level block, unchanged. migration_policy: # optional; fractions not percents max_overall_regression_rate: float = 0.30 (0..1) @@ -608,7 +626,7 @@ migration_policy: # optional; fractions not percents # fresh suite REPORTS regressions instead of failing on a few reworded tool arguments. # Tighten as the suite grows: `init --profile cost-reduction|quantization|provider-switch| # local-model` scaffold tighter numbers. - slices: {slice_name: {same fields, all nullable -> inherit top level}} + slices: {: {same fields, all nullable -> inherit top level}} # keyed by example tag retention: max_runs_per_suite: int = 20 (>=0; 0 disables) run_ttl_days: int>=1|null = null @@ -683,6 +701,8 @@ estimated capture cost uses DeepSeek's API price. "tools": [{"name": str, "description": str, "input_schema": {...}}]} # the alternative to # toolset_ref: inline toolset for a hand-authored suite. # [] is a real "no tools offered" value, not an absence. + # tools: [] (or a toolset_ref naming the empty toolset) + # carries the same incompatibility as expected_no_tools. Loader collects ALL parse/schema errors before raising; duplicate ids rejected. Every example must carry a toolset (toolset_ref XOR tools) -- every model call records the toolset it was @@ -876,8 +896,9 @@ Verdicts: pass | conditional_pass | fail | inconclusive. Written to migration_de - Semantic drift above min_similarity counts as equivalent, not regression. min_equivalence_rate floors the NON-REGRESSION rate: equivalent and improved both count toward it. - cost/latency increase = max(0, (target_avg - source_avg)/source_avg) over non-errored calls. -CI gates: analyze/all --gate critical,high (exit 1 on matching severities; allowed: -critical,high,medium,low); --policy-gate (exit 1 on fail OR conditional_pass). +CI gates: analyze/compare --gate critical,high (exit 1 on matching severities; allowed: +critical,high,medium,low); --policy-gate (exit 1 on fail OR conditional_pass, or when no +migration_policy is configured; inconclusive exits 0). - migration_policy in evalshift.yaml is the single source of truth for these budgets. MigrationDecision.policy carries the resolved policy this verdict was computed under (every top-level budget w/ its default applied, plus slices) into migration_decision.json. @@ -885,7 +906,9 @@ critical,high,medium,low); --policy-gate (exit 1 on fail OR conditional_pass). at bundle time (evaluate_migration_policy(policy=cfg.migration_policy, ...), or inconclusive_decision when unset -- hosted/bundle.py), so the bundle's decision.policy is the config's policy AS OF THE BUNDLE, not a copy of migration_decision.json -- editing - migration_policy between analyze and push changes what push uploads. Either way this is + migration_policy between analyze and bundle changes what push uploads. `push RUN_ID` builds + a bundle ONLY when run_bundle.json.gz is missing -- an existing (stale) bundle is uploaded + as-is, so re-run `bundle RUN_ID` after a config edit. Either way this is what lets the hosted gate check a PR against the exact budgets the bundle's own verdict used, instead of a separate, web-edited policy -- see `evalshift push` above and docs/hosted.md. null on the inconclusive_decision path (no migration_policy configured) and @@ -937,9 +960,9 @@ Model: defaults.insights_model, falling back to defaults.judge_model. (input + both outputs, each truncated to 2000 chars). Same data exposure as llm_judge. - Cost: 1 model call per run (2 on a rejected generation). Cached in insights.json keyed on state.config_hash + model id, so re-running report/push is free; either moving = cache miss. -- Skipped (warning, never an error) on --no-insights, no provider API key for - the chosen model, or no loadable evalshift.yaml. Any generation failure is swallowed and - leaves the narrative absent — a narrative NEVER fails a run. +- Skipped, never an error: silently on --no-insights; with a warning when there is no provider + API key for the chosen model or no loadable evalshift.yaml. Any generation failure is + swallowed and leaves the narrative absent — a narrative NEVER fails a run. - insights.json is a cache envelope {config_hash, insight}; only the inner `insight` reaches a bundle (the server's Insights model is extra="forbid"). Unrecognised envelope = cache miss. - Server caps enforced client-side before upload: <=10 findings, <=2000 chars per summary and @@ -965,24 +988,25 @@ content leaves the machine. PR number, the LOCAL suite file path as a string (can reveal directory/user names), eval_config_hash, dataset_hash, timestamp, cli_version. - examples[] (one per prompt x example): `inputs` VERBATIM; `expected` VERBATIM; both models' - FULL output text; traces (tool names + arguments; imported agent traces also tool results - capped 16KB each, retrieval queries/documents, guardrail verdicts; final text; refusal/ - error text; stream capped 256KB/side); per-evaluator scores + error strings; per-side - cost/latency; tags/slice names; turn_index. + FULL output text; traces = the replay's own tool calls (names, arguments, call ids, + round markers; final text; refusal messages; no tool results; stream capped 256KB/side, + flagged truncated); imported agent traces are NOT uploaded; per-evaluator scores + error + strings; per-side cost/latency; tags/slice names; turn_index. - aggregate/analysis/decision/economics: numbers and verdict labels, not content. - methodology_notes: model ids + statistical-contract sentences. - insights: the machine-written narrative — prose that can quote the regressions it summarizes. - evaluator_config: config version; prompt list METADATA (names, file paths, variable - names — every prompt body replaced by content_hash); defaults (model ids, concurrency, - cache, max_cost_usd, max_tokens); slices; full evaluators block INCLUDING each llm_judge + names — every prompt body replaced by content_hash); the whole defaults block (model ids, + concurrency, cache, max_cost_usd, max_tokens, samples_per_example); slices (the top-level + block, recorded though not applied); full evaluators block INCLUDING each llm_judge criterion_prompt text (keep judge criteria free of secrets). - dataset_snapshot: suite path, size, slice names, examples_hash. No example content. - NEVER uploads: provider API keys; the hosted token (never inside a bundle); prompt bodies / system prompts (manual content -> content_hash; python_string bodies never enter the config); suite conversation histories (`history`, embedded system messages included); tool - definitions/schemas (toolsets); raw.jsonl; the SQLite response cache; .evalshift/captures/; - state.json; report.json; report.html. + definitions/schemas (toolsets); raw.jsonl; imported agent traces (traces.jsonl); the SQLite + response cache; .evalshift/captures/; state.json; report.json; report.html. - Hashes that replace content (dataset_hash, examples_hash, prompts[].content_hash) are SHA-256 over canonical JSON, so hosted diffs/baselines align without the content. - Residual risk: inputs/expected/outputs/traces upload verbatim — customer data or secrets a @@ -999,9 +1023,10 @@ content leaves the machine. plain literals accepted (f-strings, concatenation, .format(), calls, names rejected). Last module-level assignment wins. Workaround for computed prompts: detection: manual. - Delta convention: delta = target_score - source_score; negative = regression. -- Upstream model-call failure or truncation -> pair scored neutral 0.5/0.5 with error attached - (cannot masquerade as regression or improvement); run always completes. Evaluator-own failure - (judge/embedding call broke) -> record stored as errored and EXCLUDED from statistics. +- Upstream model-call failure or truncation, and evaluator-own failure (judge/embedding call + broke), are handled the same: an errored row (0.5/0.5 placeholder, error set) kept in + scores.jsonl and EXCLUDED from slicing, paired tests and policy rates (cannot masquerade as + regression or improvement); run always completes. - Every example is validated against every prompt (template variables covered) BEFORE any model call is dispatched or money spent. - Cost estimate is worst-case (every completion at the registry default_max_tokens, 4096); @@ -1084,8 +1109,10 @@ content leaves the machine. content (inputs + history), NOT the envelope input_hash (the SDK salts that with conversation_id and derives it from the agent's bound args). Seeded from already-promoted cases, so dedup spans sync runs. -- Slices: filter is a tag literal matched against example.tags; each evaluator is analyzed - overall AND per slice; slice policy budgets inherit unset fields from the top level. +- Slices: every distinct example tag is a slice under its own name, plus "all"; each evaluator is + analyzed overall AND per slice; per-slice budgets go under migration_policy.slices keyed by the + tag and inherit unset fields from the top level. The top-level slices: block is validated and + recorded in the bundle but NOT applied: name/filter/applies_to have no effect today. - Slice dedup: slices holding identical (prompt, evaluator, example) triples collapse to one before any test runs. Duplicates restate the same finding AND skew BH-FDR anti-conservatively: k extra copies of a p-value raise both n and the rank the copies reach, and (n+k)/(r+k) < n/r, @@ -1119,7 +1146,7 @@ content leaves the machine. - Report is a single self-contained HTML file: no external assets, works offline. - Run pruning never touches an in-progress run or the run just finished. - Exit codes: 0 success; 1 handled errors and tripped CI gates; doctor exits 1 only on invalid - existing config; init exits 2 on unknown --provider. + existing config; init exits 2 on unknown --provider; Typer usage errors exit 2. ## Minimal examples @@ -1132,7 +1159,7 @@ prompts: variables: [input] defaults: source_model: gemini-3.1-flash-lite-preview - target_model: gemini-3.1-pro-preview + # target_model: gemini-3.1-pro-preview # init writes it commented out; set it or pass --to concurrency: 4 evaluators: semantic: {embedding_model: gemini/gemini-embedding-001, min_similarity: 0.9, blocking: false} @@ -1163,8 +1190,7 @@ prompts: path: prompts.py variable: AGENT_SYSTEM_PROMPT variables: [query] -slices: - - {name: security, filter: security} +# No slices: block needed -- the "security" tag on the rows below is already a slice. # >>> evalshift suites (managed by `evalshift capture sync`) >>> suites: briefing: # tool-free rows -> no evaluators: block @@ -1185,7 +1211,7 @@ suites: # golden.jsonl rows {"id": "ex_security_01", "inputs": {"query": "User account_42 had 5 failed login attempts in the last hour"}, "tags": ["security"], "expected_tools": [{"tool_name": "notify_security_team", "match_strategy": "subset"}], "toolset_ref": "sha256:1a2b3c..."} {"id": "ex_text_only_01", "inputs": {"query": "What is your refund policy?"}, "tags": ["text_only"], "expected_no_tools": true, "tools": []} -{"id": "conv1_t2", "inputs": {"input": "1pm works"}, "conversation_id": "conv_9f2", "turn_index": 2, "history": [{"role": "system", "content": "You are a scheduling assistant."}, {"role": "user", "content": "Can we move my appointment?"}, {"role": "assistant", "content": "", "tool_calls": [{"id": "c1", "name": "get_calendar", "arguments": {"day": "tue"}}]}, {"role": "tool", "tool_call_id": "c1", "content": "{\"slots\": [\"1pm\"]}"}, {"role": "assistant", "content": "Sure — what time works?"}]} +{"id": "conv1_t2", "inputs": {"input": "1pm works"}, "conversation_id": "conv_9f2", "turn_index": 2, "tools": [{"name": "get_calendar", "description": "List free slots on a day.", "input_schema": {"type": "object", "properties": {"day": {"type": "string"}}, "required": ["day"]}}], "history": [{"role": "system", "content": "You are a scheduling assistant."}, {"role": "user", "content": "Can we move my appointment?"}, {"role": "assistant", "content": "", "tool_calls": [{"id": "c1", "name": "get_calendar", "arguments": {"day": "tue"}}]}, {"role": "tool", "tool_call_id": "c1", "content": "{\"slots\": [\"1pm\"]}"}, {"role": "assistant", "content": "Sure — what time works?"}]} # Capture-first flow (real project) evalshift init --provider gemini # minimal config only @@ -1247,8 +1273,14 @@ with: {token: "${{ secrets.EVALSHIFT_TOKEN }}", config: evalshift.yaml, # Action inputs: token (required), host, config=evalshift.yaml, suite-name (a `suites:` key) # OR suite=golden.jsonl (a path) -- mutually exclusive, # fail-on=policy(default: hosted migration-policy verdict, falls back to regression gating when -# unreachable)|never|regression|any-slice-regression, evalshift-version, create-project=true, -# comment=true. +# unreachable)|never|regression|any-slice-regression, evalshift-version, python-version=3.12, +# require-policy=false (fail-on: policy only; true fails the job when the pushed run carries no +# migration policy, false reports it ungated with a warning and passes), branch, base-branch +# (overrides; auto-detected), create-project=true, comment=true, github-token (default: the +# workflow's github.token; PR comment + commit status), repo-private (default: GitHub context; +# private-repo CI entitlement check). +# policy mode: fail fails; pass, conditional_pass, inconclusive pass. No baseline yet: gating +# passes under regression/any-slice-regression; policy mode still follows the run's verdict. # Behavior: pushes candidate run, finds latest compatible base-branch run, fetches hosted diff, # maintains one marked PR comment, sets `evalshift/regression` commit status. # CI pin drift: the CLI that READS evalshift.yaml in CI must be >= the CLI that WROTE it @@ -1266,7 +1298,8 @@ with: {token: "${{ secrets.EVALSHIFT_TOKEN }}", config: evalshift.yaml, ## Troubleshooting checklist Run seems expensive -> estimate is worst-case at max_tokens; actual usually far lower; cache -makes repeats free; iterate with a small suite. +makes repeats of tool-less examples free (tool-calling examples are always live); iterate with a +small suite. All severities "none" -> usually genuinely no significant difference; check n (<5 insufficient, <20 uncertain) and remember BH correction raises the bar; zero-variance comparisons skip. Verdict "inconclusive" -> (1) all evaluators advisory (fresh init: flip blocking: true as the @@ -1276,11 +1309,13 @@ sample size; more pairs from the same setup are more excluded rows. "broken eval harness" row -> the SOURCE model failed ground truth captured from itself. The suite does not describe the model under test: re-capture it against the agent actually running, or set conformance: off. Any verdict printed beside it is arithmetic over the wrong suite. -analyze/all print the specific reason + recommended fix under the verdict line (also in +analyze/compare print the specific reason + recommended fix under the verdict line (also in migration_decision.json as reason/recommendations). ---resume aborts -> config/suite changed since the run started (config_hash mismatch); start fresh. -Failed calls -> recorded in raw.jsonl with error, pair scored neutral 0.5/0.5; fresh run -retries them (cache serves the successes). +--resume aborts -> config or suite path changed since the run started (config_hash mismatch); start + fresh. Suite contents are not hashed: an edited suite at the same path resumes silently. +Failed calls -> recorded in raw.jsonl with error, pair gets an errored 0.5/0.5 row excluded from +the statistics; fresh run +retries them (cache serves the tool-less successes; tool-calling examples are all re-dispatched). Config rejected -> extra="forbid": check for typo'd keys; error names the exact path. "`thresholds` was removed" -> the key is gone from evalshift.yaml; delete it. It gated nothing; migration_policy is the single source of truth. Nothing replaced it. From 7ec8df4cb026000a6df1e39ea8d0d552855f0f51 Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 13:03:23 +0200 Subject: [PATCH 4/7] docs: correct user-facing pages and examples Audit findings P1-P28 plus the twins of the reference fixes: imported traces are not uploaded, the bundle's trace streams carry no tool results or model_call events, slices come from tags (the example configs' slices: blocks are annotated as not applied), the tool-less-only cache, resume and policy-gate semantics, action no-baseline and require-policy behaviour, record_model_call's required tools=, suite auto-selection, errored rows, ls -t for the newest run, init --provider, EVALSHIFT_NONINTERACTIVE, the git-SHA requirement, push reusing an existing bundle, and a housekeeping command list. CHANGELOG gains one Unreleased Fixed entry. Co-Authored-By: Claude Opus 5.5 (1M context) --- AGENTS.md | 3 +- CHANGELOG.md | 28 +++++++++++++++ README.md | 4 +-- docs/agents.md | 12 ++++--- docs/configuration.md | 60 ++++++++++++++++++++++---------- docs/conversations.md | 2 +- docs/evaluators.md | 13 ++++--- docs/faq.md | 21 ++++++----- docs/getting-started.md | 17 ++++++--- docs/github-action.md | 7 ++-- docs/hosted.md | 46 +++++++++++++++++------- docs/index.md | 15 +++++++- docs/methodology.md | 4 ++- docs/sdk.md | 6 +++- docs/traces.md | 23 ++++++------ examples/agent/README.md | 2 +- examples/agent/evalshift.yaml | 3 ++ examples/capture-first/README.md | 2 +- examples/simple/README.md | 2 +- examples/simple/evalshift.yaml | 3 ++ 20 files changed, 197 insertions(+), 76 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index cbffed8..8465e4d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -60,7 +60,8 @@ Detailed operating rules — commands, repo map, hard rules — live in uv venv --python 3.11 && source .venv/bin/activate uv pip install -e ".[dev]" pytest # full suite -pre-commit run --all-files # exactly what CI runs +make ci # exactly what CI runs (also the pre-push hook) +pre-commit run --all-files # commit-stage hooks only (file hygiene, ruff --fix, format, mypy) ``` - Config models use `extra="forbid"`; any new `evalshift.yaml` field needs docs diff --git a/CHANGELOG.md b/CHANGELOG.md index 4fe68c5..82bd8e6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -83,6 +83,34 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 whole source tree. The floor sits below the measured 94% on purpose, so a real regression fails CI while an honest refactor does not. +- An audit of the docs against the code found them wrong in a few dozen + places, and all of them now say what the CLI does. Nothing about CLI + behaviour changed. The one that matters most concerns slices: the docs + said a top-level `slices:` block picks examples by tag, names the slice, + and scopes it to prompts. It does none of that. The block is validated + (the reserved `overall` name is still rejected) and recorded in the run + bundle, but analysis never reads it. Slices come from example `tags`, one + per distinct tag plus `all`, and per-slice budgets are keyed by tag under + `migration_policy.slices`. The docs now say the block is not applied, and + so does the `SliceConfig` docstring, which described `filter` as a Python + expression. Imported agent traces (`traces import`) stay local; the bundle + carries only the replay's own tool-call trace, without tool results or + `model_call` events. The response cache serves only tool-less examples, + so every `run` of an agent suite is live and full price. `--resume` + hashes the suite's path, not its contents. `push ` uploads an + existing bundle as-is instead of rebuilding it, and `bundle` needs a git + SHA. `--policy-gate` also fails when no `migration_policy` is configured. + The `init` profile table had the wrong `model-upgrade` numbers and no + tool-divergence column. The multi-turn suite example failed to load + because it had no `tools`. The failure-label list was missing + `TOOL_GROUND_TRUTH_MISS`. Upstream model-call failures and evaluator + failures are handled the same way, as errored rows excluded from the + statistics. The GitHub Action docs gained `require-policy` and the other + missing inputs. `record_model_call` examples now pass the required + `tools=`. DOCS.md's header said version 1.0.1; a new check in + `tests/unit/test_docs_currency.py` keeps the version in DOCS.md and + llms-full.txt equal to the package's. + ## [1.1.0] - 2026-09-19 Shipped as a minor deliberately. The `thresholds` removal below is breaking by diff --git a/README.md b/README.md index 85426c3..0379201 100644 --- a/README.md +++ b/README.md @@ -226,8 +226,8 @@ The short version: statistics, the analysis, the migration decision, economics, and the machine-written insights narrative. * **Never uploads**: provider API keys, prompt bodies and system prompts, - suite conversation histories, tool definitions/schemas, `raw.jsonl`, the - response cache, captures, and `report.html`. Prompt and dataset content is + suite conversation histories, tool definitions/schemas, `raw.jsonl`, + imported agent traces, the response cache, captures, and `report.html`. Prompt and dataset content is replaced by SHA-256 hashes so diffs still align across runs. * **Can still be sensitive**: inputs, expected outputs, model outputs, and traces carry whatever content your suite or your models put in them. Redact diff --git a/docs/agents.md b/docs/agents.md index efb09ee..f589398 100644 --- a/docs/agents.md +++ b/docs/agents.md @@ -46,7 +46,7 @@ row carrying its own `toolset_ref`: cd examples/agent export GOOGLE_API_KEY= evalshift run --yes --from gemini-2.5-flash --to gemini-3.1-flash-lite-preview -RUN_ID=$(ls .evalshift/runs/ | head -1) +RUN_ID=$(ls -t .evalshift/runs/ | head -1) evalshift evaluate "$RUN_ID" evalshift analyze "$RUN_ID" evalshift report "$RUN_ID" --open @@ -136,7 +136,7 @@ Nothing in `evalshift.yaml` wires a toolset to a prompt — dispatch reads it off each golden-suite *example* instead (`toolset_ref` or inline `tools`, see [Suite ground truth](#suite-ground-truth) below), so the same prompt can legitimately dispatch some examples with tools and others without, in one -run. `tools.yaml` above is just this project's human-readable record of what +run. [`examples/agent/tools.yaml`](https://github.com/babaliauskas/evalshift-cli/blob/main/examples/agent/tools.yaml) is just this project's human-readable record of what those tools are; it accepts either Anthropic-shape (`name` / `description` / `input_schema`) or OpenAI-shape (`{ "type": "function", "function": {...} }`) entries — `evalshift run` serialises whatever a toolset resolves to in @@ -507,9 +507,11 @@ expectation are ignored here. Which keys are compared follows each expectation's `match_strategy`: `exact` also flags arguments the ground truth did not record, while `subset` (what `capture promote` writes) scores the recorded keys only. -Check `evalshift doctor` first. If it warns about **tool argument shape**, your -recorded arguments use keys the declared schema does not have, and every -ground-truth comparison will score 0 for a reason that is not the model's fault. +If every ground-truth comparison scores 0, check whether your recorded +arguments use keys the declared schema does not have — a capture of a decorated +function's parameters rather than the model's own arguments scores 0 for a +reason that is not the model's fault. See +[Wrapper arguments are unwrapped](#wrapper-arguments-are-unwrapped-legacy-captures). A ground-truth field that **neither** model produced is dropped from that call's denominator on both sides and disclosed as `unmeasured_fields` in the record's diff --git a/docs/configuration.md b/docs/configuration.md index a9a089f..cacb7ef 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -4,20 +4,21 @@ Every EvalShift run is driven by a single `evalshift.yaml` file. This page documents every field — types, defaults, and what they do. `evalshift init` writes a minimal, capture-first `evalshift.yaml` — -a passthrough `replay` prompt, default models, evaluators, and an empty +a passthrough `replay` prompt, default models for the provider picked with +`--provider gemini|openai|anthropic|deepseek` (default `gemini`), evaluators, and an empty managed `suites:` block you fill in with `evalshift capture sync`. Below is the canonical reference. ## Top-level shape ```yaml -version: 1 # required, must be 1 +version: 1 # optional (default 1); must be 1 when set project: org/project # optional, required for hosted push unless passed by flag migration_policy: {...} # optional local migration verdict policy prompts: [...] # required, at least one defaults: {...} # optional evaluators: {...} # optional (but at least one is needed for `evaluate`) -slices: [...] # optional +slices: [...] # optional; validated but not applied (see below) suites: {...} # optional named suites for `run --suite-name`, each with # its own optional `evaluators:` block retention: {...} # optional run-history pruning policy @@ -197,7 +198,7 @@ output (cosine ~0.98) that stays within `min_similarity` is treated as *equivalent*, not a regression — so the policy gate and the report agree. Per-slice overrides live under `migration_policy.slices` — a map of -slice name to a partial policy block; unset fields inherit the top +slice name (an example tag: every distinct tag is a slice) to a partial policy block; unset fields inherit the top level. A slice budget gates the run exactly like a top-level one: a conclusively breached slice budget **fails** the run, an unconfirmed breach makes it `inconclusive`, and `recommendations` names which slice @@ -258,8 +259,14 @@ Two behaviours to know: Verdicts are `pass`, `conditional_pass`, `fail`, or `inconclusive`. When configured, `analyze` writes `migration_decision.json` next to `analysis.json`; `report` renders it as the top-level migration verdict. -Use `--policy-gate` on `analyze` or `compare` to fail CI for `fail` and -`conditional_pass`. +CI gating, on `analyze` and `compare`: + +- `--policy-gate` exits 1 when the verdict is `fail` or `conditional_pass`, + and also when no `migration_policy` is configured. `inconclusive` exits 0. +- `--gate critical,high` exits 1 when any comparison has one of the listed + severities (allowed: `critical`, `high`, `medium`, `low`). +- When `$GITHUB_STEP_SUMMARY` is set, `analyze` appends a markdown results + table to the GitHub Actions job summary. `evalshift.yaml`'s `migration_policy` is the single source of truth for these budgets. `analyze` stamps the resolved policy it computed the verdict under — @@ -267,8 +274,10 @@ every top-level budget with its default applied, plus `slices` — onto `migration_decision.json` as `policy`. `bundle` and `push` do not read that file: `bundle` re-resolves the verdict from `evalshift.yaml` at bundle time, so the bundle's `decision.policy` is the policy the config held *when the -bundle was built* — edit `migration_policy` between `analyze` and `push` and -the bundle carries the new numbers, not the ones `analyze` last wrote. Either +bundle was built* — edit `migration_policy` between `analyze` and `bundle` and +the bundle carries the new numbers, not the ones `analyze` last wrote. `push +` builds a bundle only when none exists, so after an edit re-run +`evalshift bundle ` before pushing. Either way, the hosted gate checks a pull request against exactly the budgets the bundle's own verdict used. `evalshift.yaml` is the source of truth for that policy; the web app's project policy view is becoming a read-only display of @@ -330,7 +339,7 @@ A list of prompt definitions. Each entry has: | `judge_model` | string | `gemini-3.1-flash-lite-preview` | Default LLM-as-judge model. | | `insights_model`| string | (none) | Model that writes the run-insights narrative rendered in `report.html` and uploaded with the bundle. Falls back to `judge_model` when unset — writing analytical prose is a harder task than a pairwise A/B verdict, so it is worth tuning separately. See [Run insights](#run-insights). | | `concurrency` | int | 10 (1 ≤ x ≤ 64) | Max in-flight LLM calls during `evalshift run` **and** `evalshift evaluate` (the embedding and judge calls made while scoring). | -| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions plus `semantic` embeddings and `llm_judge` verdicts. | +| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions of tool-less examples (examples that offer tools are always dispatched live) plus `semantic` embeddings and `llm_judge` verdicts. | | `max_cost_usd` | float | 50.0 | Soft ceiling reserved for future enforcement. The pre-flight cost prompt currently triggers above $10 (skip with `--yes`). | | `max_tokens` | int | 4096 (`> 0`) | Completion length cap sent to every model call. Raise it if outputs are being truncated (the provider returns `finish_reason == "length"`); a `prompts[].max_tokens` entry overrides it per prompt. Truncated calls are detected, surfaced in the report, and **excluded from the regression statistics** so a cut-off output can't manufacture a false regression. | | `samples_per_example` | int | 1 (1 ≤ x ≤ 20) | How many times each `(prompt, example)` is sent to **each** model. Above 1, every sample is its own live call (the cache keys on the sample index), sample *i* of the source is scored against sample *i* of the target, and the example's row in `scores.jsonl` becomes the **mean over samples** with the per-sample scores and the within-example `delta_variance` under `metadata.samples`. The paired tests still run over examples, not samples, so this reduces noise without inflating `n`. Cost and the call count multiply by it; only worth turning on for a model that samples non-deterministically (see the report banner). See [Methodology](methodology.md#limitations-to-be-aware-of). | @@ -523,7 +532,7 @@ values still scores by its strategy, so a wrong value is still 0.0. Under `against: expected`, a ground-truth field that **neither** model produced is dropped from that call's denominator on both sides and disclosed as -`unmeasured_fields` in the record's per-call metadata (`scores.json`). It is a +`unmeasured_fields` in the record's per-call metadata (`scores.jsonl`). It is a stale expectation, not a model defect: scored, it would cap the call below 1.0 for good, since no model change could ever lift it. A field only one side omitted is unaffected — that is `optional_fields_scored`' business. A call whose @@ -605,8 +614,15 @@ writes `false` — see [`blocking`](#blocking-every-evaluator) for why. ## `slices` -A list of named subsets used for slice-level statistical analysis. -The implicit `"all"` slice always exists. +Slices come from the suite, not from this block: every distinct example +`tag` becomes a slice under its own name, alongside the implicit `"all"` +slice, and each is analysed separately. Per-slice budgets go under +[`migration_policy.slices`](#migration_policy), keyed by the tag. + +The top-level `slices:` list is still accepted — it is validated and copied +into the run bundle's evaluator config — but analysis does not read it today: +`name`, `filter` and `applies_to` rename, filter and scope nothing, and a run +reports the same slices with or without it. `overall` is reserved and cannot be used as a slice `name`, as an example tag, or as a `migration_policy.slices` key. It names the run-level scope in the run @@ -616,9 +632,9 @@ suite loads. | Field | Type | Required | Description | | ------------- | ------ | -------- | ----------- | -| `name` | string | yes | Slice name surfaced in reports. | -| `filter` | string | yes | A tag string. Currently the filter is a literal tag — examples whose `tags` list contains the value land in this slice. | -| `applies_to` | list | optional | Glob list of prompt ids this slice applies to (default `["*"]`). | +| `name` | string | yes | Slice name. Not applied: reports name each slice after its tag. | +| `filter` | string | yes | A literal tag. Not applied: every tag is already its own slice. | +| `applies_to` | list | optional | Glob list of prompt ids (default `["*"]`). Not applied. | Slices with identical membership are collapsed to one before analysis, so duplicate tags cannot inflate the Benjamini–Hochberg correction. `all` and any @@ -673,7 +689,7 @@ walkthrough — all optional, additive, single-turn suites parse unchanged): | Field | Type | Required | Description | | ----------------- | ------------------------------------- | -------- | ----------- | -| `history` | list of `{role, content}` or `null` | optional | Conversation prefix replayed verbatim before the current turn (teacher-forced). `role` is `system`, `user`, or `assistant`; at most one `system` message, and it must come first if present. `null` (the default) means single-turn — no message-mode dispatch. | +| `history` | list of `{role, content, tool_calls?, tool_call_id?}` or `null` | optional | Conversation prefix replayed verbatim before the current turn (teacher-forced). `role` is `system`, `user`, `assistant`, or `tool`; `tool_calls` only on `assistant`; `tool_call_id` required on `tool` and forbidden elsewhere; at most one `system` message, and it must come first if present. `null` (the default) means single-turn — no message-mode dispatch. | | `conversation_id` | string or `null` | optional | Id of the recorded conversation this turn came from. Provenance only. | | `turn_index` | integer (`>= 0`) or `null` | optional | Zero-based position of this turn within its conversation. Shown as a `turn N` badge in the HTML report. | | `generation_config` | object or `null` | optional | Generation settings recorded by the SDK on the capture's first model call (`temperature`, `response_mime_type`, `response_schema`, `tool_choice`, `parallel_tool_calls`, `tool_config`, ...). Written by `capture promote`/`sync`; the runner translates it at dispatch — `temperature` overrides the model default, `response_mime_type: application/json` (plus an optional `response_schema`) becomes a LiteLLM `response_format`, and `tool_choice` / `tool_config` / `parallel_tool_calls` become an OpenAI-style `tool_choice` + `parallel_tool_calls` that LiteLLM maps per provider — all on both the source and target calls. See [Agents → Tool-choice constraints are replayed too](agents.md#tool-choice-constraints-are-replayed-too). Delete the field to disable the override; keys the runner cannot translate are ignored with a warning. | @@ -702,8 +718,11 @@ suites: | `evaluators` | block | optional | Evaluators this suite is scored with, replacing the top-level `evaluators:` family by family. Omitted (the default) scores the suite with the top-level block unchanged. | | `managed` | bool | optional | Whether `capture sync` owns this entry. Default `true`. | -Resolution precedence for `run`: an explicit `--suite ` wins, then -`--suite-name ` (looked up here), then the default `golden.jsonl`. +Resolution precedence for `run` and `compare`: an explicit `--suite ` +wins, then `--suite-name ` (looked up here). With neither flag, a config +that wires exactly one suite uses it; one that wires two or more is an error +that lists a ready-to-run command per suite; `./golden.jsonl` is the fallback +only when `suites:` is empty. ### Per-suite `evaluators` @@ -821,6 +840,11 @@ evalshift run --suite-name support_agent --yes # score a candidate model agains evalshift capture clean # prune already-promoted captures ``` +`capture clean []` deletes promoted captures by default (`--promoted`); +`--all` deletes every capture, promoted or not. It never touches promoted +suites, then offers to sweep toolset sidecars no surviving capture or suite +references. Each deletion asks first; `--yes`/`-y` skips both prompts. + `evalshift capture sync` is the one-shot path: it promotes **every** capture under `.evalshift/captures/` into golden suites at `.evalshift/suites//golden.jsonl` **and** injects the resulting diff --git a/docs/conversations.md b/docs/conversations.md index 297934d..6333bc8 100644 --- a/docs/conversations.md +++ b/docs/conversations.md @@ -34,7 +34,7 @@ with capture.agent_session( turn_index=2, parent_capture_id="cap_prev_turn_id", ): - record_model_call(model_id="claude-opus-4-8", input=messages, output=reply) + record_model_call(model_id="claude-opus-4-8", tools=None, input=messages, output=reply) ``` (`capture.agent_session_async(...)` is the `async with` equivalent, for async diff --git a/docs/evaluators.md b/docs/evaluators.md index d36027a..ee27820 100644 --- a/docs/evaluators.md +++ b/docs/evaluators.md @@ -16,8 +16,10 @@ suite sizes their noise would gate the verdict. See When an evaluator's own measurement breaks (judge call fails, embedding call fails), the record is stored as **errored and excluded from the statistics** — not silently scored neutral. Upstream *model-call* -failures are different: the pair gets a neutral 0.5/0.5 record with the -error attached, so the run always completes. +failures and truncated calls are recorded the same way: an errored row +(a 0.5/0.5 placeholder with `error` set), kept in `scores.jsonl` for +inspection and excluded from slicing, the paired tests and the policy +rates. The run always completes. EvalShift ships four families (plus `agent_trace` for [imported external traces](traces.md)): @@ -219,7 +221,7 @@ Per (prompt, example) pair, each evaluator means: | llm_judge | 1 judge model completion | | tool_selection | $0 (compares parsed traces only) | | tool_trace_structure | $0 (compares parsed traces only) | -| tool_arguments | $0 normally; embedding calls per `semantic`-strategy field if you opt in | +| tool_arguments | $0 without an `evaluators.semantic` block (free text uses `difflib`). With one, embedding calls per free-text or `semantic`-strategy field (cached) | A 100-example suite with 1 prompt and 4 evaluators (2 structural + 1 semantic + 1 judge) is: @@ -228,6 +230,7 @@ A 100-example suite with 1 prompt and 4 evaluators (2 structural + * Evaluate: 200 embedding calls + 100 judge calls LiteLLM's pricing data drives the pre-flight estimate; the local -SQLite cache absorbs identical re-runs, evaluate-stage embedding and -judge calls included. Evaluate dispatches its calls under +SQLite cache absorbs identical re-runs of tool-less examples, evaluate-stage +embedding and judge calls included. Examples that offer tools are dispatched +live on every run; only their evaluate-stage calls are cached. Evaluate dispatches its calls under `defaults.concurrency`, same as the run stage. diff --git a/docs/faq.md b/docs/faq.md index 5a95fc5..758c03e 100644 --- a/docs/faq.md +++ b/docs/faq.md @@ -41,9 +41,10 @@ itself when no API key is configured. Disable it with ## What happens if a single LLM call fails? The orchestrator records the error in `raw.jsonl` (with `error="..."`) -and moves on. The run still completes; failed calls are recorded with -a neutral 0.5/0.5 score in the evaluation phase so the analysis can -account for them rather than silently dropping examples. +and moves on. The run still completes. In the evaluation phase the pair +gets an errored row (a 0.5/0.5 placeholder with `error` set): it stays in +`scores.jsonl` for inspection and is excluded from the statistics, so a +failed call can't masquerade as a regression or an improvement. ## What models does EvalShift support? @@ -92,12 +93,13 @@ DeepSeek API, and its estimated capture cost uses DeepSeek's API price. ## Can I resume a run after Ctrl+C / a crash? Yes. `evalshift run --resume` finds the latest in-progress run for -the project, validates that the config + suite haven't changed since, -and continues from where it left off. Already-completed calls +the project, validates that the config and the suite path haven't changed +since, and continues from where it left off. Already-completed calls (including ones that errored at the LLM layer) are skipped. -A config or suite change between attempts aborts the resume — start -a fresh run instead. +A config change or a different suite path between attempts aborts the +resume — start a fresh run instead. The suite's *contents* are not checked, +so after editing examples, start a fresh run yourself. ## How do I push a run to hosted EvalShift? @@ -135,7 +137,10 @@ you see `≤ $0.17` and the run actually cost $0.03, that's expected. ## How do I lower the cost of a run? * **Set the SQLite cache to be on** (it's the default). A re-run of - the exact same configuration is free. + the exact same configuration makes no run-stage calls for tool-less + examples. Examples that offer tools are not cached: an agent suite's + run stage is live, at full price, every time. Evaluate-stage embedding + and judge calls are cached either way. * **Use cheaper models.** The model registry assigns sensible defaults but you can drop everything to flash/mini/haiku tier. * **Skip the LLM judge.** Structural and semantic evaluators are diff --git a/docs/getting-started.md b/docs/getting-started.md index 829e066..aed7d39 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -38,7 +38,8 @@ fastest. EvalShift calls the provider you configure — any provider LiteLLM supports — directly using your own keys. Local runs do not send prompts or outputs to an EvalShift-operated server. -Provider responses are cached locally in `~/.evalshift/cache.db`. +Provider responses to tool-less examples are cached locally in `~/.evalshift/cache.db`; +examples that offer tools are dispatched live on every run. Set whichever providers you intend to use: @@ -54,9 +55,12 @@ export DEEPSEEK_API_KEY= ```bash mkdir my-eval cd my-eval -evalshift init +evalshift init --provider gemini # or openai, anthropic, deepseek ``` +`--provider` picks the model ids the scaffold uses. Omit it and `init` asks on +a terminal, or defaults to `gemini` when there is no terminal to ask on. + You can also pick a migration profile: ```bash @@ -67,8 +71,10 @@ The default `model-upgrade` profile scaffolds a `migration_policy` block that powers the verdict in `analyze`, `compare`, and `report`. This writes a single, minimal, capture-first `evalshift.yaml`: a -passthrough `replay` prompt, advisory semantic + LLM-judge evaluators, an -empty managed `suites:` block for `capture sync` to fill, and the migration +passthrough `replay` prompt, an advisory LLM-judge evaluator and — for the +Gemini and OpenAI scaffolds — an advisory semantic evaluator (the Anthropic +and DeepSeek scaffolds write it commented out, since neither provider has an +embedding endpoint), an empty managed `suites:` block for `capture sync` to fill, and the migration policy. `init` refuses to clobber an existing `evalshift.yaml`; pass `--force` to overwrite, or `--directory my-eval/` to scaffold into a different folder. @@ -154,7 +160,8 @@ verdict block. Warnings raised along the way (LiteLLM deprecation notices, insights retries) are held back and printed as one `⚠` section directly under the pipeline block; errors are never deferred. `run`/`compare` estimate worst-case cost up front and prompt for -confirmation above $10 (skip with `--yes`). +confirmation above $10 (skip with `--yes`, or set `EVALSHIFT_NONINTERACTIVE=1` +in CI). If you want to drive each stage by hand (useful when re-running just one stage after fixing config, or in CI where you stage artefacts): diff --git a/docs/github-action.md b/docs/github-action.md index d75366c..7649372 100644 --- a/docs/github-action.md +++ b/docs/github-action.md @@ -105,13 +105,15 @@ On pull requests, the Action: - Sets commit status `evalshift/regression`. If no compatible baseline exists, the comment explains that the run was pushed -but there is no baseline yet. Gating passes in that case. +but there is no baseline yet. Under `regression` / `any-slice-regression`, +gating passes. Under the default `policy` mode, the job still follows the +run's policy verdict. ## `fail-on` modes | Mode | Behavior | | --- | --- | -| `policy` (default) | Ask hosted EvalShift for the migration-policy verdict — the verdict the run itself computed against the `migration_policy` limits in `evalshift.yaml` and carried in its bundle, returned rather than re-scored. `fail` fails; `pass`/`conditional_pass` pass. If the policy check is unreachable, falls back to `regression` gating and says so. | +| `policy` (default) | Ask hosted EvalShift for the migration-policy verdict — the verdict the run itself computed against the `migration_policy` limits in `evalshift.yaml` and carried in its bundle, returned rather than re-scored. `fail` fails; `pass`/`conditional_pass`/`inconclusive` pass. A run pushed with no `migration_policy` is reported as ungated with a workflow warning and passes, unless `require-policy: true`. If the policy check is unreachable, falls back to `regression` gating and says so. | | `never` | Do not fail the workflow for hosted regressions. | | `regression` | Fail when the hosted diff reports one or more regressed examples. | | `any-slice-regression` | Fail when any slice pass rate moves down. | @@ -131,6 +133,7 @@ Common inputs: | `suite-name` | — | Name of a suite wired under `suites:` in `evalshift.yaml`. Preferred — see [Selecting a suite](#selecting-a-suite-name-not-path). Needs a CLI pin of `0.14.0` or newer. | | `suite` | `golden.jsonl` | Suite path, for a file that is not wired into the config (one suite per invocation). Mutually exclusive with `suite-name`. | | `fail-on` | `policy` | Gate mode — see the table above. | +| `require-policy` | `false` | Fail the job when the pushed run carries no migration policy (`policy` mode only). | | `evalshift-version` | action default (may lag) | Exact CLI version installed from PyPI. Always set it: it must be at least as new as the CLI that writes your `evalshift.yaml` (reader ≥ writer). `init --ci` pins it to the scaffolding CLI. | | `create-project` | `true` | Allow project auto-create when permissions allow it. | | `comment` | `true` | Post or update the PR comment on pull requests. | diff --git a/docs/hosted.md b/docs/hosted.md index 48ebf8c..38ec05e 100644 --- a/docs/hosted.md +++ b/docs/hosted.md @@ -19,6 +19,11 @@ evalshift whoami approve the CLI login, then stores the returned API token in `~/.evalshift/credentials` with owner-only file permissions. Use `--no-browser` on remote shells where the browser cannot open automatically. +`--timeout ` (default 900) bounds how long it waits for the approval. + +Re-running `login` while the stored token for that host still works reuses it +instead of minting a new one, and prints `already logged in as `. To +switch accounts, run `evalshift logout` first, then `evalshift login`. You can still paste an existing hosted API token manually: @@ -97,10 +102,12 @@ The output is `.evalshift/runs//run_bundle.json.gz`. It carries: record it), - one row per example: inputs, both models' outputs, per-evaluator scores, and the cost/latency deltas, -- each example's `traces` — one stream per model side, holding the ordered - tool calls with their arguments, any final text, and round markers. - `model_call` input and output payloads are deliberately excluded, and - oversized tool results are shortened rather than dropped, +- each example's `traces` — the replay's own tool-call traces, one stream per + model side: the ordered tool calls with their names, arguments and call ids, + any final text, refusal messages and round markers. No `model_call` events + or tool results are included. A stream over 256 KB keeps its leading events + and is flagged `truncated`. Imported agent traces (`traces import`) stay + local and are not uploaded, - the aggregate, `analysis`, and the policy `decision`, - `economics` — a run-level per-role rollup of calls, tokens, cost and latency, - `methodology_notes` and the evaluator config and dataset snapshot, @@ -124,6 +131,19 @@ Push a local run: evalshift push ``` +`push ` builds `run_bundle.json.gz` only when the run directory has +none; an existing bundle is uploaded as-is. After editing `evalshift.yaml` +(a `migration_policy` budget, say), re-run `evalshift bundle ` before +pushing, or the upload carries the numbers from when the bundle was built. + +`bundle` — and `push ` when it has to build one — records git +metadata and needs a commit to point at: it fails with `could not determine a +valid 40-character git SHA; run inside git or set GITHUB_SHA` outside a git +checkout unless `GITHUB_SHA` is set. The branch comes from `GITHUB_HEAD_REF`, +then `GITHUB_REF_NAME`, then git, falling back to `local`; the pull request +number from `GITHUB_REF` (`refs/pull//…`) or the event payload at +`GITHUB_EVENT_PATH`. + Push a prebuilt bundle: ```bash @@ -159,8 +179,9 @@ Two more notices can appear before `push` reports success. A bundle with no `migration_policy` configured carries no `decision.policy`, so the hosted gate has nothing of this run's own to check: unless the project still has an old web-app policy for the server to fall back on, the gate reports `inconclusive` -and the pull request it belongs to is never blocked — a silence that reads -exactly like a passing gate unless `push` says so. The warning prints before +and the pull request it belongs to is never blocked (unless the GitHub Action +runs with `require-policy: true`, which fails the job for such a run) — a +silence that reads exactly like a passing gate unless `push` says so. The warning prints before the network is touched, so it cannot yet know which of the two this project is; it hedges accordingly, and prints once: @@ -267,11 +288,11 @@ The bundle itself contains: | Block | What is inside | | --- | --- | | `manifest` | Run id, `org/project` slug, source and target model ids, suite name, git commit SHA, branch name, PR number, the **local suite file path** as a string (it can reveal directory or user names), two content hashes, the run timestamp, and the CLI version. | -| `examples[]` — one row per prompt × example | The example's template variables (`inputs`) **verbatim**; its `expected` reference output **verbatim**; both models' **full output text**; tool-call traces (tool names and arguments; for imported agent traces also tool results capped at 16 KB each, retrieval queries and documents, and guardrail verdicts; plus any final text and refusal/error messages, the whole stream capped at 256 KB per side); per-evaluator scores and error strings; per-side cost and latency; tags and slice names. | +| `examples[]` — one row per prompt × example | The example's template variables (`inputs`) **verbatim**; its `expected` reference output **verbatim**; both models' **full output text**; tool-call traces from the replay: each side's tool calls (names, arguments, call ids) with round markers, any final text and refusal messages, capped at 256 KB per side. Imported agent traces (`traces import`) are not uploaded; per-evaluator scores and error strings; per-side cost and latency; tags and slice names. | | `aggregate`, `analysis`, `decision`, `economics` | Pass/fail counts, statistical comparisons, the migration verdict, and per-role token/cost/latency rollups. Numbers and verdict labels, not content. `decision.policy` is the resolved `migration_policy` this run's verdict was computed under — every top-level budget with its default applied, plus `slices` — or `null` when no `migration_policy` is configured. It is what lets the hosted gate check a pull request against the exact budgets the verdict used, instead of a separate policy configured elsewhere. | | `methodology_notes` | The model ids and the statistical-contract sentences shown in every report. | | `insights` | The machine-written run narrative, when one was generated. It is prose *about* your run and can paraphrase or quote the regressions it summarizes. | -| `evaluator_config` | Config version; the prompt list **metadata only** — prompt names, file paths, and variable names, with every prompt body replaced by a `content_hash`; `defaults` (model ids, concurrency, cache flag, cost ceiling, max_tokens); slice definitions; and the full evaluators block — which includes each `llm_judge` entry's `criterion_prompt` text, so keep judge criteria free of secrets. | +| `evaluator_config` | Config version; the prompt list **metadata only** — prompt names, file paths, and variable names, with every prompt body replaced by a `content_hash`; the whole `defaults` block (model ids, concurrency, cache flag, cost ceiling, max_tokens, samples_per_example); slice definitions; and the full evaluators block — which includes each `llm_judge` entry's `criterion_prompt` text, so keep judge criteria free of secrets. | | `dataset_snapshot` | Suite path, example count, slice names, and one `examples_hash`. **No example content.** | ### What never leaves your machine @@ -288,8 +309,8 @@ The bundle itself contains: in the bundle; only the calls a model actually made at run time appear, in the traces. - **Local artefacts**: `raw.jsonl` (the raw provider requests and responses), - the SQLite response cache, `.evalshift/captures/`, `state.json`, - `report.json`, and `report.html`. + imported agent traces (`traces.jsonl`), the SQLite response cache, + `.evalshift/captures/`, `state.json`, `report.json`, and `report.html`. The content hashes that replace this data (`dataset_hash`, `examples_hash`, `prompts[].content_hash`) are SHA-256 digests, so hosted diffs and baselines @@ -318,7 +339,7 @@ report included, works without an account. Run insights are a separate exposure from the hosted upload: generating them sends the worst regressions' inputs and outputs to `defaults.insights_model`, the same way an `llm_judge` criterion sends outputs to its judge. Disable with -`evalshift report --no-insights` (or `all --no-insights`). +`evalshift report --no-insights` (or `compare --no-insights`). Do not put hosted API tokens or provider API keys in config files. Use the credential file locally and repository secrets in CI. @@ -330,7 +351,8 @@ credential file locally and repository secrets in CI. | `missing hosted token` | No flag, env var, or credentials file token is available. | Run `evalshift login --host `, paste a token with `evalshift login --token --host `, or set `EVALSHIFT_TOKEN`. | | `host uses plain http` warning | The host is non-local HTTP. | Use HTTPS for non-local hosts. | | `hosted project is required` | No `project` in config and no `--project` flag. | Add `project: org/project` or pass `--project`. | +| `could not determine a valid 40-character git SHA` | `bundle` (or `push ` building a bundle) ran outside a git checkout, or in one with no commits, and `GITHUB_SHA` is unset. | Run inside a git repository with at least one commit, or set `GITHUB_SHA`. | | `project was not found` | The project does not exist and auto-create is disabled or not allowed. | Ask an owner to create it, use an org-scoped owner token, or enable auto-create. | | `cannot auto-create at ` | The message names the host it talked to and the server's status. Most often the host is not the one you meant: with no `--host` and no `EVALSHIFT_HOST`, an unset credentials file falls back to `https://api.evalshift.dev`, where your org does not exist. | Run `evalshift whoami` and check the host it prints. If it is wrong, `evalshift login --host `. If the host is right and the status is 403, the token lacks org access — see [Project auto-create](#project-auto-create). | | `this run needs a paid plan` | The org's plan does not cover this push, or the subscription has stopped paying. | Open the upgrade URL printed with the message, or wait for the monthly reset and push the same run id again. See [Plan limits](#plan-limits). | -| `this run carries no migration policy` warning | No `migration_policy` is configured in `evalshift.yaml`, so the bundle has no `decision.policy`. | Add `migration_policy` to `evalshift.yaml` (see [Configuration](configuration.md#migration_policy)). Until then the gate has only whatever old web-app policy the project still has; with none, it reports `inconclusive` and never blocks the pull request. | +| `this run carries no migration policy` warning | No `migration_policy` is configured in `evalshift.yaml`, so the bundle has no `decision.policy`. | Add `migration_policy` to `evalshift.yaml` (see [Configuration](configuration.md#migration_policy)). Until then the gate has only whatever old web-app policy the project still has; with none, it reports `inconclusive` and never blocks the pull request unless the GitHub Action sets `require-policy: true`. | diff --git a/docs/index.md b/docs/index.md index 407d59a..ee982df 100644 --- a/docs/index.md +++ b/docs/index.md @@ -68,9 +68,13 @@ Each stage writes its artefact under `.evalshift/runs//`: | ---------- | ---------------- | | `run` | `raw.jsonl` | | `evaluate` | `scores.jsonl` | -| `analyze` | `analysis.json` | +| `analyze` | `analysis.json` (+ `migration_decision.json` with a `migration_policy`) | | `report` | `report.html` + `report.json` (+ `insights.json`) | +Optional steps add their own: `traces import` writes `traces.jsonl`, +`bundle`/`push` write `run_bundle.json.gz`, and `push` keeps a +`push_state.json` only while an upload is in flight. + Hosted commands add optional sharing and CI workflows: ``` @@ -81,6 +85,15 @@ evalshift push # upload a bundle to hosted EvalShift evalshift compare --suite-name --push # run locally, then push ``` +Housekeeping: + +``` +evalshift inspect [--failed] # per-row deltas (--failed: negative or errored only) +evalshift bundle -o out.json.gz # write the bundle somewhere else +evalshift cache clear # wipe the local response cache +evalshift runs clean # prune old runs (--keep, --older-than, --dry-run) +``` + ## Local-first by design For local commands, your prompts and your suite never leave your machine. diff --git a/docs/methodology.md b/docs/methodology.md index 377e959..ef01f1e 100644 --- a/docs/methodology.md +++ b/docs/methodology.md @@ -377,7 +377,9 @@ bundle reported no drift denominator; once it did, the hosted gate — which has always counted drift as a proportion — began confirming breaches the CLI still called outright failures, so one run could read `fail` locally and `inconclusive` hosted off the CLI's own numbers. Both engines now compute the -interval over the same three budgets, from the same two-sided 95% quantile +interval over the same three budgets — regression rate, equivalence rate and +tool-argument drift; the CLI's fourth, `max_tool_divergence`, has no hosted +interval yet — from the same two-sided 95% quantile (`1.959963984540054`, not the rounded `1.96`), and apply the same asymmetric rule, so a local verdict and a hosted one no longer disagree about whether a breach was confirmed. diff --git a/docs/sdk.md b/docs/sdk.md index 367468d..7a1edf8 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -47,7 +47,9 @@ def issue_refund(order_id: str) -> dict: @capture.agent(suite="support_agent", redact=True, tools=[]) def handle_ticket(query: str) -> str: response = client.messages.create(model="claude-sonnet-5", messages=messages) - record_model_call(model_id="claude-sonnet-5", input=messages, output=response.text) + record_model_call( + model_id="claude-sonnet-5", tools=None, input=messages, output=response.text + ) # tools=None inherits the session's toolset ... ``` @@ -63,6 +65,8 @@ value — `None` included — raises `TypeError`. Details: `tools` is required on the same entry points: the toolset the agent was offered, or `[]` if it never calls tools. Omitting it is a `TypeError` too. +`record_model_call` and `capture.model_call` require `tools=` as well; pass +`None` to inherit the session's toolset. **Provider client wrappers (SDK 0.4.0+).** If the agent calls OpenAI, Anthropic or Google GenAI directly, wrap the client once instead of calling diff --git a/docs/traces.md b/docs/traces.md index 3e492fc..a99d437 100644 --- a/docs/traces.md +++ b/docs/traces.md @@ -3,8 +3,9 @@ EvalShift can compare externally recorded agent timelines. The traces come from your own runtime — EvalShift never runs your agent — and are imported into a completed local run, after which `evaluate`, `analyze`, and `report` work as -usual. Import is local, but imported traces are not local-only: they travel to -hosted EvalShift inside the run bundle if you `push` (see [Hosted](#hosted)). +usual. Imported traces stay local: `evaluate` scores them and the local report +shows them, but `push` uploads only the replay's own tool-call trace (see +[Hosted](#hosted)). ## Import @@ -121,13 +122,13 @@ evalshift replay case --model target --trace ## Hosted -`evalshift bundle` / `evalshift push` carry imported traces into the run bundle -as one event stream per model side, so the hosted run-detail page renders the -timeline and not just the text. A new round starts at each `model_call`; events -before the first one stay in round 0. +Imported traces are not uploaded. They stay in +`.evalshift/runs//traces.jsonl`, where `evaluate` (the `agent_trace` +evaluator), the local report and `diff case` / `inspect case` / `replay case` +read them. -Not everything travels verbatim. `model_call` `input` and `output` payloads are -excluded, and oversized content is shortened rather than dropped: a serialized -`tool_result.result` over 16 KB becomes a truncated preview, and a stream over -256 KB keeps its leading events and is flagged `truncated`. The full bundle -contract is in [Hosted EvalShift](hosted.md#bundle-and-push). +`evalshift bundle` / `evalshift push` carry only the replay's own tool-call +trace — one stream per model side with the tool calls (names, arguments, call +ids), round markers, any final text and refusal messages, capped at 256 KB per +side. The full bundle contract is in +[Hosted EvalShift](hosted.md#bundle-and-push). diff --git a/examples/agent/README.md b/examples/agent/README.md index 656fdec..adb459f 100644 --- a/examples/agent/README.md +++ b/examples/agent/README.md @@ -18,7 +18,7 @@ end to end. cd examples/agent export GOOGLE_API_KEY= evalshift run --yes --from gemini-2.5-flash --to gemini-3.1-flash-lite-preview -RUN_ID=$(ls .evalshift/runs/ | head -1) +RUN_ID=$(ls -t .evalshift/runs/ | head -1) evalshift evaluate "$RUN_ID" evalshift analyze "$RUN_ID" evalshift report "$RUN_ID" --open diff --git a/examples/agent/evalshift.yaml b/examples/agent/evalshift.yaml index 94d24b8..888da21 100644 --- a/examples/agent/evalshift.yaml +++ b/examples/agent/evalshift.yaml @@ -36,6 +36,9 @@ evaluators: divergence: set severity_floor: high +# Slices come from each example's `tags` automatically: one per distinct tag, +# plus `all`. This block is validated but not applied by analysis today -- +# the run reports the same slices without it. slices: - name: security filter: security diff --git a/examples/capture-first/README.md b/examples/capture-first/README.md index 4ca6c00..fdc6f1b 100644 --- a/examples/capture-first/README.md +++ b/examples/capture-first/README.md @@ -177,7 +177,7 @@ evalshift compare --suite-name oncall_triage --to gemini-3.1-pro-preview ``` `--suite-name` looks the suite up in the `suites:` block, which is where its -path *and* its evaluator overrides come from. Step by step instead of `all`: +path *and* its evaluator overrides come from. Step by step instead of `compare`: ```bash evalshift run --yes --suite-name oncall_triage --to gemini-3.1-pro-preview diff --git a/examples/simple/README.md b/examples/simple/README.md index 274b2ed..ad236e5 100644 --- a/examples/simple/README.md +++ b/examples/simple/README.md @@ -9,7 +9,7 @@ cd examples/simple export GOOGLE_API_KEY= evalshift run --yes --from gemini-2.5-flash --to gemini-2.5-pro -RUN_ID=$(ls .evalshift/runs/ | head -1) +RUN_ID=$(ls -t .evalshift/runs/ | head -1) evalshift evaluate "$RUN_ID" evalshift analyze "$RUN_ID" evalshift report "$RUN_ID" --open diff --git a/examples/simple/evalshift.yaml b/examples/simple/evalshift.yaml index 1fb07df..9d60b2d 100644 --- a/examples/simple/evalshift.yaml +++ b/examples/simple/evalshift.yaml @@ -34,6 +34,9 @@ evaluators: min_chars: 5 max_chars: 200 +# Slices come from each example's `tags` automatically: one per distinct tag, +# plus `all`. This block is validated but not applied by analysis today -- +# the run reports the same slices without it. slices: - name: formal filter: formal From 7f5159139328bf211528e5d19dfa7e51c2ca3b0d Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 13:16:05 +0200 Subject: [PATCH 5/7] docs: CONTRIBUTING names make ci as the CI mirror; rewrap two lines `pre-commit run --all-files` without --hook-stage runs only the commit-stage hooks, the same false "everything" claim already corrected in AGENTS.md. Also re-wraps README.md's never-uploads bullet and the migration_policy.slices paragraph in docs/configuration.md to the surrounding line width. Co-Authored-By: Claude Opus 5.5 (1M context) --- CONTRIBUTING.md | 3 ++- README.md | 5 +++-- docs/configuration.md | 17 ++++++++++------- 3 files changed, 15 insertions(+), 10 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a56f18e..4312d6c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -35,7 +35,8 @@ pytest -m "not integration" # unit tests only ruff check . # lint ruff format . # auto-format mypy --strict src/evalshift_cli # type-check -pre-commit run --all-files # everything pre-commit runs +pre-commit run --all-files # commit-stage hooks only +make ci # exactly what CI runs (also the pre-push hook) ``` ## Style diff --git a/README.md b/README.md index 0379201..76dfa4f 100644 --- a/README.md +++ b/README.md @@ -227,8 +227,9 @@ The short version: machine-written insights narrative. * **Never uploads**: provider API keys, prompt bodies and system prompts, suite conversation histories, tool definitions/schemas, `raw.jsonl`, - imported agent traces, the response cache, captures, and `report.html`. Prompt and dataset content is - replaced by SHA-256 hashes so diffs still align across runs. + imported agent traces, the response cache, captures, and `report.html`. + Prompt and dataset content is replaced by SHA-256 hashes so diffs still + align across runs. * **Can still be sensitive**: inputs, expected outputs, model outputs, and traces carry whatever content your suite or your models put in them. Redact at capture time (see the SDK's redaction boundary) and inspect before diff --git a/docs/configuration.md b/docs/configuration.md index cacb7ef..ce19b2b 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -198,12 +198,13 @@ output (cosine ~0.98) that stays within `min_similarity` is treated as *equivalent*, not a regression — so the policy gate and the report agree. Per-slice overrides live under `migration_policy.slices` — a map of -slice name (an example tag: every distinct tag is a slice) to a partial policy block; unset fields inherit the top -level. A slice budget gates the run exactly like a top-level one: a -conclusively breached slice budget **fails** the run, an unconfirmed -breach makes it `inconclusive`, and `recommendations` names which slice -budget blocked. A slice that fails on comparison *severity* rather than -a budget still only downgrades an overall `pass` to `conditional_pass`. +slice name (an example tag: every distinct tag is a slice) to a partial +policy block; unset fields inherit the top level. A slice budget gates +the run exactly like a top-level one: a conclusively breached slice +budget **fails** the run, an unconfirmed breach makes it `inconclusive`, +and `recommendations` names which slice budget blocked. A slice that +fails on comparison *severity* rather than a budget still only +downgrades an overall `pass` to `conditional_pass`. ```yaml migration_policy: @@ -622,7 +623,9 @@ slice, and each is analysed separately. Per-slice budgets go under The top-level `slices:` list is still accepted — it is validated and copied into the run bundle's evaluator config — but analysis does not read it today: `name`, `filter` and `applies_to` rename, filter and scope nothing, and a run -reports the same slices with or without it. +reports the same slices with or without it. It is, however, part of the +bundle's `eval_config_hash`, so editing or removing it breaks hosted baseline +compatibility with earlier runs. `overall` is reserved and cannot be used as a slice `name`, as an example tag, or as a `migration_policy.slices` key. It names the run-level scope in the run From 6eed983b7b34bd42dcbf521d070c2ca0e95eeaea Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 13:16:05 +0200 Subject: [PATCH 6/7] docs: note that the unapplied slices: block still feeds eval_config_hash The top-level slices: block has no effect on analysis, but the bundle's eval_config_hash is computed over the evaluator_config snapshot that includes it, and hosted EvalShift only pairs a candidate with a baseline of equal eval_config_hash. Editing or removing the block therefore breaks baseline compatibility with earlier hosted runs; every "not applied" site now says so, and docs/hosted.md marks the recorded slice definitions as not applied. Co-Authored-By: Claude Opus 5.5 (1M context) --- DOCS.md | 4 ++-- docs/hosted.md | 2 +- examples/agent/evalshift.yaml | 4 +++- examples/simple/evalshift.yaml | 4 +++- llms-full.txt | 9 +++++++-- 5 files changed, 16 insertions(+), 7 deletions(-) diff --git a/DOCS.md b/DOCS.md index a66cbf3..d896d95 100644 --- a/DOCS.md +++ b/DOCS.md @@ -314,7 +314,7 @@ suites: {} | `prompts` | list, required, ≥1 | Prompt definitions (unique ids enforced) | | `defaults` | block | Run defaults, below | | `evaluators` | block | Evaluator configs, see [Evaluators](#evaluators) | -| `slices` | list | Validated and recorded in the bundle, but **not applied** — slices come from example tags. See below | +| `slices` | list | Validated and recorded in the bundle (and in its `eval_config_hash`), but **not applied** — slices come from example tags. See below | | `migration_policy` | block \| absent | Regression budgets, see [Migration policy](#migration-policy-and-ci-gating) | | `suites` | map | Named suites (`{name: {source: captured\|jsonl, path: ..., evaluators: ..., managed: true}}`); the block between the `>>> evalshift suites` markers is managed by `capture sync`. See [Per-suite evaluators](#per-suite-evaluators) | | `retention` | block | `max_runs_per_suite` (default 20, `0` disables), `run_ttl_days` (default off) | @@ -352,7 +352,7 @@ slices: # validated and recorded in the run bundle; NOT ap applies_to: ["*"] # glob list of prompt ids ``` -The top-level `slices:` block still loads — it is validated and copied into the run bundle's evaluator config — but analysis does not read it today: `name`, `filter` and `applies_to` rename, filter and scope nothing, and a run reports the same slices with or without it. `overall` is reserved — it names the run-level scope in the run bundle — and is rejected as a slice `name`, as an example tag, and as a `migration_policy.slices` key. +The top-level `slices:` block still loads — it is validated and copied into the run bundle's evaluator config — but analysis does not read it today: `name`, `filter` and `applies_to` rename, filter and scope nothing, and a run reports the same slices with or without it. It is, however, part of the bundle's `eval_config_hash`, so editing or removing it breaks hosted baseline compatibility with earlier runs. `overall` is reserved — it names the run-level scope in the run bundle — and is rejected as a slice `name`, as an example tag, and as a `migration_policy.slices` key. Slices holding exactly the same examples are collapsed to one before any test runs — duplicates restate the same numbers as if they were independent findings and skew the Benjamini–Hochberg correction anti-conservatively (extra copies of a p-value shrink every adjusted p-value in the family, so results look more significant than they are). `all` and any slice named under `migration_policy.slices` always survive; otherwise the provenance tag `captured` (written by `capture promote`) loses to an ordinary tag, then alphabetical order decides. Drops are reported on the terminal and as `collapsed_slices` in `analysis.json`. See [docs/methodology.md](docs/methodology.md). diff --git a/docs/hosted.md b/docs/hosted.md index 38ec05e..f6831be 100644 --- a/docs/hosted.md +++ b/docs/hosted.md @@ -292,7 +292,7 @@ The bundle itself contains: | `aggregate`, `analysis`, `decision`, `economics` | Pass/fail counts, statistical comparisons, the migration verdict, and per-role token/cost/latency rollups. Numbers and verdict labels, not content. `decision.policy` is the resolved `migration_policy` this run's verdict was computed under — every top-level budget with its default applied, plus `slices` — or `null` when no `migration_policy` is configured. It is what lets the hosted gate check a pull request against the exact budgets the verdict used, instead of a separate policy configured elsewhere. | | `methodology_notes` | The model ids and the statistical-contract sentences shown in every report. | | `insights` | The machine-written run narrative, when one was generated. It is prose *about* your run and can paraphrase or quote the regressions it summarizes. | -| `evaluator_config` | Config version; the prompt list **metadata only** — prompt names, file paths, and variable names, with every prompt body replaced by a `content_hash`; the whole `defaults` block (model ids, concurrency, cache flag, cost ceiling, max_tokens, samples_per_example); slice definitions; and the full evaluators block — which includes each `llm_judge` entry's `criterion_prompt` text, so keep judge criteria free of secrets. | +| `evaluator_config` | Config version; the prompt list **metadata only** — prompt names, file paths, and variable names, with every prompt body replaced by a `content_hash`; the whole `defaults` block (model ids, concurrency, cache flag, cost ceiling, max_tokens, samples_per_example); slice definitions (recorded, not applied); and the full evaluators block — which includes each `llm_judge` entry's `criterion_prompt` text, so keep judge criteria free of secrets. | | `dataset_snapshot` | Suite path, example count, slice names, and one `examples_hash`. **No example content.** | ### What never leaves your machine diff --git a/examples/agent/evalshift.yaml b/examples/agent/evalshift.yaml index 888da21..72f454d 100644 --- a/examples/agent/evalshift.yaml +++ b/examples/agent/evalshift.yaml @@ -38,7 +38,9 @@ evaluators: # Slices come from each example's `tags` automatically: one per distinct tag, # plus `all`. This block is validated but not applied by analysis today -- -# the run reports the same slices without it. +# the run reports the same slices without it. It is still part of the hosted +# bundle's eval_config_hash, so editing it breaks baseline compatibility with +# earlier hosted runs. slices: - name: security filter: security diff --git a/examples/simple/evalshift.yaml b/examples/simple/evalshift.yaml index 9d60b2d..ad31554 100644 --- a/examples/simple/evalshift.yaml +++ b/examples/simple/evalshift.yaml @@ -36,7 +36,9 @@ evaluators: # Slices come from each example's `tags` automatically: one per distinct tag, # plus `all`. This block is validated but not applied by analysis today -- -# the run reports the same slices without it. +# the run reports the same slices without it. It is still part of the hosted +# bundle's eval_config_hash, so editing it breaks baseline compatibility with +# earlier hosted runs. slices: - name: formal filter: formal diff --git a/llms-full.txt b/llms-full.txt index 64fd8de..3ac213c 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -591,7 +591,10 @@ slices: # VALIDATED + recorded in the bundle, NOT AP applies_to: ["*"] # come from example tags instead: one per distinct # tag, plus "all". "overall" is RESERVED (run-level # scope in the bundle): rejected as slice name, - # example tag, and policy slice key. + # example tag, and policy slice key. Still part of + # the bundle's eval_config_hash: editing or removing + # it breaks hosted baseline compatibility with + # earlier runs. suites: # managed block; capture sync rewrites between : # ">>> evalshift suites" markers source: captured|jsonl = captured @@ -1112,7 +1115,9 @@ content leaves the machine. - Slices: every distinct example tag is a slice under its own name, plus "all"; each evaluator is analyzed overall AND per slice; per-slice budgets go under migration_policy.slices keyed by the tag and inherit unset fields from the top level. The top-level slices: block is validated and - recorded in the bundle but NOT applied: name/filter/applies_to have no effect today. + recorded in the bundle but NOT applied: name/filter/applies_to have no effect on analysis + today. It IS part of the bundle's eval_config_hash, so editing or removing it breaks hosted + baseline compatibility with earlier runs. - Slice dedup: slices holding identical (prompt, evaluator, example) triples collapse to one before any test runs. Duplicates restate the same finding AND skew BH-FDR anti-conservatively: k extra copies of a p-value raise both n and the rank the copies reach, and (n+k)/(r+k) < n/r, From c58ebd07075a162fd4fffba227483811b0d9d2a4 Mon Sep 17 00:00:00 2001 From: Lukas Babaliauskas Date: Wed, 30 Sep 2026 13:20:11 +0200 Subject: [PATCH 7/7] docs: pin-drift summaries include newer pins utils/ci_pin.py reports three statuses -- stale, unpinned and ahead (every pin newer than the local CLI) -- and capture sync, init, validate and doctor print whichever finding check_ci_pin returns. The Pin drift sections already said so; the per-command summaries still described older or missing pins only. They now cover newer pins too, with the fix the ahead finding prints (pip install -U evalshift), and the configuration.md paragraph keeps its reader >= writer argument while explaining why an ahead pin still warns. Co-Authored-By: Claude Opus 5.5 (1M context) --- DOCS.md | 6 +++--- docs/configuration.md | 5 ++++- docs/getting-started.md | 12 ++++++------ docs/github-action.md | 2 +- llms-full.txt | 10 ++++++---- 5 files changed, 20 insertions(+), 15 deletions(-) diff --git a/DOCS.md b/DOCS.md index d896d95..60baea7 100644 --- a/DOCS.md +++ b/DOCS.md @@ -176,7 +176,7 @@ evalshift capture clean # delete promoted capture files + sweep o 7. Skips captures whose turn recorded an `error` event — a turn that died before the agent acted is not ground truth, and promoting it would assert `expected_no_tools: true` on a question that needed a tool. `--allow-errored` promotes it anyway (still never asserting `expected_no_tools`). `capture promote` exits non-zero on the same condition. Separately and unconditionally — `--allow-errored` does not help — a capture whose first `model_call` has no `toolset_ref` is refused: the SDK did not record what tools were offered, so there is nothing to carry, and re-capturing with a current `evalshift-sdk` is the only fix. 8. Warns when two captures claim the same `(conversation_id, turn_index)` (a retried turn), and when a promoted turn contains a failed tool result (`error`, or `{"success": false}`). Both stay warnings — see [Agent evals → What does not belong in a golden suite](docs/agents.md). 9. Writes `.evalshift/suites//golden.jsonl` and rewrites the managed `suites:` block in `evalshift.yaml` (between the `>>> evalshift suites` markers). -10. After the write (or after printing the block for you to paste), checks the CI pin: if a workflow under `.github/workflows/` uses `babaliauskas/evalshift-action` with an `evalshift-version` older than this CLI, or with no pin at all, it prints a warning naming the workflow and job plus the exact `evalshift-version: ""` line to set. Advisory only — sync never edits a workflow and the exit code is unchanged. See [Pin drift](#pin-drift). +10. After the write (or after printing the block for you to paste), checks the CI pin: if a workflow under `.github/workflows/` uses `babaliauskas/evalshift-action` with an `evalshift-version` older than this CLI, with no pin at all, or with pins that are all newer than this CLI, it prints a warning naming the workflow and job plus the fix — the exact `evalshift-version: ""` line to set, or, for a newer pin, `pip install -U evalshift` locally. Advisory only — sync never edits a workflow and the exit code is unchanged. See [Pin drift](#pin-drift). Strictness knobs for the derived tool expectations: `--strict-args` (exact argument matches), `--names-only` (ignore arguments), `--tool-count` (also pin the call count, scoped the same way as `expected_tools`), `--rounds {first,all}` (which agent rounds become ground truth, default `first`). `--tag` attaches extra slice tags; `--print` previews the `suites:` block without writing. @@ -807,7 +807,7 @@ Common conventions: `-c/--config` defaults to `./evalshift.yaml`; run artefacts **`evalshift init`** — scaffold a minimal capture-first `evalshift.yaml`. `-f/--force` · `-d/--directory ` · `--ci` · `--wire-agents/--no-wire-agents` (default on) · `--provider gemini|openai|anthropic|deepseek` · `--profile model-upgrade|cost-reduction|local-model|quantization|provider-switch` (default `model-upgrade`) -Without `--ci`, warns after writing when an existing workflow under `.github/workflows/` pins an older CLI than this one, or none at all (see [Pin drift](#pin-drift)); `init --ci` writes the pin itself and does not warn about the file it just wrote. +Without `--ci`, warns after writing when an existing workflow under `.github/workflows/` pins an older or newer CLI than this one, or none at all (see [Pin drift](#pin-drift)); `init --ci` writes the pin itself and does not warn about the file it just wrote. **`evalshift doctor`** — environment/config check. Exit 1 only on an invalid existing config. Row 2, `evalshift-sdk`, confirms `import evalshift` is the SDK (`warn` when missing or shadowed, never a failure). Reports the toolset each configured suite carries and flags a suite whose examples carry more than one distinct toolset. The suite-side checks cover every suite in the config's `suites:` block, falling back to `./golden.jsonl` when none are wired. Adds a `ci pin` row when a workflow uses the GitHub Action (`warn` on pin drift, never a failure) and a `judge family` row when an `llm_judge` judge shares a provider with a configured arm (`warn`, never a failure). @@ -855,7 +855,7 @@ All `run` flags, plus `--gate` · `--policy-gate` · `--open` · `--push` · `-- ### Hidden debug commands -**`evalshift validate`** — load config + suite + prompts, cross-check compatibility. `-s/--suite` · `-c/--config`. After the success line, prints the [Pin drift](#pin-drift) warning if a workflow pins an older CLI (advisory; exit code unchanged, and a no-op in CI where the running CLI is the pin). +**`evalshift validate`** — load config + suite + prompts, cross-check compatibility. `-s/--suite` · `-c/--config`. After the success line, prints the [Pin drift](#pin-drift) warning if a workflow pins an older or newer CLI, or none at all (advisory; exit code unchanged, and a no-op in CI where the running CLI is the pin). **`evalshift test-call`** — one live smoke-test call. `-m/--model` (required) · `-p/--prompt` · `-t/--temperature` (0–2, default 0) · `--max-tokens` (1–8192, default 256) · `--tools ` (prints a ToolTrace) --- diff --git a/docs/configuration.md b/docs/configuration.md index ce19b2b..5f69624 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -47,7 +47,10 @@ practice that means the `evalshift-version` your CI workflow installs must be at least the version you run `capture sync` and `init` with locally. The CI pin check is the mechanism that enforces it: `capture sync`, `init`, `doctor`, and `validate` warn when a workflow under `.github/workflows/` pins an older -CLI, or none at all, and print the exact line to set. See +CLI, or none at all, and print the exact line to set. They also warn when +every pin is newer than the local CLI. That does not break the rule — CI +reads with the newer CLI — but local runs then disagree with CI, and the fix +printed is `pip install -U evalshift`. See [Pin drift](github-action.md#pin-drift). ## `project` diff --git a/docs/getting-started.md b/docs/getting-started.md index aed7d39..ed686a8 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -98,8 +98,8 @@ environment is the capture SDK — yellow when it is missing or shadowed by an older CLI install. If a workflow under `.github/workflows/` uses the GitHub Action, the table -also has a `ci pin` row — yellow when CI pins an older CLI than yours (or -none at all); see [Pin drift](github-action.md#pin-drift). +also has a `ci pin` row — yellow when CI pins an older or newer CLI than +yours (or none at all); see [Pin drift](github-action.md#pin-drift). If everything is green or yellow, you're ready to run. @@ -141,10 +141,10 @@ evalshift capture sync `.evalshift/suites//golden.jsonl` and injects the matching `suites:` block into `evalshift.yaml`. See [Configuration](configuration.md) for the full capture lifecycle. If a -workflow under `.github/workflows/` pins an older CLI than the one you just -synced with, it ends with an advisory warning and the exact -`evalshift-version` line to set — see -[Pin drift](github-action.md#pin-drift). +workflow under `.github/workflows/` pins an older or newer CLI than the one +you just synced with, or none at all, it ends with an advisory warning and the +fix — the exact `evalshift-version` line to set, or `pip install -U evalshift` +when CI is ahead — see [Pin drift](github-action.md#pin-drift). ## 7. Run the pipeline diff --git a/docs/github-action.md b/docs/github-action.md index 7649372..70949cc 100644 --- a/docs/github-action.md +++ b/docs/github-action.md @@ -29,7 +29,7 @@ checklist) with three jobs: `evalshift.yaml` and pushed inside the bundle. `evalshift-version` is pinned to the CLI that scaffolded the project: the CLI that *reads* the config in CI must be at least as new as the CLI that *wrote* it locally (`extra: forbid` rejects newer keys), and - the CLI warns when the pin falls behind — see [Pin drift](#pin-drift). `max-parallel` defaults to 1 — + the CLI warns when the pin falls behind (or runs ahead of the local CLI) — see [Pin drift](#pin-drift). `max-parallel` defaults to 1 — raise it toward your hosted plan's in-flight-run ceiling (Free 1, Pro 5, Team 10). The PR comment is posted by the first matrix job only: the comment marker is a constant, so multiple suites would overwrite one diff --git a/llms-full.txt b/llms-full.txt index 3ac213c..4c6bfae 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -131,8 +131,8 @@ evalshift init [-f/--force] [-d/--directory DIR] [--ci] [--wire-agents/--no-wire EVALSHIFT.md and points AGENTS.md/CLAUDE.md/GEMINI.md/.cursorrules/copilot-instructions at it; creates AGENTS.md when none of those files exist. Idempotent. --provider prompted on TTY, else gemini. Does NOT write prompts.py/tools.yaml/golden.jsonl. - Without --ci, warns after writing when an existing .github/workflows/*.yml pins an older - evalshift-version than this CLI, or none (advisory, exit 0; see CI pin drift). --ci writes + Without --ci, warns after writing when an existing .github/workflows/*.yml pins an older or + newer evalshift-version than this CLI, or none (advisory, exit 0; see CI pin drift). --ci writes the pin itself and does not warn about the file it just wrote. PROFILE budgets (regression/critical/equivalence/arg-drift/tool-divergence/cost/latency): model-upgrade (default): .30/1/.75/.20/.20/.30/.30 (= MigrationPolicy defaults) @@ -151,7 +151,8 @@ evalshift doctor ok ("pinned to "; "not pinned (version unknown)" when every pin is a ${{ }} expression) when nothing drifts; warn (exit 0) on pin drift -- stale (pin < local), unpinned (input absent -> action default may lag), ahead (all pins > - local). Fix line names the exact `evalshift-version: ""` to set. See CI pin drift. + local). Fix line names the exact `evalshift-version: ""` to set (ahead: `pip install -U + evalshift` instead). See CI pin drift. Adds a `judge family` row when the config wires llm_judge AND names both defaults.source_model and target_model: warn (exit 0), one row per distinct judge_model whose resolved provider equals an arm's ("judge `` shares a model family () with the @@ -292,7 +293,8 @@ evalshift capture sync [--suite S] [--input-var NAME=input] [--tag T]... [--stri Reports "wired generation config for N case(s)" when promoted captures carried one. After writing (or printing the block to paste) warns on CI pin drift: a workflow step `uses: babaliauskas/evalshift-action@...` whose evalshift-version is older than this CLI or - absent gets a warning naming workflow + job and the `evalshift-version: ""` to set. + absent gets a warning naming workflow + job and the `evalshift-version: ""` to set; pins + all newer than this CLI get one too, with the fix `pip install -U evalshift`. Advisory only: never edits the workflow, exit code unchanged. `init` (without --ci) warns the same way next to an existing workflow; `validate` prints it after its success line. Each suite's entry is DERIVED from its own rows, evaluators included: