diff --git a/CHANGELOG.md b/CHANGELOG.md index 4eda1e5..6fe1dd9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,28 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ("1 schema problem found"), which names neither the offending key nor the fix; `validate` prints both. +- Repeat runs of agent suites are now served from the response cache. Every + example that offers tools used to bypass the cache, a leftover from v0.2 + when a cache entry could not hold a parsed tool trace, so each `run` of an + agent suite paid for every call again, one per replayed round. Each replayed + round is now its own entry. It is keyed on the canonical model, the prompt + and inputs, the exact message list that round sends (history, current turn, + and the recorded rounds and fixture results fed back), the tool list exactly + as sent and in order (including `strict`), `generation_config` (so + `tool_choice` and `parallel_tool_calls`), the effective temperature and + `max_tokens`, the round index and the sample index. A hit restores the + parsed trace, tokens, cost, latency and finish reason, so the `raw.jsonl` + row is identical to the live one apart from `cached`, which is true only + when every round hit, and the new `cached_rounds` count. A row with any + round served from the cache carries latency from an earlier run, so it stays + out of the report's live latency figures and its latency delta is marked not + comparable, in `report.json` and in the bundle alike. Errors are never + cached: the next run re-sends a failed round and serves the rounds before it + from the cache. Truncated responses are cached and stay flagged, and + `defaults.cache: false` still sends everything live. Existing + `~/.evalshift/cache.db` files keep working: the new `trace_json` column is + added in place on open, and every cached text response still hits. + ### Removed - **The top-level `slices:` key is gone from `evalshift.yaml`, and its @@ -72,6 +94,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `docs/github-action.md`, and `docs/hosted.md` now name `run:create` + `run:read` + `policy:read` for the CI key. +- A run whose target was served entirely from the cache while the source ran + live showed a -100% latency change in the HTML report header and in the + run insights. Latency averages cover only calls measured live on this run, + so a role with none of them averages 0, which means unmeasured, not + instant. Both now say the latency change is not comparable unless both + roles measured some latency live. + +- Two `evalshift` processes opening the response cache at the same moment + (parallel CI jobs, or a `run` beside an `evaluate`) could crash one of them + with `table cached_calls already exists` when the cache file was new. Both + had checked for the table, and the slower one then tried to create it too. + Opening the cache now treats that error, and its migration counterpart + `duplicate column name`, as "another process already did it" and carries + on; any other schema error is still raised. + - Under SQLAlchemy 2.1, which fresh installs resolve (`sqlalchemy>=2.0`), a `CacheStore` opened on an in-memory SQLite database could silently lose concurrent writes. The default on-disk cache used by CLI runs was not @@ -142,20 +179,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 block itself is removed in this release (see Removed above). Imported agent traces (`traces import`) stay local; the bundle carries only the replay's own tool-call trace, without tool results or `model_call` events. - The response cache serves only tool-less examples, so every `run` of an - agent suite is live and full price. `--resume` hashes the suite's path, - not its contents. `push ` uploads an existing bundle as-is instead - of rebuilding it, and `bundle` needs a git SHA. `--policy-gate` also fails - when no `migration_policy` is configured. The `init` profile table had the - wrong `model-upgrade` numbers and no tool-divergence column. The - multi-turn suite example failed to load because it had no `tools`. The - failure-label list was missing `TOOL_GROUND_TRUTH_MISS`. Upstream - model-call failures and evaluator failures are handled the same way, as - errored rows excluded from the statistics. The GitHub Action docs gained - `require-policy` and the other missing inputs. `record_model_call` - examples now pass the required `tools=`. DOCS.md's header said version - 1.0.1; a new check in `tests/unit/test_docs_currency.py` keeps the version - in DOCS.md and llms-full.txt equal to the package's. + The response cache served only tool-less examples, so every `run` of an + agent suite was live and full price (it now serves them too; see Changed). + `--resume` hashes the suite's path, not its contents. `push ` + uploads an existing bundle as-is instead of rebuilding it, and `bundle` + needs a git SHA. `--policy-gate` also fails when no `migration_policy` is + configured. The `init` profile table had the wrong `model-upgrade` numbers + and no tool-divergence column. The multi-turn suite example failed to load + because it had no `tools`. The failure-label list was missing + `TOOL_GROUND_TRUTH_MISS`. Upstream model-call failures and evaluator + failures are handled the same way, as errored rows excluded from the + statistics. The GitHub Action docs gained `require-policy` and the other + missing inputs. `record_model_call` examples now pass the required + `tools=`. DOCS.md's header said version 1.0.1; a new check in + `tests/unit/test_docs_currency.py` keeps the version in DOCS.md and + llms-full.txt equal to the package's. ## [1.1.0] - 2026-09-19 diff --git a/DOCS.md b/DOCS.md index 262386f..e5dbffc 100644 --- a/DOCS.md +++ b/DOCS.md @@ -208,7 +208,7 @@ init → doctor → run → evaluate → analy ``` - **`doctor`** validates local config and shows which provider keys are visible. Exit 1 only when an existing `evalshift.yaml` fails validation — the row gives the problem count and points at `evalshift validate`, which prints each problem; missing keys are soft warnings. Its second row, `evalshift-sdk`, reports the SDK version the `evalshift` import name resolves to in this environment (`warn` when the SDK is missing or shadowed by an older CLI's leftover files; never a failure). It also reports the toolset each configured suite carries (or the flat `golden.jsonl`) and flags a suite whose examples carry more than one distinct toolset — legal (each example dispatches its own), but also the shape a wiring mistake takes. When a workflow under `.github/workflows/` uses the GitHub Action it adds a `ci pin` row: `ok` (`pinned to `) when CI installs this CLI version, `warn` when the pin is older, absent, or newer than the local CLI (see [Pin drift](#pin-drift)). When the config wires an `llm_judge` evaluator and names both `defaults.source_model` and `target_model`, a `judge family` row warns for every `judge_model` that resolves to the same provider as an arm (self-preference bias; `ok` "from a third family" otherwise, no row when either arm is unset) — advisory, never a failure; `validate` prints the same line and the report repeats it above the verdict (see [`evaluators.llm_judge`](docs/configuration.md#evaluatorsllm_judge)). -- **`run`** parses prompts, validates every example against every prompt, estimates cost, then dispatches `(prompt × example × {source, target})` calls through an async orchestrator under a concurrency semaphore. Responses to tool-less examples are cached (tool-calling examples are always dispatched live — see [Response cache](#response-cache)); progress is checkpointed every 50 completions. +- **`run`** parses prompts, validates every example against every prompt, estimates cost, then dispatches `(prompt × example × {source, target})` calls through an async orchestrator under a concurrency semaphore. Responses are cached — per call, and per replayed round for tool-calling examples (see [Response cache](#response-cache)); progress is checkpointed every 50 completions. - **`evaluate`** scores each (source, target) pair with the configured evaluators, one `EvalRecord` per pair × evaluator. Scoring runs under the same `defaults.concurrency` semaphore as `run`, and the embedding/judge calls it makes go through the same response cache. - **`analyze`** runs paired statistics per `(prompt, evaluator, slice)`, applies Benjamini–Hochberg FDR correction, classifies severities, and — when a `migration_policy` is configured — computes a pass/fail verdict. - **`report`** renders the single-file HTML report (no external assets; works offline and attaches cleanly to a PR or email), and writes the machine-written [run insights](#run-insights) narrative unless `--no-insights` is passed. The page opens on a verdict / advisory-signal / economics panel row and a six-cell run strip (examples, calls, failed-or-truncated, spend, latency Δ, mean score Δ), then the executive summary, the narrative, one section per prompt, and the methodology. Every figure on it is derived from the run's own artefacts; the deltas in the header are the run-level rollup of the per-prompt economics. Top regressions are collapsed cards — expand one for the trace diff, the tool diffs and the conversation context. The report is dark-only. @@ -237,7 +237,7 @@ Run ids look like `r_20260722_golden_a1b2c3` (`r___`). Un ### Response cache -Live responses to **tool-less** examples are cached in SQLite at `~/.evalshift/cache.db`, keyed by SHA-256 over canonical JSON of `(model, prompt, inputs, temperature, max_tokens[, history])` — plus `generation_config`, the toolset fingerprint, the round index and the sample index when each is set — with a 7-day TTL. Re-running an identical tool-less evaluation costs no run-stage calls. **Examples with a non-empty toolset are not cached:** every `run` of an agent suite dispatches them live, at full price. The evaluate-stage embedding and judge caches below still apply to them. Disable per-project with `defaults.cache: false`; wipe with `evalshift cache clear`. +Live run-stage responses are cached in SQLite at `~/.evalshift/cache.db`, keyed by SHA-256 over canonical JSON of `(model, prompt, inputs, temperature, max_tokens[, history])` — plus `generation_config`, the tools array as sent, the round index and the sample index when each is set — with a 7-day TTL. Re-running an identical evaluation costs no run-stage calls, agent suites included. An example with a non-empty toolset gets one entry per replayed round, keyed on the canonical model, the prompt and inputs, **the exact message list that round sends** (history, current turn, and the recorded rounds and fixture results teacher forcing feeds back), the tool list exactly as sent, in order (tool names, descriptions, schemas and `strict` — reordering the tools is a miss), `generation_config` (so `tool_choice` and `parallel_tool_calls`), the effective temperature and `max_tokens`, the round index and the sample index — editing a fixture re-sends only the rounds after it. A hit restores the parsed tool trace (calls with their ids and arguments, final text, refusal), tokens, cost, latency and finish reason, so the `raw.jsonl` row matches the live one apart from `cached` and `cached_rounds`; a multi-round example counts as cached only when every round was a hit. A row with any round served from the cache (`cached_rounds > 0`) carries latency recorded on an earlier run, so it is left out of the report's live latency figures and its latency delta is marked not comparable. Errors are never cached (the next run re-sends a failed round; the rounds before it come from the cache), truncated responses are cached and stay flagged, and a hit records the original call's cost and latency. A `cache.db` written by an earlier version is upgraded in place on open and keeps its entries; several `evalshift` processes may open it at once. Disable per-project with `defaults.cache: false`; wipe with `evalshift cache clear`. The cache covers the evaluate stage too: `semantic` embeddings are keyed by `(embedding model, text)`, and `llm_judge` verdicts by `(judge model, criterion, source output, target output)`. The judge key uses a canonical A/B ordering, so the per-call orientation randomization doesn't halve the hit rate — the orientation that was actually used is recorded with the verdict and replayed on a hit, leaving `metadata.target_was_a` faithful. @@ -918,11 +918,11 @@ Keys are consumed by LiteLLM at call time; EvalShift itself never stores or tran ### Will `run` cost me money? -Yes — every `run` calls a real model. Before dispatch you get a worst-case cost estimate (assumes every completion hits the registry `default_max_tokens`, 4096 — actual cost is usually much lower); above $10 it asks for confirmation. The cache makes repeat runs of unchanged tool-less calls free; tool-calling examples are dispatched live (full price) on every run. Cheapest iteration loop: small suite first, cache on. +Yes — every `run` calls a real model. Before dispatch you get a worst-case cost estimate (assumes every completion hits the registry `default_max_tokens`, 4096 — actual cost is usually much lower); above $10 it asks for confirmation. The cache makes repeat runs of unchanged calls free, tool-calling examples included. Cheapest iteration loop: small suite first, cache on. ### A model call failed mid-run -The error is recorded on that call in `raw.jsonl`; the run completes. At evaluate time the affected pair gets an errored row (a 0.5/0.5 placeholder with the error attached) that is excluded from the statistics, so it can't masquerade as a regression or an improvement. Re-running the same command re-uses cached successes (tool-less examples only — tool-calling examples are all dispatched again) and retries the failures (errored calls in a *resumed* run are not retried — start a fresh run to retry them). +The error is recorded on that call in `raw.jsonl`; the run completes. At evaluate time the affected pair gets an errored row (a 0.5/0.5 placeholder with the error attached) that is excluded from the statistics, so it can't masquerade as a regression or an improvement. Re-running the same command re-uses cached successes (for a multi-round tool example, the rounds before the failed one) and retries the failures (errored calls in a *resumed* run are not retried — start a fresh run to retry them). ### `--resume` aborts with a config-hash mismatch diff --git a/docs/agents.md b/docs/agents.md index f589398..e4e3cd2 100644 --- a/docs/agents.md +++ b/docs/agents.md @@ -82,7 +82,6 @@ defaults: source_model: gemini-2.5-flash target_model: gemini-3.1-flash-lite-preview judge_model: gemini-3.1-pro-preview - cache: false evaluators: # Top-level: what every suite is scored with. Tool evaluators do not belong @@ -223,6 +222,15 @@ the results recorded in the same round; a name match never reaches across a example as diverged if **any** replayed round diverged. * The cost estimate counts one call per replayed round; the progress bar still counts examples. +* Each round is its own response-cache entry, keyed on exactly what that round + sends: the prompt, the recorded rounds and fixture results fed back, the + tool list exactly as sent and in order (including `strict`), + `generation_config` and the round index. A repeat run is served from the + cache; editing round *k*'s fixtures re-sends only the rounds after it, and a + round that errored is re-sent while the rounds before it are not. The + example's row counts as cached only when every round was a hit; a partly + cached row's latency is left out of the live latency figures, since some of + it was measured on an earlier run. `expected_tools` is `expected_tool_rounds[0]` under both settings. `--rounds all` no longer flattens every round into `expected_tools` — that yardstick was diff --git a/docs/configuration.md b/docs/configuration.md index c679e71..31961a9 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -366,7 +366,7 @@ A list of prompt definitions. Each entry has: | `judge_model` | string | `gemini-3.1-flash-lite-preview` | Default LLM-as-judge model. | | `insights_model`| string | (none) | Model that writes the run-insights narrative rendered in `report.html` and uploaded with the bundle. Falls back to `judge_model` when unset — writing analytical prose is a harder task than a pairwise A/B verdict, so it is worth tuning separately. See [Run insights](#run-insights). | | `concurrency` | int | 10 (1 ≤ x ≤ 64) | Max in-flight LLM calls during `evalshift run` **and** `evalshift evaluate` (the embedding and judge calls made while scoring). | -| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions of tool-less examples (examples that offer tools are always dispatched live) plus `semantic` embeddings and `llm_judge` verdicts. | +| `cache` | bool | `true` | Read/write the local SQLite cache at `~/.evalshift/cache.db`. Covers run-stage completions — for examples that offer tools, one entry per replayed round — plus `semantic` embeddings and `llm_judge` verdicts. See [Response cache](https://github.com/babaliauskas/evalshift-cli/blob/main/DOCS.md#response-cache) for the key. | | `max_cost_usd` | float | 50.0 | Soft ceiling reserved for future enforcement. The pre-flight cost prompt currently triggers above $10 (skip with `--yes`). | | `max_tokens` | int | 4096 (`> 0`) | Completion length cap sent to every model call. Raise it if outputs are being truncated (the provider returns `finish_reason == "length"`); a `prompts[].max_tokens` entry overrides it per prompt. Truncated calls are detected, surfaced in the report, and **excluded from the regression statistics** so a cut-off output can't manufacture a false regression. | | `samples_per_example` | int | 1 (1 ≤ x ≤ 20) | How many times each `(prompt, example)` is sent to **each** model. Above 1, every sample is its own live call (the cache keys on the sample index), sample *i* of the source is scored against sample *i* of the target, and the example's row in `scores.jsonl` becomes the **mean over samples** with the per-sample scores and the within-example `delta_variance` under `metadata.samples`. The paired tests still run over examples, not samples, so this reduces noise without inflating `n`. Cost and the call count multiply by it; only worth turning on for a model that samples non-deterministically (see the report banner). See [Methodology](methodology.md#limitations-to-be-aware-of). | diff --git a/docs/evaluators.md b/docs/evaluators.md index ee27820..b98b1e5 100644 --- a/docs/evaluators.md +++ b/docs/evaluators.md @@ -230,7 +230,6 @@ A 100-example suite with 1 prompt and 4 evaluators (2 structural + * Evaluate: 200 embedding calls + 100 judge calls LiteLLM's pricing data drives the pre-flight estimate; the local -SQLite cache absorbs identical re-runs of tool-less examples, evaluate-stage -embedding and judge calls included. Examples that offer tools are dispatched -live on every run; only their evaluate-stage calls are cached. Evaluate dispatches its calls under +SQLite cache absorbs identical re-runs, examples that offer tools and +evaluate-stage embedding and judge calls included. Evaluate dispatches its calls under `defaults.concurrency`, same as the run stage. diff --git a/docs/faq.md b/docs/faq.md index 758c03e..be0e6d3 100644 --- a/docs/faq.md +++ b/docs/faq.md @@ -137,10 +137,9 @@ you see `≤ $0.17` and the run actually cost $0.03, that's expected. ## How do I lower the cost of a run? * **Set the SQLite cache to be on** (it's the default). A re-run of - the exact same configuration makes no run-stage calls for tool-less - examples. Examples that offer tools are not cached: an agent suite's - run stage is live, at full price, every time. Evaluate-stage embedding - and judge calls are cached either way. + the exact same configuration makes no run-stage calls, agent suites + included: examples that offer tools are cached one entry per replayed + round. Evaluate-stage embedding and judge calls are cached too. * **Use cheaper models.** The model registry assigns sensible defaults but you can drop everything to flash/mini/haiku tier. * **Skip the LLM judge.** Structural and semantic evaluators are diff --git a/docs/getting-started.md b/docs/getting-started.md index 8f79b1b..f30e698 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -38,8 +38,8 @@ fastest. EvalShift calls the provider you configure — any provider LiteLLM supports — directly using your own keys. Local runs do not send prompts or outputs to an EvalShift-operated server. -Provider responses to tool-less examples are cached locally in `~/.evalshift/cache.db`; -examples that offer tools are dispatched live on every run. +Provider responses are cached locally in `~/.evalshift/cache.db`, so re-running an +unchanged suite (agent suites included) makes no new calls. Set whichever providers you intend to use: diff --git a/llms-full.txt b/llms-full.txt index 3d4ce50..09b6764 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -90,12 +90,21 @@ as one section under the pipeline block (also on stage failure); errors are neve | traces.jsonl | traces import | optional BYO agent traces | | run_bundle.json.gz | bundle/push | optional hosted upload bundle | -Cache: SQLite ~/.evalshift/cache.db, key = SHA-256 of canonical JSON -{model, prompt, inputs, temperature, max_tokens[, history]} + generation_config, toolset -fingerprint, round index, sample index when set; TTL 7 days. TOOL-LESS examples only: an example -with a non-empty toolset is NOT cached -- every run of an agent suite is live and full price. -Evaluate-stage embedding/judge caches still apply. defaults.cache: false disables; `evalshift -cache clear` wipes. +Cache: SQLite ~/.evalshift/cache.db, key = SHA-256 of canonical JSON {model, prompt, inputs, +temperature, max_tokens[, history]} + generation_config, tools array as sent, round index, sample +index when set; TTL 7 days. Tool-calling examples ARE cached, one entry per replayed round, keyed +on canonical model + prompt + inputs + the exact message list that round sends (history, current +turn, recorded rounds + fixture results fed back) + the tool list exactly as sent, in order +(names, descriptions, schemas, strict; reordering = miss) + generation_config (tool_choice, +parallel_tool_calls) + effective temperature/max_tokens + round index + sample index; editing a +fixture re-sends only the rounds after it. A hit restores the parsed trace (call ids, arguments, +final text, refusal), tokens, cost, latency, finish_reason: the raw.jsonl row equals the live one +except `cached` (true only when every round hit) and `cached_rounds` (rounds served from cache). +cached_rounds > 0 = latency partly from an earlier run: excluded from live latency stats, +latency_comparable false. Errors never cached (next run re-sends the failed round; earlier rounds +hit); truncated responses cached and still flagged; a hit records the original cost and latency. +An older cache.db is upgraded in place and keeps its entries; concurrent opens by several +processes are safe. defaults.cache: false disables; `evalshift cache clear` wipes. Resume: `run --resume` continues newest in_progress run; requires config_hash match (canonical config + suite PATH; suite contents are NOT hashed, so start fresh after editing examples), skips (prompt_id, example_id, role, sample_index) keys already in raw.jsonl; errored calls @@ -490,7 +499,7 @@ defaults: judge_model: str = gemini-3.1-flash-lite-preview insights_model: str|null # run-narrative model; falls back to judge_model concurrency: int = 10 (1..64) # applies to run AND evaluate - cache: bool = true # covers tool-less completions, embeddings, judge verdicts + cache: bool = true # covers run completions (tool rounds too), embeddings, judge verdicts max_cost_usd: float = 50.0 # soft ceiling, reserved for future enforcement max_tokens: int = 4096 # truncated calls are EXCLUDED from stats samples_per_example: int = 1 (1..20) # repeats per (prompt, example) per model; scores averaged per example @@ -1308,8 +1317,7 @@ with: {token: "${{ secrets.EVALSHIFT_TOKEN }}", config: evalshift.yaml, ## Troubleshooting checklist Run seems expensive -> estimate is worst-case at max_tokens; actual usually far lower; cache -makes repeats of tool-less examples free (tool-calling examples are always live); iterate with a -small suite. +makes repeats of unchanged examples free, tool-calling ones included; iterate with a small suite. All severities "none" -> usually genuinely no significant difference; check n (<5 insufficient, <20 uncertain) and remember BH correction raises the bar; zero-variance comparisons skip. Verdict "inconclusive" -> (1) all evaluators advisory (fresh init: flip blocking: true as the @@ -1325,7 +1333,7 @@ migration_decision.json as reason/recommendations). fresh. Suite contents are not hashed: an edited suite at the same path resumes silently. Failed calls -> recorded in raw.jsonl with error, pair gets an errored 0.5/0.5 row excluded from the statistics; fresh run -retries them (cache serves the tool-less successes; tool-calling examples are all re-dispatched). +retries them (cache serves the successes; for a multi-round tool example, the rounds before the failed one). Config rejected -> extra="forbid": check for typo'd keys; error names the exact path. "`thresholds` was removed" -> the key is gone from evalshift.yaml; delete it. It gated nothing; migration_policy is the single source of truth. Nothing replaced it. diff --git a/src/evalshift_cli/cache/schema.py b/src/evalshift_cli/cache/schema.py index df0a97a..fe3a766 100644 --- a/src/evalshift_cli/cache/schema.py +++ b/src/evalshift_cli/cache/schema.py @@ -54,6 +54,12 @@ class CachedCall(Base): so a cache hit stays flagged as truncated. Rows written before this column existed read back as ``NULL`` (treated as not truncated) and age out within the TTL. + trace_json: For a tool-calling round, the parsed + :class:`~evalshift_cli.evaluators.tool_models.ToolTrace` as JSON + (``ToolTrace.model_dump_json()``); ``response_text`` then holds + the trace's ``final_text`` (or ``""``). ``NULL`` for text-only + responses. Nullable and backfilled additively on open, so a DB + written before tool calls were cached keeps serving its text rows. created_at: When the cache row was written. Driver of TTL. """ @@ -69,6 +75,7 @@ class CachedCall(Base): cost_usd: Mapped[float] = mapped_column(Float, nullable=False) latency_ms: Mapped[int] = mapped_column(Integer, nullable=False) finish_reason: Mapped[str | None] = mapped_column(String(32), nullable=True, default=None) + trace_json: Mapped[str | None] = mapped_column(Text, nullable=True, default=None) created_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, diff --git a/src/evalshift_cli/cache/store.py b/src/evalshift_cli/cache/store.py index 0c1c6a3..db605f3 100644 --- a/src/evalshift_cli/cache/store.py +++ b/src/evalshift_cli/cache/store.py @@ -24,8 +24,10 @@ from pathlib import Path from typing import Any +from pydantic import ValidationError from sqlalchemy import delete, select, text from sqlalchemy.dialects.sqlite import insert as sqlite_insert +from sqlalchemy.exc import OperationalError from sqlalchemy.ext.asyncio import AsyncEngine, AsyncSession, async_sessionmaker from sqlalchemy.pool import SingletonThreadPool, StaticPool @@ -35,13 +37,20 @@ create_engine, default_database_url, ) +from evalshift_cli.evaluators.tool_models import ToolTrace DEFAULT_TTL_DAYS: int = 7 @dataclass(frozen=True, slots=True) class CachedResponse: - """The cache-side view of a previously-completed LLM call.""" + """The cache-side view of a previously-completed LLM call. + + ``trace`` is set only for a tool-calling round: the parsed + :class:`ToolTrace` the live call produced, restored field for field. + Text-only responses (and every row written before tool calls were + cached) carry ``None``. + """ response_text: str input_tokens: int @@ -50,6 +59,7 @@ class CachedResponse: latency_ms: int created_at: datetime finish_reason: str | None = None + trace: ToolTrace | None = None def cache_key( @@ -61,7 +71,7 @@ def cache_key( max_tokens: int, history: Sequence[Mapping[str, str]] | None = None, generation_config: Mapping[str, Any] | None = None, - toolset_fingerprint: str | None = None, + tools_payload: Sequence[Mapping[str, Any]] | None = None, round_index: int | None = None, sample_index: int | None = None, ) -> str: @@ -73,38 +83,37 @@ def cache_key( Args: history: Multi-turn conversation prefix (recorded turns dispatched - ahead of ``prompt_text``). Included in the hashed payload only - when not ``None``, so single-turn calls (``history=None``) - produce byte-identical keys to before this parameter existed — - existing cache entries stay valid. An empty list is still - included (and hashes differently from ``None``) since it marks - the call as message-mode. + ahead of ``prompt_text``); the tool path passes the round's whole + dispatched message list here instead (history, current turn and the + teacher-forced recorded rounds), so every byte the provider sees is + keyed. Included in the hashed payload only when not ``None``, so + single-turn calls (``history=None``) produce byte-identical keys to + before this parameter existed — existing cache entries stay valid. + An empty list is still included (and hashes differently from + ``None``) since it marks the call as message-mode. generation_config: Recorded per-example generation config applied at dispatch. Same inclusion rule as ``history``: hashed only when not ``None``, so config-less calls keep their pre-existing keys. - toolset_fingerprint: Content-address of the toolset this call was - dispatched with (``"sha256:"`` from - :func:`evalshift_cli.captures.toolset.fingerprint_tools`). Same - inclusion rule as ``history``/``generation_config``: hashed only - when not ``None``, so a call that never sends a ``tools`` - parameter to the provider at all keeps its pre-existing key. An - example's toolset, whether spelled as an inline ``tools:`` list - or a ``toolset_ref`` sidecar, resolves to the same fingerprint - before it reaches this function — see - :func:`evalshift_cli.runner.orchestrator._fingerprint_toolset` — so - the two spellings of one toolset never fork the cache, while two - genuinely different toolsets always produce different keys. + tools_payload: The ``tools`` array exactly as sent to the provider, + in order (:func:`evalshift_cli.models.client.serialize_tools`, the + same helper dispatch uses). Same inclusion rule as + ``history``/``generation_config``: hashed only when not ``None``, + so a call that never sends a ``tools`` parameter keeps its + pre-existing key. ``sort_keys`` orders each tool's own keys but + never the list, so reordering the tools, or changing a name, + description, schema or ``strict`` flag, changes the key — each of + those changes what the provider receives. An inline ``tools:`` + list and a ``toolset_ref`` sidecar share entries exactly when they + resolve to the same tools in the same order. round_index: 0-based round of a teacher-forced multi-round replay (see :meth:`evalshift_cli.suite.models.SuiteExample.rounds_to_replay`). Same inclusion rule as the three above: hashed only when not - ``None``, so a single-shot call keeps its pre-existing key. ``0`` - is a real round and hashes *differently* from ``None`` — round 0 - of a replayed loop is dispatched with a different message list - than the same example replayed single-shot would be. The tool - path bypasses the cache entirely (unchanged since v0.2), so today - nothing passes this; it exists so that when tool-call caching - lands the round dimension is already in the key and no cache - migration is needed. + ``None``, so the text path (which never replays rounds and passes + ``None``) keeps its pre-existing keys. ``0`` is a real round and + hashes *differently* from ``None``. The tool path passes the real + round for every round it dispatches, single-shot examples + included, so a tool-calling key can never collide with a text + one. sample_index: 0-based sample of a repeated-sampling run (``defaults.samples_per_example > 1``). Same inclusion rule as ``round_index``: hashed only when not ``None``, so every @@ -125,8 +134,8 @@ def cache_key( payload["history"] = [dict(m) for m in history] if generation_config is not None: payload["generation_config"] = dict(generation_config) - if toolset_fingerprint is not None: - payload["toolset_fingerprint"] = toolset_fingerprint + if tools_payload is not None: + payload["tools_payload"] = [dict(t) for t in tools_payload] if round_index is not None: payload["round_index"] = round_index if sample_index is not None: @@ -182,9 +191,12 @@ async def open( """ url = database_url or default_database_url(path) engine = create_engine(url) - async with engine.begin() as conn: - await conn.run_sync(Base.metadata.create_all) - await _ensure_finish_reason_column(conn) + try: + await _ensure_schema(engine) + except BaseException: + # No store owns the engine yet, so nothing else would close it. + await engine.dispose() + raise return cls(engine, ttl=ttl) async def close(self) -> None: @@ -221,6 +233,14 @@ async def get(self, key: str) -> CachedResponse | None: created_at = _ensure_utc(row.created_at) if created_at < cutoff: return None + trace: ToolTrace | None = None + if row.trace_json is not None: + try: + trace = ToolTrace.model_validate_json(row.trace_json) + except ValidationError: + # Written by a version whose trace shape this one cannot + # read: re-dispatch rather than fail the run. + return None return CachedResponse( response_text=row.response_text, input_tokens=row.input_tokens, @@ -229,6 +249,7 @@ async def get(self, key: str) -> CachedResponse | None: latency_ms=row.latency_ms, created_at=created_at, finish_reason=row.finish_reason, + trace=trace, ) async def put( @@ -244,9 +265,14 @@ async def put( cost_usd: float, latency_ms: int, finish_reason: str | None = None, + trace: ToolTrace | None = None, ) -> None: """Insert (or replace) a cache entry. + ``trace`` is the parsed tool trace of a tool-calling round, stored as + JSON beside ``response_text`` and restored by :meth:`get`; leave it + ``None`` for a text-only response. + A single atomic ``INSERT ... ON CONFLICT DO UPDATE``: concurrent callers routinely miss the same key and race to write it back (the evaluate stage scores many pairs at once, and identical model @@ -264,6 +290,7 @@ async def put( "cost_usd": cost_usd, "latency_ms": latency_ms, "finish_reason": finish_reason, + "trace_json": trace.model_dump_json() if trace is not None else None, "created_at": _utcnow(), } async with self._session() as session: @@ -307,19 +334,60 @@ def _pool_shares_one_connection(engine: AsyncEngine) -> bool: return isinstance(engine.pool, StaticPool | SingletonThreadPool) -async def _ensure_finish_reason_column(conn: Any) -> None: - """Additively backfill the ``finish_reason`` column on pre-existing DBs. +# Nullable columns added after the table first shipped, with the DDL that adds +# each one to a DB created before it existed. Append-only. +_ADDITIVE_COLUMNS: tuple[tuple[str, str], ...] = ( + ("finish_reason", "VARCHAR(32)"), + ("trace_json", "TEXT"), +) + + +async def _ensure_schema(engine: AsyncEngine) -> None: + """Create the table and backfill :data:`_ADDITIVE_COLUMNS`, tolerating races. + + Several ``evalshift`` processes can open one cache DB at once (parallel CI + jobs, a ``run`` next to an ``evaluate``). Each step is check-then-change, + so another process can make the same change in between: ``create_all`` + then fails with "table … already exists" and a backfill with "duplicate + column name". Both mean the schema the loser wanted is already there, so + they are swallowed; any other error is raised. Each change runs in its own + transaction so a lost race rolls back only that statement. + """ + try: + async with engine.begin() as conn: + await conn.run_sync(Base.metadata.create_all) + except OperationalError as exc: + if not _lost_schema_race(exc, "already exists"): + raise + async with engine.connect() as conn: + columns = await _existing_columns(conn) + for name, ddl_type in _ADDITIVE_COLUMNS: + if name in columns: + continue + try: + async with engine.begin() as conn: + await conn.execute(text(f"ALTER TABLE cached_calls ADD COLUMN {name} {ddl_type}")) + except OperationalError as exc: + if not _lost_schema_race(exc, "duplicate column name"): + raise + + +async def _existing_columns(conn: Any) -> set[str]: + """Column names of ``cached_calls`` as this connection sees them now. ``create_all`` only creates missing *tables*, never alters existing ones, - and the disposable 7-day cache has no migration framework. A cache DB - created before this column existed would be missing it, so probe - ``PRAGMA table_info`` and ``ALTER TABLE ... ADD COLUMN`` when absent. A - fresh DB already has the column via ``create_all``, making this a no-op. + and the disposable 7-day cache has no migration framework, so a DB created + before a column in :data:`_ADDITIVE_COLUMNS` existed is missing it until + :func:`_ensure_schema` adds it. Existing rows read back ``NULL`` there (not + truncated, no tool trace), so they keep serving the text path. """ result = await conn.execute(text("PRAGMA table_info(cached_calls)")) - columns = {row[1] for row in result.fetchall()} - if "finish_reason" not in columns: - await conn.execute(text("ALTER TABLE cached_calls ADD COLUMN finish_reason VARCHAR(32)")) + return {row[1] for row in result.fetchall()} + + +def _lost_schema_race(exc: OperationalError, message: str) -> bool: + """Whether ``exc`` is SQLite reporting that a concurrent opener got there first.""" + return message in str(exc.orig) def _utcnow() -> datetime: diff --git a/src/evalshift_cli/hosted/bundle.py b/src/evalshift_cli/hosted/bundle.py index f7e7996..8461e89 100644 --- a/src/evalshift_cli/hosted/bundle.py +++ b/src/evalshift_cli/hosted/bundle.py @@ -365,7 +365,8 @@ def _build_examples( Delta conventions match ``reports/json.py::_build_example_rows`` exactly — target minus source, with latency forced to 0 and flagged as - incomparable whenever either side replayed from cache. + incomparable whenever either side replayed any round from cache + (``Call.latency_replayed``). ``cached_rounds`` itself is not uploaded. """ # One output per (prompt, example, role) — sample 0 on a repeated-sampling # run — so a later sample never overwrites the one shown. @@ -392,7 +393,10 @@ def _build_examples( row_score = sum(score_values) / len(score_values) if score_values else None worst = min((item.delta for item in scored), default=None) latency_comparable = ( - source is not None and target is not None and not source.cached and not target.cached + source is not None + and target is not None + and not source.latency_replayed + and not target.latency_replayed ) delta_latency = ( target.latency_ms - source.latency_ms diff --git a/src/evalshift_cli/insights/facts.py b/src/evalshift_cli/insights/facts.py index 810ad76..a81ff29 100644 --- a/src/evalshift_cli/insights/facts.py +++ b/src/evalshift_cli/insights/facts.py @@ -218,9 +218,16 @@ def build_facts( "cost_source_usd": _usd(source.total_cost_usd), "cost_target_usd": _usd(target.total_cost_usd), "cost_delta_pct": _relative_pct(source.total_cost_usd, target.total_cost_usd), - # Cache hits carry ``latency_ms = 0`` by convention, so a role with no - # live calls has no measured latency and no percentage to report. - "latency_delta_pct": _relative_pct(source.latency_ms_avg, target.latency_ms_avg), + # Latency averages cover only calls measured fully live on this run + # (cache hits keep their original latency and are excluded via + # ``Call.latency_replayed``). A role with no such call averages 0.0, + # meaning "unmeasured", so either side lacking one leaves no + # percentage to report — not -100% or a division by zero. + "latency_delta_pct": ( + _relative_pct(source.latency_ms_avg, target.latency_ms_avg) + if source.live_calls and target.live_calls + else NOT_COMPARABLE + ), "cost_ceiling_pct": budget_limits.get("max_cost_increase", NOT_AVAILABLE), "latency_ceiling_pct": budget_limits.get("max_latency_increase", NOT_AVAILABLE), "regression_rate_pct": ( diff --git a/src/evalshift_cli/models/client.py b/src/evalshift_cli/models/client.py index 89c728a..b50a98c 100644 --- a/src/evalshift_cli/models/client.py +++ b/src/evalshift_cli/models/client.py @@ -27,7 +27,7 @@ import random import sys import time -from collections.abc import Iterator +from collections.abc import Iterator, Sequence from contextlib import contextmanager from dataclasses import dataclass from typing import Any, Final, cast @@ -654,9 +654,7 @@ async def complete_messages_with_tools( mt = meta.default_max_tokens if max_tokens is None else max_tokens provider = detect_provider(canonical) - tools_payload = [ - t.to_anthropic() if provider == "anthropic" else t.to_openai() for t in tools - ] + tools_payload = serialize_tools(canonical, tools) kwargs: dict[str, Any] = { "model": canonical, @@ -790,6 +788,20 @@ async def _dispatch_with_retry( # --------------------------------------------------------------------------- +def serialize_tools(canonical: str, tools: Sequence[ToolSpec]) -> list[dict[str, Any]]: + """The ``tools`` array sent to ``canonical``, in the order given. + + Anthropic models get :meth:`ToolSpec.to_anthropic`; everything else gets + :meth:`ToolSpec.to_openai` (Gemini and DeepSeek accept the OpenAI shape via + LiteLLM). :meth:`ModelClient.complete_messages_with_tools` sends exactly + this list, and the run cache keys on it, so the key and the wire share one + source: anything that changes what the provider receives — a tool's + schema, description, ``strict`` flag, or the list order — changes the key. + """ + provider = detect_provider(canonical) + return [t.to_anthropic() if provider == "anthropic" else t.to_openai() for t in tools] + + def _build_result( canonical: str, response: Any, @@ -963,4 +975,5 @@ def _map_exception(exc: BaseException) -> ModelClientError: "RateLimitError", "RetryPolicy", "ToolCompletionResult", + "serialize_tools", ] diff --git a/src/evalshift_cli/reports/economics.py b/src/evalshift_cli/reports/economics.py index dbc861f..002463b 100644 --- a/src/evalshift_cli/reports/economics.py +++ b/src/evalshift_cli/reports/economics.py @@ -17,9 +17,12 @@ class RoleEconomics: """Per-prompt × role rollup of operational stats from raw.jsonl. - `latency_ms_avg` / `_p95` are computed only over live (non-cached, - non-error) calls — cache hits replay text from disk so their - `latency_ms` is 0 and would skew the averages downward. + `latency_ms_avg` / `_p95` are computed only over calls whose latency was + measured on this run (no error, no round replayed from cache — + `Call.latency_replayed`). A cache hit carries the latency recorded when it + originally ran, so counting it would report an old measurement as new; + `live_calls` is that sample count. A partly cached multi-round row is + neither live here nor `cached_calls` (it spent money on this run). """ calls: int @@ -87,7 +90,7 @@ def role_economics(calls: list[Call]) -> RoleEconomics: failed = sum(1 for c in calls if c.error is not None) truncated = sum(1 for c in calls if c.truncated) empty_output = sum(1 for c in calls if is_empty_output(c)) - live_latencies = [c.latency_ms for c in calls if not c.cached and c.error is None] + live_latencies = [c.latency_ms for c in calls if not c.latency_replayed and c.error is None] if live_latencies: sorted_l = sorted(live_latencies) avg_ms = sum(sorted_l) / len(sorted_l) diff --git a/src/evalshift_cli/reports/html.py b/src/evalshift_cli/reports/html.py index d2008d8..f38c920 100644 --- a/src/evalshift_cli/reports/html.py +++ b/src/evalshift_cli/reports/html.py @@ -362,6 +362,19 @@ def _pct_delta(source: float, target: float) -> float | None: return (target - source) / source * 100.0 +def _run_latency_delta_pct(totals: Mapping[str, Mapping[str, float]]) -> float | None: + """Run-level latency change, or ``None`` unless both roles measured some live. + + The means cover only calls measured fully live on this run (cache hits and + rows with any cached round are excluded), so a role with none averages + 0.0 — unmeasured, not instant. :func:`_pct_delta` already refuses a zero + source; a zero target would otherwise read as a -100% change. + """ + if not (totals["source"]["live_calls"] and totals["target"]["live_calls"]): + return None + return _pct_delta(totals["source"]["latency_ms_avg"], totals["target"]["latency_ms_avg"]) + + def _run_role_totals(sections: Sequence[Any]) -> dict[str, dict[str, float]]: """Roll every prompt's economics up into one source/target pair. @@ -505,10 +518,7 @@ def render_html(report: ReportData, *, insight: Insight | None = None) -> str: source_cost_usd=role_totals["source"]["cost"], target_cost_usd=role_totals["target"]["cost"], cost_delta_pct=_pct_delta(role_totals["source"]["cost"], role_totals["target"]["cost"]), - latency_delta_pct=_pct_delta( - role_totals["source"]["latency_ms_avg"], - role_totals["target"]["latency_ms_avg"], - ), + latency_delta_pct=_run_latency_delta_pct(role_totals), avg_score_delta=_avg_score_delta(report.executive_summary), headline=_headline_comparison(report.prompt_sections), budgets_passed=budgets_passed, diff --git a/src/evalshift_cli/reports/json.py b/src/evalshift_cli/reports/json.py index 179f747..035ee56 100644 --- a/src/evalshift_cli/reports/json.py +++ b/src/evalshift_cli/reports/json.py @@ -200,7 +200,7 @@ class ExampleRow: example_id: str tags: list[str] delta_latency_ms: int - latency_comparable: bool # False when either side cached (delta forced to 0) + latency_comparable: bool # False when either side replayed a round from cache (delta 0) delta_cost_usd: float worst_delta_score: float | None # Whether the target held every tool measurement the source did — i.e. @@ -636,8 +636,9 @@ def _build_example_rows( """Assemble one ExampleRow per example for a single prompt. Δ values are target − source. Latency is reported as 0 when either - side cached (cached calls carry latency_ms = 0 by convention) so - the figure stays meaningful only on live × live pairs. + side replayed any round from the cache (`Call.latency_replayed`: its + `latency_ms` is the original run's, not this one's) so the figure stays + meaningful only on live × live pairs. """ sides_by_example: dict[str, dict[str, Call]] = {} for c in calls: @@ -654,7 +655,7 @@ def _build_example_rows( tgt = sides.get("target") if src is None or tgt is None: continue # incomplete pair; skip rather than misreport - latency_comparable = not (src.cached or tgt.cached) + latency_comparable = not (src.latency_replayed or tgt.latency_replayed) delta_lat = tgt.latency_ms - src.latency_ms if latency_comparable else 0 delta_cost = tgt.cost_usd - src.cost_usd diff --git a/src/evalshift_cli/reports/templates/report.html.j2 b/src/evalshift_cli/reports/templates/report.html.j2 index b408d1e..a2f2fe1 100644 --- a/src/evalshift_cli/reports/templates/report.html.j2 +++ b/src/evalshift_cli/reports/templates/report.html.j2 @@ -450,7 +450,7 @@ -

Latency stats cover live calls only ({{ section.economics.source.live_calls }} source / {{ section.economics.target.live_calls }} target); cache hits replay from disk.

+

Latency stats cover calls measured fully live on this run ({{ section.economics.source.live_calls }} source / {{ section.economics.target.live_calls }} target); cache hits and rows with any cached round are excluded.

@@ -464,7 +464,7 @@ Example Tags - Time Δ + Time Δ Cost Δ Worst score Δ Tool match diff --git a/src/evalshift_cli/runner/models.py b/src/evalshift_cli/runner/models.py index 7b37add..f7c1fce 100644 --- a/src/evalshift_cli/runner/models.py +++ b/src/evalshift_cli/runner/models.py @@ -247,10 +247,21 @@ class Call(_StrictModel): over replayed rounds. cost_usd: Per-call cost (``litellm.completion_cost``), summed over replayed rounds; ``0.0`` when the model isn't priced. - latency_ms: Wall time of the live call, summed over replayed - rounds. ``0`` for cache hits after the first run (we keep the - *original* latency). - cached: ``True`` if the response came from the local cache. + latency_ms: Wall time of the provider call, summed over replayed + rounds. A round served from the cache contributes the latency + recorded when it originally ran live, not ``0`` — so whenever + :attr:`latency_replayed` is true the figure is not a measurement + of this run. + cached: ``True`` if the response came from the local cache — for a + multi-round replay, only when every round did (each round is + its own cache entry). ``cached`` means "nothing was spent on this + run"; a row with any live round is not cached. + cached_rounds: How many of the row's provider rounds were served + from the cache: ``0`` for a fully live row (and for every row + written before the field existed), the round count for a fully + cached one, in between for a multi-round replay that re-sent only + some rounds. Anything above ``0`` means ``latency_ms`` mixes in + replayed latencies — see :attr:`latency_replayed`. error: ``None`` on success; the stringified error on failure. A multi-round replay that fails part-way names the round it died in (``"round 2/3: "``) and records no ``trace`` — a @@ -278,6 +289,8 @@ class Call(_StrictModel): cost_usd: float = 0.0 latency_ms: int = 0 cached: bool = False + # Defaulted so pre-existing raw.jsonl lines still validate on resume. + cached_rounds: int = Field(default=0, ge=0) error: str | None = None # v0.2 — populated only for tool-aware calls; ``None`` for plain text # ones. The orchestrator switches between ``ModelClient.complete`` and @@ -300,6 +313,18 @@ def truncated(self) -> bool: """True when the provider cut the output off at the token cap.""" return self.finish_reason == "length" + @property + def latency_replayed(self) -> bool: + """True when any part of ``latency_ms`` was replayed from the cache. + + Such a row's latency is not a measurement of this run, so it stays out + of live latency statistics and makes a latency delta incomparable. A + partly cached multi-round row is not :attr:`cached` (it spent money) + but is latency-replayed. ``cached`` is checked too because rows + written before ``cached_rounds`` existed read it back as ``0``. + """ + return self.cached or self.cached_rounds > 0 + def representative_calls(calls: Iterable[Call]) -> list[Call]: """The one call per ``(prompt, example, role)`` a consumer should display. diff --git a/src/evalshift_cli/runner/orchestrator.py b/src/evalshift_cli/runner/orchestrator.py index b414686..d55a9f4 100644 --- a/src/evalshift_cli/runner/orchestrator.py +++ b/src/evalshift_cli/runner/orchestrator.py @@ -48,7 +48,6 @@ from evalshift_cli.cache.store import CacheStore, cache_key from evalshift_cli.captures.reader import CaptureError, capture_base, load_toolset -from evalshift_cli.captures.toolset import fingerprint_tools from evalshift_cli.config.models import EvalShiftConfig from evalshift_cli.evaluators.tool_models import ToolCall, ToolSpec, ToolTrace from evalshift_cli.models.capabilities import ( @@ -57,7 +56,12 @@ silently_unsent_params, unsupported_params, ) -from evalshift_cli.models.client import ModelClient, ModelClientError +from evalshift_cli.models.client import ( + ModelClient, + ModelClientError, + ToolCompletionResult, + serialize_tools, +) from evalshift_cli.models.registry import resolve_model from evalshift_cli.parsers.base import PromptParseError, PromptTemplate from evalshift_cli.parsers.manual import ManualParser @@ -540,19 +544,6 @@ def resolve_suite_tools( } -def _fingerprint_toolset(tools: Sequence[ToolSpec]) -> str: - """Content-address a resolved toolset the same way regardless of its source. - - An inline ``tools:`` list and a ``toolset_ref`` sidecar both resolve to the - same ``list[ToolSpec]`` shape by the time dispatch sees them. Fingerprinting - that resolved list — via Task 2's - :func:`~evalshift_cli.captures.toolset.fingerprint_tools` — rather than trusting - a ``toolset_ref`` string verbatim guarantees the two spellings of the same - toolset produce the same fingerprint, and therefore the same cache key. - """ - return fingerprint_tools([t.to_anthropic() for t in tools]) - - def _setup_run( *, config: EvalShiftConfig, @@ -1206,24 +1197,12 @@ async def _execute( ``defaults.samples_per_example > 1`` so each sample is its own live call. For agent-style work items (``item.tools`` non-empty), dispatches to - :meth:`ModelClient.complete_with_tools` and stores the parsed - :class:`ToolTrace` on the resulting :class:`Call`. The local SQLite - cache is intentionally bypassed for tool calls in v0.2 — caching - serialised traces is a v0.3 polish. + :func:`_execute_with_tools`, which stores the parsed :class:`ToolTrace` + on the resulting :class:`Call` and caches each replayed round on its own. """ meta = resolve_model(item.model_id) messages = build_messages(item.example, prompt_text) - if item.tools: - return await _execute_with_tools( - client=client, - run_id=run_id, - item=item, - prompt_text=prompt_text, - canonical_id=meta.id, - messages=messages, - ) - # Effective cap: prompt/run config override, else the registry default. # The same value is keyed AND sent so a cache hit matches the live call. effective_max_tokens = ( @@ -1237,19 +1216,24 @@ async def _execute( gen_temperature if gen_temperature is not None else meta.default_temperature ) + if item.tools: + return await _execute_with_tools( + client=client, + cache=cache, + run_id=run_id, + item=item, + prompt_text=prompt_text, + canonical_id=meta.id, + messages=messages, + gen_temperature=gen_temperature, + gen_extra=gen_extra, + key_temperature=effective_temperature, + key_max_tokens=effective_max_tokens, + cache_enabled=cache_enabled, + cache_sample_index=cache_sample_index, + ) + history_for_key = history_for_cache_key(item.example) - # item.tools is always empty here — a non-empty toolset routes to - # _execute_with_tools above and never reaches this line — so this is - # always None in practice today. Written as a real conditional (not - # hard-coded) because that empty-ness is a routing fact, not a cache-key - # rule: this is the one call site that mirrors _execute_with_tools's own - # `_fingerprint_toolset(item.tools) if item.tools else None` shape, so the - # two stay in lockstep if either path's routing condition ever changes. - # None (omit from the payload) matches the history/generation_config - # precedent below: this call never sends a `tools` parameter to the - # provider at all, so it keeps its pre-existing cache key rather than - # forking on a toolset dimension that doesn't apply to it. - toolset_fingerprint = _fingerprint_toolset(item.tools) if item.tools else None key = cache_key( model_id=meta.id, prompt_text=prompt_text, @@ -1258,11 +1242,10 @@ async def _execute( max_tokens=effective_max_tokens, history=history_for_key, generation_config=item.example.generation_config, - toolset_fingerprint=toolset_fingerprint, - # None: this path is single-shot by construction. Only the tool path - # replays rounds, and it bypasses the cache entirely (unchanged since - # v0.2), so nothing passes a real round today — the key's round - # dimension exists so tool-call caching can land without a migration. + # None for both: this call never sends a ``tools`` parameter and is + # single-shot by construction, so it keeps its pre-existing key. The + # tool path always sets both, so the two paths' keys never collide. + tools_payload=None, round_index=None, sample_index=cache_sample_index, ) @@ -1288,6 +1271,7 @@ async def _execute( cost_usd=hit.cost_usd, latency_ms=hit.latency_ms, cached=True, + cached_rounds=1, finish_reason=hit.finish_reason, ) @@ -1354,11 +1338,18 @@ async def _execute( async def _execute_with_tools( *, client: ModelClient, + cache: CacheStore, run_id: str, item: WorkItem, prompt_text: str, canonical_id: str, + gen_temperature: float | None, + gen_extra: dict[str, Any] | None, + key_temperature: float, + key_max_tokens: int, + cache_enabled: bool, messages: list[dict[str, Any]] | None = None, + cache_sample_index: int | None = None, ) -> Call: """Tool-aware call path: teacher-forced replay loop + one Call per example. @@ -1379,11 +1370,32 @@ async def _execute_with_tools( single-turn examples, which keeps the existing single-prompt :meth:`ModelClient.complete_with_tools` path so a single-shot example makes a byte-identical client call to before this loop existed. Rounds ``k >= 1`` - are always message-mode. The local cache is bypassed throughout — unchanged - from before. + are always message-mode. + + Each round is its own cache entry. Teacher forcing makes the rounds + independent requests — round *k* is sent the recording, never the + candidate's earlier rounds — so a round is keyed on exactly what it sends: + the dispatched message list (``None`` for a plain-prompt round 0), the + tools array exactly as sent (:func:`serialize_tools`, in order), the + generation config, the effective temperature and token cap + (``key_temperature`` / ``key_max_tokens``, as the text path keys them), the + round index and ``cache_sample_index``. ``gen_temperature`` / ``gen_extra`` + are :func:`translate_generation_config`'s output, computed once by + :func:`_execute` so its warnings fire once per call. A hit restores the + round's :class:`ToolCompletionResult` (trace, tokens, cost, latency, finish + reason) and feeds the same merge a live round does, so the :class:`Call` is + identical apart from ``cached`` and ``cached_rounds``. Policies follow the + text path: an errored round is not cached (earlier rounds, genuine + responses to their own requests, stay cached); a truncated round is cached + and warned about on a hit; ``cache_enabled=False`` neither reads nor + writes. The :class:`Call` records ``cached_rounds`` hits and is ``cached`` + only when every round was a hit — a row with any live round spent money on + this run, but its summed latency is no longer a fresh measurement + (:attr:`Call.latency_replayed`). """ - gen_temperature, gen_extra = translate_generation_config(item.example.generation_config) rounds = item.example.rounds_to_replay() + tools_payload = serialize_tools(canonical_id, item.tools) + cached_rounds = 0 merged_calls: list[ToolCall] = [] input_tokens = 0 @@ -1401,40 +1413,62 @@ async def _execute_with_tools( if round_index == 0 else build_round_messages(item.example, prompt_text, round_index) ) - try: - if round_messages is not None: - result = await client.complete_messages_with_tools( - model=canonical_id, - messages=round_messages, - tools=list(item.tools), + key = cache_key( + model_id=canonical_id, + prompt_text=prompt_text, + inputs=item.example.inputs, + temperature=key_temperature, + max_tokens=key_max_tokens, + history=round_messages, + generation_config=item.example.generation_config, + tools_payload=tools_payload, + round_index=round_index, + sample_index=cache_sample_index, + ) + result = await _cached_tool_round(cache, key, canonical_id) if cache_enabled else None + if result is not None: + cached_rounds += 1 + else: + try: + result = await _dispatch_tool_round( + client=client, + item=item, + canonical_id=canonical_id, + prompt_text=prompt_text, + round_messages=round_messages, temperature=gen_temperature, - max_tokens=item.max_tokens, extra=gen_extra, ) - else: - result = await client.complete_with_tools( - model=canonical_id, - prompt=prompt_text, - tools=list(item.tools), - temperature=gen_temperature, - max_tokens=item.max_tokens, - extra=gen_extra, + except ModelClientError as exc: + # A partially replayed example is an unmeasured example: the + # rounds that did complete are dropped, exactly as a failed + # single-shot call records no trace. The round is named so the + # failure is attributable without re-running; a single-shot + # call keeps the bare provider error it always carried. Nothing + # is cached for the failed round. + return Call( + run_id=run_id, + prompt_id=item.prompt.id, + example_id=item.example.id, + model_id=canonical_id, + role=item.role, + sample_index=item.sample_index, + error=f"round {round_index + 1}/{rounds}: {exc}" if rounds > 1 else str(exc), + ) + if cache_enabled: + await cache.put( + key, + model_id=canonical_id, + prompt_text=prompt_text, + inputs=item.example.inputs, + response_text=result.trace.final_text or "", + input_tokens=result.input_tokens, + output_tokens=result.output_tokens, + cost_usd=result.cost_usd, + latency_ms=result.latency_ms, + finish_reason=result.finish_reason, + trace=result.trace, ) - except ModelClientError as exc: - # A partially replayed example is an unmeasured example: the rounds - # that did complete are dropped, exactly as a failed single-shot - # call records no trace. The round is named so the failure is - # attributable without re-running; a single-shot call keeps the - # bare provider error it always carried. - return Call( - run_id=run_id, - prompt_id=item.prompt.id, - example_id=item.example.id, - model_id=canonical_id, - role=item.role, - sample_index=item.sample_index, - error=f"round {round_index + 1}/{rounds}: {exc}" if rounds > 1 else str(exc), - ) input_tokens += result.input_tokens output_tokens += result.output_tokens @@ -1489,11 +1523,75 @@ async def _execute_with_tools( output_tokens=output_tokens, cost_usd=cost_usd, latency_ms=latency_ms, + cached=cached_rounds == rounds, + cached_rounds=cached_rounds, trace=trace, finish_reason=finish_reason, ) +async def _cached_tool_round( + cache: CacheStore, + key: str, + canonical_id: str, +) -> ToolCompletionResult | None: + """A cached tool round rebuilt as the :class:`ToolCompletionResult` it came from. + + ``None`` on a miss, and on a hit without a trace (never written by the tool + path, so not something this round can be served from). + ``raw_provider_response`` comes back empty: the orchestrator never + persists or reads it. + """ + hit = await cache.get(key) + if hit is None or hit.trace is None: + return None + if hit.finish_reason == "length": + log.warning( + "cached tool response for model %s was truncated (finish_reason=length)", + canonical_id, + ) + return ToolCompletionResult( + trace=hit.trace, + model_id=canonical_id, + input_tokens=hit.input_tokens, + output_tokens=hit.output_tokens, + cost_usd=hit.cost_usd, + latency_ms=hit.latency_ms, + raw_provider_response={}, + finish_reason=hit.finish_reason, + ) + + +async def _dispatch_tool_round( + *, + client: ModelClient, + item: WorkItem, + canonical_id: str, + prompt_text: str, + round_messages: list[dict[str, Any]] | None, + temperature: float | None, + extra: dict[str, Any] | None, +) -> ToolCompletionResult: + """Send one tool round live: message-mode when there are messages, else the plain prompt.""" + if round_messages is not None: + return await client.complete_messages_with_tools( + model=canonical_id, + messages=round_messages, + tools=list(item.tools), + temperature=temperature, + max_tokens=item.max_tokens, + extra=extra, + ) + return await client.complete_with_tools( + model=canonical_id, + prompt=prompt_text, + tools=list(item.tools), + temperature=temperature, + max_tokens=item.max_tokens, + extra=extra, + ) + + # Re-exported so the CLI can catch them with one import. __all__ = [ "CHECKPOINT_EVERY", diff --git a/tests/conftest.py b/tests/conftest.py index 6ee54d3..347eaaf 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -340,6 +340,26 @@ def build(self, **overrides: Any) -> BundleBuildResult: return build_bundle(self.run_id, **kwargs) +@pytest.fixture(autouse=True) +def _isolated_response_cache( + tmp_path_factory: pytest.TempPathFactory, + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Point the default on-disk response cache at a fresh per-test file. + + Command-level tests (``run``, ``compare``, ``evaluate`` and the integration + pipelines) open the cache without passing a URL, which resolves to + ``~/.evalshift/cache.db``. Unredirected, they wrote the developer's real + cache, and a row left there by one test run could serve the next one. + ``tmp_path_factory`` rather than ``tmp_path``, so tests that inspect their + own ``tmp_path`` never find a stray ``cache.db`` in it. + """ + monkeypatch.setattr( + "evalshift_cli.cache.schema.DEFAULT_CACHE_PATH", + tmp_path_factory.mktemp("response-cache") / "cache.db", + ) + + @pytest.fixture def run_fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> RunFixture: """A completed two-example run, one live pair and one cache-replayed pair. diff --git a/tests/unit/test_bundle_shape.py b/tests/unit/test_bundle_shape.py index 7689b3a..b34163a 100644 --- a/tests/unit/test_bundle_shape.py +++ b/tests/unit/test_bundle_shape.py @@ -784,3 +784,37 @@ def _rewrite_suite_history(suite_path: Path, system_prompt: str) -> None: for row in rows: row["history"] = [{"role": "system", "content": system_prompt}] suite_path.write_text("".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8") + + +def test_bundle_latency_is_incomparable_when_a_side_replayed_rounds() -> None: + from evalshift_cli.hosted.bundle import _build_examples + + def call(role: str, example_id: str, latency: int, cached_rounds: int = 0) -> Call: + return Call( + run_id="r_20260601_abc123", + prompt_id="p", + example_id=example_id, + model_id="m", + role=role, # type: ignore[arg-type] + latency_ms=latency, + cached_rounds=cached_rounds, + ) + + rows = _build_examples( + suite=Suite(examples=[]), + calls=[ + call("source", "live", 100), + call("target", "live", 130), + call("source", "mixed", 100), + call("target", "mixed", 130, cached_rounds=2), + ], + scores=[], + tool_evaluator_names=frozenset(), + ) + by_id = {r["example_id"]: r for r in rows} + assert by_id["live"]["latency_comparable"] is True + assert by_id["live"]["delta_latency_ms"] == 30 + assert by_id["mixed"]["latency_comparable"] is False + assert by_id["mixed"]["delta_latency_ms"] == 0 + # The count itself stays local: the bundle row schema has no such field. + assert "cached_rounds" not in by_id["mixed"] diff --git a/tests/unit/test_cache.py b/tests/unit/test_cache.py index 90dde06..fca1998 100644 --- a/tests/unit/test_cache.py +++ b/tests/unit/test_cache.py @@ -14,17 +14,18 @@ from collections.abc import AsyncIterator from datetime import timedelta from pathlib import Path +from typing import Any, ClassVar import pytest import pytest_asyncio -from sqlalchemy import event +from sqlalchemy import event, text from sqlalchemy.ext.asyncio import AsyncEngine from typer.testing import CliRunner from evalshift_cli.cache.schema import Base, create_engine from evalshift_cli.cache.store import CacheStore, cache_key -from evalshift_cli.captures.toolset import fingerprint_tools from evalshift_cli.cli.main import app +from evalshift_cli.evaluators.tool_models import ToolCall, ToolTrace # Use an in-memory database for every test to keep them fast and hermetic. IN_MEMORY_DB = "sqlite+aiosqlite:///:memory:" @@ -261,10 +262,10 @@ def test_same_history_same_args_same_key(self) -> None: ) assert a == b - # -- toolset_fingerprint (Task 8: per-example toolsets reach the cache key) -- + # -- tools_payload (the tool path keys the tools array exactly as sent) -- - def test_no_toolset_fingerprint_is_byte_identical_to_pre_toolset_payload(self) -> None: - """Regression: omitting ``toolset_fingerprint`` must not change the hashed payload. + def test_no_tools_payload_is_byte_identical_to_pre_toolset_payload(self) -> None: + """Regression: omitting ``tools_payload`` must not change the hashed payload. Mirrors ``test_no_history_is_byte_identical_to_pre_history_payload``. A call dispatched via ``complete``/``complete_messages`` never sends a ``tools`` @@ -294,120 +295,60 @@ def test_no_toolset_fingerprint_is_byte_identical_to_pre_toolset_payload(self) - ) expected = hashlib.sha256(old_payload.encode("utf-8")).hexdigest() - actual = cache_key( - model_id=model_id, - prompt_text=prompt_text, - inputs=inputs, - temperature=temperature, - max_tokens=max_tokens, - ) - assert actual == expected - - actual_explicit_none = cache_key( - model_id=model_id, - prompt_text=prompt_text, - inputs=inputs, - temperature=temperature, - max_tokens=max_tokens, - toolset_fingerprint=None, - ) - assert actual_explicit_none == expected - - def test_toolset_fingerprint_changes_key(self) -> None: - base = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - ) - with_toolset = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint="sha256:" + "a" * 64, - ) - assert base != with_toolset - - def test_different_toolsets_produce_different_keys(self) -> None: - """The cache key differs across differing toolsets (Task 8 requirement).""" - a = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint="sha256:" + "a" * 64, - ) - b = cache_key( + base = { + "model_id": model_id, + "prompt_text": prompt_text, + "inputs": inputs, + "temperature": temperature, + "max_tokens": max_tokens, + } + assert cache_key(**base) == expected # type: ignore[arg-type] + assert cache_key(**base, tools_payload=None) == expected # type: ignore[arg-type] + + def _tool_key(self, tools_payload: list[dict[str, Any]] | None) -> str: + return cache_key( model_id="m", prompt_text="hi", inputs={}, temperature=0.0, max_tokens=1024, - toolset_fingerprint="sha256:" + "b" * 64, + tools_payload=tools_payload, ) - assert a != b - def test_same_toolset_fingerprint_same_key(self) -> None: - a = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint="sha256:" + "c" * 64, - ) - b = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint="sha256:" + "c" * 64, - ) - assert a == b + _SEARCH: ClassVar[dict[str, Any]] = { + "name": "search_orders", + "description": "Look up orders.", + "input_schema": {}, + } + _REFUND: ClassVar[dict[str, Any]] = { + "name": "issue_refund", + "description": "Refund an order.", + "input_schema": {}, + } - def test_inline_and_ref_resolved_toolset_fingerprint_the_same_key(self) -> None: - """An inline toolset and a ``toolset_ref`` to the same tools key identically. + def test_tools_payload_changes_key(self) -> None: + assert self._tool_key([self._SEARCH]) != self._tool_key(None) - Simulates the two spellings ``SuiteExample`` allows for one toolset: - ``fp_inline`` fingerprints a hand-authored ``tools:`` list directly; - ``fp_from_sidecar`` fingerprints the *same* tools as they would come back - off a promoted sidecar -- a freshly-built, differently-ordered list of - equivalent dicts (``fingerprint_tools`` sorts by name, so list order must - not matter). Both must fingerprint identically, and two ``cache_key()`` - calls built from each must collide. - """ - inline_tools = [ - {"name": "search_orders", "description": "Look up orders.", "input_schema": {}}, - {"name": "issue_refund", "description": "Refund an order.", "input_schema": {}}, - ] - sidecar_tools = list(reversed(inline_tools)) # same tools, different order + def test_empty_tools_payload_differs_from_none(self) -> None: + assert self._tool_key([]) != self._tool_key(None) - fp_inline = fingerprint_tools(inline_tools) - fp_from_sidecar = fingerprint_tools(sidecar_tools) - assert fp_inline == fp_from_sidecar + def test_different_tools_produce_different_keys(self) -> None: + assert self._tool_key([self._SEARCH]) != self._tool_key([self._REFUND]) - a = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint=fp_inline, + def test_same_tools_same_order_same_key(self) -> None: + assert self._tool_key([self._SEARCH, self._REFUND]) == self._tool_key( + [dict(self._SEARCH), dict(self._REFUND)] ) - b = cache_key( - model_id="m", - prompt_text="hi", - inputs={}, - temperature=0.0, - max_tokens=1024, - toolset_fingerprint=fp_from_sidecar, + + def test_reordered_tools_produce_different_keys(self) -> None: + """The provider receives the list in order, so order is part of the request.""" + assert self._tool_key([self._SEARCH, self._REFUND]) != self._tool_key( + [self._REFUND, self._SEARCH] ) - assert a == b + + def test_dict_key_order_inside_a_tool_does_not_matter(self) -> None: + reordered = dict(reversed(list(self._SEARCH.items()))) + assert self._tool_key([self._SEARCH]) == self._tool_key([reordered]) # -- round_index (teacher-forced multi-round replay) -- @@ -415,7 +356,7 @@ def test_no_round_index_is_byte_identical_to_pre_round_payload(self) -> None: """Omitting ``round_index`` must not change the hashed payload. Same inclusion rule as ``history`` / ``generation_config`` / - ``toolset_fingerprint``: hashed only when not ``None``, so every key + ``tools_payload``: hashed only when not ``None``, so every key minted before the round dimension existed stays valid. """ import hashlib @@ -748,3 +689,367 @@ def test_zero_differs_from_none(self) -> None: def test_different_samples_produce_different_keys(self) -> None: assert self._key(0) != self._key(1) + + +# --------------------------------------------------------------------------- +# Tool-call responses — the value carries a parsed ToolTrace +# --------------------------------------------------------------------------- + + +def _rich_trace() -> ToolTrace: + """A trace exercising every field a downstream consumer reads.""" + return ToolTrace( + calls=[ + ToolCall( + tool_name="search_orders", + arguments={"customer": "Zoë", "limit": 5, "filters": {"open": True}}, + call_id="toolu_01ABC", + sequence_index=0, + round_index=0, + ), + ToolCall( + tool_name="issue_refund", + arguments={"order_id": "A-1", "amount": 12.5, "items": [1, 2]}, + call_id="call_xyz", + parent_call_id="toolu_01ABC", + sequence_index=1, + round_index=1, + ), + ToolCall( + tool_name="broken", + arguments={"_parse_error": True}, + call_id=None, + sequence_index=2, + round_index=1, + ), + ], + final_text="Refunded order A-1.", + raised_refusal=True, + refusal_text="I can only refund once.", + round_count=2, + ) + + +def _tool_put_kwargs(trace: ToolTrace) -> dict[str, object]: + kw = _put_kwargs() + kw["response_text"] = trace.final_text or "" + return kw + + +class TestCacheStoreToolTrace: + async def test_round_trip_preserves_the_trace_exactly(self, store: CacheStore) -> None: + trace = _rich_trace() + kw = _tool_put_kwargs(trace) + await store.put("k", **kw, finish_reason="tool_calls", trace=trace) # type: ignore[arg-type] + got = await store.get("k") + assert got is not None + assert got.trace == trace + assert got.trace is not None + assert got.trace.model_dump() == trace.model_dump() + assert got.response_text == "Refunded order A-1." + assert got.finish_reason == "tool_calls" + assert (got.input_tokens, got.output_tokens, got.latency_ms) == (10, 5, 250) + assert got.cost_usd == pytest.approx(0.0001) + + async def test_tool_only_trace_round_trips(self, store: CacheStore) -> None: + trace = ToolTrace( + calls=[ToolCall(tool_name="t", arguments={}, call_id="c1", sequence_index=0)], + final_text=None, + ) + await store.put("k", **_tool_put_kwargs(trace), trace=trace) # type: ignore[arg-type] + got = await store.get("k") + assert got is not None + assert got.trace == trace + assert got.trace is not None + assert got.trace.final_text is None + + async def test_text_rows_carry_no_trace(self, store: CacheStore) -> None: + await store.put("k", **_put_kwargs()) # type: ignore[arg-type] + got = await store.get("k") + assert got is not None + assert got.trace is None + + async def test_a_trace_that_no_longer_validates_is_a_miss(self, tmp_path: Path) -> None: + # A row written by some other version whose trace shape this code + # cannot read must be re-dispatched, not crash the run. + url = f"sqlite+aiosqlite:///{tmp_path / 'c.db'}" + store = await CacheStore.open(database_url=url) + try: + trace = _rich_trace() + await store.put("k", **_tool_put_kwargs(trace), trace=trace) # type: ignore[arg-type] + async with store._engine.begin() as conn: + await conn.execute( + text("UPDATE cached_calls SET trace_json = :j"), + {"j": '{"calls": "not a list"}'}, + ) + assert await store.get("k") is None + finally: + await store.close() + + +# DDL exactly as `CacheStore.open` created it on origin/main (af4e6fe), dumped +# from `sqlite_master` of a DB that code wrote — the shape every existing +# user's ~/.evalshift/cache.db has today. +_ORIGIN_MAIN_DDL = ( + "CREATE TABLE cached_calls (\n" + "\tcache_key VARCHAR(64) NOT NULL, \n" + "\tmodel_id VARCHAR(128) NOT NULL, \n" + "\tprompt_text TEXT NOT NULL, \n" + "\tinputs_json TEXT NOT NULL, \n" + "\tresponse_text TEXT NOT NULL, \n" + "\tinput_tokens INTEGER NOT NULL, \n" + "\toutput_tokens INTEGER NOT NULL, \n" + "\tcost_usd FLOAT NOT NULL, \n" + "\tlatency_ms INTEGER NOT NULL, \n" + "\tfinish_reason VARCHAR(32), \n" + "\tcreated_at DATETIME NOT NULL, \n" + "\tPRIMARY KEY (cache_key)\n" + ")" +) + + +class TestCacheMigrationFromOriginMain: + async def _legacy_db(self, tmp_path: Path) -> str: + from datetime import UTC, datetime + + from sqlalchemy.ext.asyncio import create_async_engine + + url = f"sqlite+aiosqlite:///{tmp_path / 'legacy.db'}" + engine = create_async_engine(url) + async with engine.begin() as conn: + await conn.execute(text(_ORIGIN_MAIN_DDL)) + await conn.execute( + text( + "INSERT INTO cached_calls (cache_key, model_id, prompt_text, inputs_json, " + "response_text, input_tokens, output_tokens, cost_usd, latency_ms, " + "finish_reason, created_at) VALUES ('old', 'm', 'p', '{}', 'old text', " + "1, 2, 0.5, 3, 'stop', :now)" + ), + {"now": datetime.now(UTC).strftime("%Y-%m-%d %H:%M:%S.%f")}, + ) + await engine.dispose() + return url + + async def test_existing_text_rows_still_hit(self, tmp_path: Path) -> None: + store = await CacheStore.open(database_url=await self._legacy_db(tmp_path)) + try: + got = await store.get("old") + assert got is not None + assert got.response_text == "old text" + assert got.finish_reason == "stop" + assert got.trace is None + finally: + await store.close() + + async def test_tool_rows_can_be_written_after_the_backfill(self, tmp_path: Path) -> None: + url = await self._legacy_db(tmp_path) + store = await CacheStore.open(database_url=url) + try: + trace = _rich_trace() + await store.put("new", **_tool_put_kwargs(trace), trace=trace) # type: ignore[arg-type] + got = await store.get("new") + assert got is not None + assert got.trace == trace + finally: + await store.close() + # Re-opening an already-migrated DB is a no-op, not a duplicate-column error. + again = await CacheStore.open(database_url=url) + try: + assert await again.count() == 2 + finally: + await again.close() + + +class TestToolRoundKeySensitivity: + """Every component the tool path keys on moves the key; identical inputs hit. + + Mirrors what :func:`evalshift_cli.runner.orchestrator._execute_with_tools` + passes per round: the dispatched message list as ``history``, the tools + array exactly as sent, the generation config (tool_choice / parallel_tool_calls), the + round index and the sample index. + """ + + _TOOLS: ClassVar[list[dict[str, Any]]] = [ + {"name": "t", "description": "d", "input_schema": {"type": "object"}} + ] + + def _kwargs(self) -> dict[str, Any]: + return { + "model_id": "anthropic/claude-sonnet-4-5", + "prompt_text": "Hello Alex", + "inputs": {"name": "Alex"}, + "temperature": 0.0, + "max_tokens": 1024, + "history": [ + {"role": "user", "content": "Hello Alex"}, + { + "role": "assistant", + "content": "", + "tool_calls": [ + { + "id": "call_r0_0", + "type": "function", + "function": {"name": "t", "arguments": "{}"}, + } + ], + }, + {"role": "tool", "tool_call_id": "call_r0_0", "content": "{}"}, + ], + "generation_config": {"tool_choice": "auto"}, + "tools_payload": [dict(t) for t in self._TOOLS], + "round_index": 1, + "sample_index": None, + } + + def test_identical_inputs_same_key(self) -> None: + assert cache_key(**self._kwargs()) == cache_key(**self._kwargs()) + + @pytest.mark.parametrize( + ("field", "value"), + [ + ("model_id", "openai/gpt-4o"), + ("prompt_text", "Hello Bea"), + ("inputs", {"name": "Bea"}), + ("temperature", 0.7), + ("max_tokens", 2048), + ("history", [{"role": "user", "content": "Hello Alex"}]), + ("generation_config", {"tool_choice": "required"}), + ("generation_config", {"tool_choice": "auto", "parallel_tool_calls": False}), + ("round_index", 2), + ("sample_index", 1), + ], + ) + def test_each_component_changes_the_key(self, field: str, value: Any) -> None: + changed = self._kwargs() + changed[field] = value + assert cache_key(**changed) != cache_key(**self._kwargs()) + + def test_a_changed_fixture_result_changes_the_key(self) -> None: + changed = self._kwargs() + changed["history"] = [dict(m) for m in changed["history"]] + changed["history"][2]["content"] = '{"hits": 1}' + assert cache_key(**changed) != cache_key(**self._kwargs()) + + def test_tool_strictness_changes_the_key(self) -> None: + changed = self._kwargs() + changed["tools_payload"] = [{**self._TOOLS[0], "strict": True}] + assert cache_key(**changed) != cache_key(**self._kwargs()) + + def test_a_changed_tool_schema_changes_the_key(self) -> None: + changed = self._kwargs() + changed["tools_payload"] = [ + {**self._TOOLS[0], "input_schema": {"type": "object", "required": ["q"]}} + ] + assert cache_key(**changed) != cache_key(**self._kwargs()) + + def test_reordering_the_tools_changes_the_key(self) -> None: + second = {"name": "u", "description": "e", "input_schema": {"type": "object"}} + a = self._kwargs() + a["tools_payload"] = [self._TOOLS[0], second] + b = self._kwargs() + b["tools_payload"] = [second, self._TOOLS[0]] + assert cache_key(**a) != cache_key(**b) + + +def test_the_suite_never_opens_the_users_real_cache() -> None: + # Several command-level tests open the default on-disk cache. Without the + # autouse redirect in tests/conftest.py they read and wrote the developer's + # ~/.evalshift/cache.db, so a cached row from an earlier test run could + # serve a later one. + from evalshift_cli.cache import schema + + assert Path.home() / ".evalshift" / "cache.db" != schema.DEFAULT_CACHE_PATH + + +class TestConcurrentSchemaSetup: + """Several `evalshift` processes can open one cache.db at the same moment. + + Each probes the schema and then changes it, so another process can make the + same change in between. Losing that race must not abort the run: the + schema the loser wanted is already there. + """ + + async def test_a_column_added_since_the_probe_is_not_an_error( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + from evalshift_cli.cache import store as store_module + + url = f"sqlite+aiosqlite:///{tmp_path / 'c.db'}" + await (await CacheStore.open(database_url=url)).close() # fully migrated + + # Stale probe: this process "saw" a legacy table, another added the columns. + async def stale(_conn: Any) -> set[str]: + return {"cache_key", "model_id"} + + monkeypatch.setattr(store_module, "_existing_columns", stale) + again = await CacheStore.open(database_url=url) + await again.close() + + async def test_a_table_created_since_the_check_is_not_an_error( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + url = f"sqlite+aiosqlite:///{tmp_path / 'c.db'}" + await (await CacheStore.open(database_url=url)).close() + + # Another process created the table between create_all's check and its CREATE. + real_create_all = Base.metadata.create_all + + def create_without_check(bind: Any, **_kw: Any) -> None: + real_create_all(bind, checkfirst=False) + + monkeypatch.setattr(Base.metadata, "create_all", create_without_check) + again = await CacheStore.open(database_url=url) + await again.close() + + async def test_other_schema_errors_still_raise( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + from sqlalchemy.exc import OperationalError + + from evalshift_cli.cache import store as store_module + + monkeypatch.setattr(store_module, "_ADDITIVE_COLUMNS", (("broken", "INTEGER DEFAULT ("),)) + disposed = self._spy_on_dispose(monkeypatch) + with pytest.raises(OperationalError): + await CacheStore.open(database_url=f"sqlite+aiosqlite:///{tmp_path / 'c.db'}") + assert disposed == [True] + + async def test_other_errors_while_creating_the_table_still_raise( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + from sqlalchemy.exc import OperationalError + + def failing_create_all(_bind: Any, **_kw: Any) -> None: + raise OperationalError("CREATE TABLE cached_calls", {}, Exception("disk I/O error")) + + monkeypatch.setattr(Base.metadata, "create_all", failing_create_all) + disposed = self._spy_on_dispose(monkeypatch) + with pytest.raises(OperationalError, match="disk I/O error"): + await CacheStore.open(database_url=f"sqlite+aiosqlite:///{tmp_path / 'c.db'}") + assert disposed == [True] + + @staticmethod + def _spy_on_dispose(monkeypatch: pytest.MonkeyPatch) -> list[bool]: + """Record every ``AsyncEngine.dispose`` so a failed open can be checked for leaks.""" + calls: list[bool] = [] + real_dispose = AsyncEngine.dispose + + async def spy(self: AsyncEngine, close: bool = True) -> None: + calls.append(True) + await real_dispose(self, close) + + monkeypatch.setattr(AsyncEngine, "dispose", spy) + return calls + + async def test_stores_opened_together_on_a_legacy_file_all_succeed( + self, tmp_path: Path + ) -> None: + url = await TestCacheMigrationFromOriginMain()._legacy_db(tmp_path) + stores = await asyncio.gather(*(CacheStore.open(database_url=url) for _ in range(4))) + try: + got = await stores[0].get("old") + assert got is not None + assert got.trace is None + finally: + for s in stores: + await s.close() diff --git a/tests/unit/test_economics.py b/tests/unit/test_economics.py index 823756d..93c770c 100644 --- a/tests/unit/test_economics.py +++ b/tests/unit/test_economics.py @@ -132,3 +132,21 @@ def test_role_economics_to_dict_is_json_ready() -> None: "latency_ms_avg", "latency_ms_p95", } + + +def test_partly_cached_rows_stay_out_of_live_latency_stats() -> None: + # A multi-round row whose earlier rounds were replayed from cache mixes a + # fresh latency with an old one: it is not a measurement of this run. + from evalshift_cli.reports.economics import role_economics + + econ = role_economics( + [ + _call(latency_ms=100), + _call(latency_ms=900, cached_rounds=2), + _call(latency_ms=700, cached=True, cached_rounds=3), + ], + ) + assert econ.live_calls == 1 + assert econ.cached_calls == 1 + assert econ.latency_ms_avg == 100.0 + assert econ.latency_ms_p95 == 100.0 diff --git a/tests/unit/test_insight_facts.py b/tests/unit/test_insight_facts.py index 3207b97..8862d23 100644 --- a/tests/unit/test_insight_facts.py +++ b/tests/unit/test_insight_facts.py @@ -67,7 +67,25 @@ def test_delta_spread_is_rendered_from_the_examples(sample_run: dict[str, Any]) def test_latency_is_not_reported_as_a_measurement_on_a_cached_replay( sample_run: dict[str, Any], ) -> None: - """Cache hits carry ``latency_ms = 0``; a percentage there is a fiction.""" + """No role measured live latency; a percentage there is a fiction.""" + assert build_facts(**sample_run).rendered["latency_delta_pct"] == "not comparable" + + +def test_latency_is_not_comparable_when_only_one_side_ran_live( + sample_run: dict[str, Any], +) -> None: + # A target served from the cache (or with some rounds replayed) has no + # live latency sample; its 0.0 average is "unmeasured", not "instant", so + # it must not render as a -100% latency change. + sample_run["economics"] = PromptEconomics( + source=role(live_calls=21, cached_calls=0, latency_ms_avg=1000.0), + target=role( + live_calls=0, + cached_calls=21, + latency_ms_avg=0.0, + total_cost_usd=COST_TARGET, + ), + ) assert build_facts(**sample_run).rendered["latency_delta_pct"] == "not comparable" diff --git a/tests/unit/test_model_client.py b/tests/unit/test_model_client.py index 6194746..bff73e6 100644 --- a/tests/unit/test_model_client.py +++ b/tests/unit/test_model_client.py @@ -35,7 +35,9 @@ RateLimitError, RetryPolicy, ToolCompletionResult, + serialize_tools, ) +from evalshift_cli.models.registry import resolve_model # --------------------------------------------------------------------------- # Fakes @@ -1264,3 +1266,54 @@ async def test_other_providers_are_sent_unchanged( tools=[_DEMO_TOOL], ) assert captured["kwargs"]["messages"] == _REPLAYED_ROUND + + +_SECOND_TOOL = ToolSpec( + name="add_note", + description="Attach a note", + input_schema={"type": "object", "properties": {"text": {"type": "string"}}}, + strict=True, +) + + +class TestSerializeTools: + """The one place a toolset becomes the ``tools`` array on the wire. + + The run cache keys on this exact list, so it must be what + ``complete_messages_with_tools`` sends, in the caller's order. + """ + + def test_openai_shape_in_the_given_order(self) -> None: + payload = serialize_tools(resolve_model("gpt-4o").id, [_SECOND_TOOL, _DEMO_TOOL]) + assert payload == [_SECOND_TOOL.to_openai(), _DEMO_TOOL.to_openai()] + + def test_anthropic_shape_for_anthropic_models(self) -> None: + payload = serialize_tools(resolve_model("claude-4.5-sonnet").id, [_DEMO_TOOL, _SECOND_TOOL]) + assert payload == [_DEMO_TOOL.to_anthropic(), _SECOND_TOOL.to_anthropic()] + + async def test_the_wire_carries_exactly_the_serialized_list( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + captured = _patch_tools_acompletion(monkeypatch, _OPENAI_SINGLE_RESPONSE) + await ModelClient().complete_messages_with_tools( + model="gpt-4o", + messages=[{"role": "user", "content": "hi"}], + tools=[_SECOND_TOOL, _DEMO_TOOL], + ) + assert captured["kwargs"]["tools"] == serialize_tools( + resolve_model("gpt-4o").id, [_SECOND_TOOL, _DEMO_TOOL] + ) + + async def test_dispatch_goes_through_serialize_tools( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + # Shared source, not a parallel copy: patching the helper changes the wire. + sentinel = [{"name": "sentinel"}] + monkeypatch.setattr(client_module, "serialize_tools", lambda _c, _t: sentinel) + captured = _patch_tools_acompletion(monkeypatch, _OPENAI_SINGLE_RESPONSE) + await ModelClient().complete_messages_with_tools( + model="gpt-4o", + messages=[{"role": "user", "content": "hi"}], + tools=[_DEMO_TOOL], + ) + assert captured["kwargs"]["tools"] == sentinel diff --git a/tests/unit/test_orchestrator.py b/tests/unit/test_orchestrator.py index 0356b30..62e356b 100644 --- a/tests/unit/test_orchestrator.py +++ b/tests/unit/test_orchestrator.py @@ -39,6 +39,7 @@ ModelClient, ModelClientError, ToolCompletionResult, + serialize_tools, ) from evalshift_cli.runner.checkpoint import ( iter_calls, @@ -49,7 +50,6 @@ RunAborted, RunResult, _build_work_list, - _fingerprint_toolset, build_messages, build_round_messages, history_for_cache_key, @@ -296,6 +296,8 @@ async def test_repeat_run_with_cache_serves_cached_calls( assert second.cached_calls == 4 assert second.live_calls == 0 assert counter["calls"] == 4 # unchanged from before + assert all(r.cached_rounds == 1 for r in iter_calls(second.run_dir)) + assert all(r.cached_rounds == 0 for r in iter_calls(first.run_dir)) async def test_effective_max_tokens_and_finish_reason( self, @@ -1386,8 +1388,8 @@ def test_resolution_cached_across_examples_sharing_one_ref(self, tmp_path: Path) ) assert second == first - def test_inline_and_ref_to_same_tools_fingerprint_identically(self, tmp_path: Path) -> None: - """The two spellings of one toolset must key the cache identically.""" + def test_inline_and_ref_to_same_tools_send_the_same_payload(self, tmp_path: Path) -> None: + """The two spellings of one toolset send (and so key) the same tools array.""" tool = self._tool_a() ref = self._write_sidecar(tmp_path, [tool]) @@ -1401,7 +1403,8 @@ def test_inline_and_ref_to_same_tools_fingerprint_identically(self, tmp_path: Pa toolset_cache={}, ) - assert _fingerprint_toolset(inline_tools) == _fingerprint_toolset(ref_tools) + for model in ("anthropic/claude-sonnet-4-5", "gemini/gemini-2.5-flash"): + assert serialize_tools(model, inline_tools) == serialize_tools(model, ref_tools) # --------------------------------------------------------------------------- @@ -2242,3 +2245,486 @@ def test_preflight_cost_counts_samples(self, tmp_path: Path) -> None: assert one.total_calls == 6 assert three.total_calls == 18 assert three.estimated_usd == pytest.approx(one.estimated_usd * 3) + + +# --------------------------------------------------------------------------- +# Tool-calling examples go through the response cache, one entry per round +# --------------------------------------------------------------------------- + + +def _round_of(kwargs: dict[str, Any]) -> int: + """The teacher-forced round a tool-path dispatch asks for. + + Derived from what was sent (not from a dispatch counter) so it stays right + when earlier rounds were served from the cache: round *k* carries exactly + *k* recorded assistant turns with ``call_r{j}_{i}`` ids. + """ + messages = kwargs.get("messages") + if messages is None: + return 0 + return sum( + 1 + for m in messages + if m.get("role") == "assistant" + and any(str(tc.get("id", "")).startswith("call_r") for tc in m.get("tool_calls") or []) + ) + + +def _install_tool_fake( + monkeypatch: pytest.MonkeyPatch, + *, + fail_rounds: set[int] | None = None, + finish_reason_for: Any = None, +) -> list[dict[str, Any]]: + """Patch both tool entry points with a fake that records every live dispatch. + + Each round's trace exercises every field a downstream consumer reads: + provider call ids, nested/unicode arguments, a refusal, final text. + """ + seen: list[dict[str, Any]] = [] + + async def fake(self: ModelClient, **kwargs: Any) -> ToolCompletionResult: + model = str(kwargs["model"]) + round_index = _round_of(kwargs) + seen.append({"model": model, "round": round_index}) + if fail_rounds is not None and round_index in fail_rounds: + raise ModelClientError(f"upstream exploded in round {round_index}") + trace = ToolTrace( + calls=[ + ToolCall( + tool_name="search", + arguments={"q": f"r{round_index}", "opts": {"lang": "é", "k": 1.5}}, + call_id=f"toolu_{model[-3:]}_{round_index}_{i}", + sequence_index=i, + ) + for i in range(round_index + 1) + ], + final_text=f"{model} answer {round_index}", + raised_refusal=round_index == 1, + refusal_text="partial refusal" if round_index == 1 else None, + ) + return ToolCompletionResult( + trace=trace, + model_id=model, + input_tokens=7 + round_index, + output_tokens=3 + round_index, + cost_usd=0.25 * (round_index + 1), + latency_ms=100 + round_index, + raw_provider_response={"id": "resp"}, + finish_reason=( + "tool_calls" if finish_reason_for is None else finish_reason_for(round_index) + ), + ) + + monkeypatch.setattr(ModelClient, "complete_with_tools", fake) + monkeypatch.setattr(ModelClient, "complete_messages_with_tools", fake) + return seen + + +def _rows(run_dir: Path) -> list[dict[str, Any]]: + """raw.jsonl rows minus the fields a cache hit is allowed to change.""" + return sorted( + (r.model_dump(exclude={"run_id", "cached", "cached_rounds"}) for r in iter_calls(run_dir)), + key=lambda r: (r["example_id"], r["role"], r["sample_index"]), + ) + + +def _tool_config(**defaults: Any) -> EvalShiftConfig: + return EvalShiftConfig( + prompts=list(_round_config().prompts), + defaults=Defaults(concurrency=4, max_cost_usd=100.0, **defaults), + ) + + +def _single_shot_tool_suite() -> Suite: + return Suite( + examples=[ + suite_example(id=f"ex{i}", inputs={"name": f"User{i}"}, tools=[_ROUND_TOOL]) + for i in range(2) + ], + ) + + +class TestToolPathCache: + def _kwargs(self, tmp_path: Path, cache: CacheStore, suite: Suite, **over: Any) -> Any: + config_path, suite_path, runs_base = _writeable_paths(tmp_path) + kwargs: dict[str, Any] = { + "config": _tool_config(), + "config_path": config_path, + "suite": suite, + "suite_path": suite_path, + "source_model": "gemini-2.5-flash", + "target_model": "gemini-2.5-pro", + "runs_base": runs_base, + "yes": True, + "cache": cache, + } + kwargs.update(over) + return kwargs + + async def test_repeat_run_of_a_single_shot_tool_suite_is_served_from_cache( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + seen = _install_tool_fake(monkeypatch) + kwargs = self._kwargs(tmp_path, cache, _single_shot_tool_suite()) + + first = await run_orchestrator(**kwargs) + assert (first.live_calls, first.cached_calls, len(seen)) == (4, 0, 4) + + second = await run_orchestrator(**kwargs) + assert len(seen) == 4 # nothing dispatched + assert (second.live_calls, second.cached_calls) == (0, 4) + assert second.total_cost_usd == pytest.approx(first.total_cost_usd) + assert _rows(second.run_dir) == _rows(first.run_dir) + assert all(r.cached and r.cached_rounds == 1 for r in iter_calls(second.run_dir)) + assert not any(r.cached or r.cached_rounds for r in iter_calls(first.run_dir)) + + @pytest.mark.parametrize( + "history", + [ + None, + [ + ChatMessage(role="user", content="earlier turn"), + ChatMessage(role="assistant", content="earlier reply"), + ], + ], + ids=["single-turn", "multi-turn"], + ) + async def test_repeat_run_of_a_teacher_forced_example_is_served_from_cache( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + history: list[ChatMessage] | None, + ) -> None: + seen = _install_tool_fake(monkeypatch) + suite = Suite(examples=[_two_round_example(history=history)]) + kwargs = self._kwargs(tmp_path, cache, suite) + + first = await run_orchestrator(**kwargs) + assert len(seen) == 6 # 3 rounds x 2 roles + assert (first.live_calls, first.cached_calls) == (2, 0) + + second = await run_orchestrator(**kwargs) + assert len(seen) == 6 + assert (second.live_calls, second.cached_calls) == (0, 2) + rows = _rows(second.run_dir) + assert rows == _rows(first.run_dir) + for row in rows: + assert row["trace"]["round_count"] == 3 + assert [c["round_index"] for c in row["trace"]["calls"]] == [0, 1, 1, 2, 2, 2] + assert row["trace"]["raised_refusal"] is True + assert row["input_tokens"] == 7 + 8 + 9 + assert row["cost_usd"] == pytest.approx(0.25 + 0.5 + 0.75) + assert row["latency_ms"] == 100 + 101 + 102 + + async def test_cache_disabled_dispatches_every_tool_round_live( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + seen = _install_tool_fake(monkeypatch) + suite = Suite(examples=[_two_round_example()]) + kwargs = self._kwargs(tmp_path, cache, suite, config=_tool_config(cache=False)) + + await run_orchestrator(**kwargs) + second = await run_orchestrator(**kwargs) + assert len(seen) == 12 + assert (second.live_calls, second.cached_calls) == (2, 0) + assert await cache.count() == 0 + + async def test_an_errored_tool_call_is_not_cached( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + _install_tool_fake(monkeypatch, fail_rounds={0}) + kwargs = self._kwargs(tmp_path, cache, _single_shot_tool_suite()) + + first = await run_orchestrator(**kwargs) + assert first.failed_calls == 4 + assert await cache.count() == 0 + + seen = _install_tool_fake(monkeypatch) + second = await run_orchestrator(**kwargs) + assert len(seen) == 4 + assert (second.live_calls, second.cached_calls, second.failed_calls) == (4, 0, 0) + + async def test_a_failed_round_is_redispatched_and_earlier_rounds_are_served( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + # Rounds are independent requests (teacher forcing feeds back the + # recording, never the candidate), so round 0's genuine response stays + # valid when round 1 fails — only the failed round and later re-run. + _install_tool_fake(monkeypatch, fail_rounds={1}) + kwargs = self._kwargs(tmp_path, cache, Suite(examples=[_two_round_example()])) + first = await run_orchestrator(**kwargs) + assert first.failed_calls == 2 + assert all(r.error and r.error.startswith("round 2/3") for r in iter_calls(first.run_dir)) + + seen = _install_tool_fake(monkeypatch) + second = await run_orchestrator(**kwargs) + assert sorted(s["round"] for s in seen) == [1, 1, 2, 2] + # A row with any live round spent money now: it counts as live. + assert (second.live_calls, second.cached_calls, second.failed_calls) == (2, 0, 0) + for row in iter_calls(second.run_dir): + assert not row.cached + assert row.cached_rounds == 1 # round 0 replayed; its latency is not fresh + assert row.latency_replayed + assert row.trace is not None + assert row.trace.round_count == 3 + + third = await run_orchestrator(**kwargs) + assert len(seen) == 4 + assert (third.live_calls, third.cached_calls) == (0, 2) + assert _rows(third.run_dir) == _rows(second.run_dir) + assert all(r.cached and r.cached_rounds == 3 for r in iter_calls(third.run_dir)) + + async def test_a_truncated_round_is_cached_and_stays_flagged( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + caplog: pytest.LogCaptureFixture, + ) -> None: + _install_tool_fake( + monkeypatch, + finish_reason_for=lambda k: "length" if k == 0 else "stop", + ) + kwargs = self._kwargs(tmp_path, cache, Suite(examples=[_two_round_example()])) + await run_orchestrator(**kwargs) + + caplog.clear() + with caplog.at_level("WARNING", logger="evalshift_cli.runner.orchestrator"): + second = await run_orchestrator(**kwargs) + assert second.cached_calls == 2 + for row in iter_calls(second.run_dir): + assert row.finish_reason == "length" + assert row.truncated + assert any("truncated" in r.getMessage() for r in caplog.records) + + @pytest.mark.parametrize( + ("change", "expected_rounds"), + [ + # Anything round 0 is sent re-dispatches every round. + ({"tools": [_ROUND_TOOL.model_copy(update={"strict": True}), _ROUND_TOOL_2]}, 3), + ( + { + "tools": [ + _ROUND_TOOL.model_copy(update={"description": "Search it."}), + _ROUND_TOOL_2, + ] + }, + 3, + ), + ({"tools": [_ROUND_TOOL_2, _ROUND_TOOL]}, 3), + ({"generation_config": {"tool_choice": "required"}}, 3), + ({"generation_config": {"temperature": 0.7}}, 3), + ({"generation_config": {"parallel_tool_calls": False}}, 3), + ({"inputs": {"name": "Bea"}}, 3), + # A recorded fixture is only sent from the round after it. + ( + { + "tool_result_fixtures": [ + [ToolResultFixture(tool_name="search", result={"hits": 2})], + [ + ToolResultFixture(tool_name="fetch", result="edited text"), + ToolResultFixture(tool_name="fetch", error="not found"), + ], + ] + }, + 1, + ), + ], + ids=[ + "tool-strict", + "tool-description", + "tool-order", + "tool-choice", + "temperature", + "parallel-tool-calls", + "inputs", + "later-fixture", + ], + ) + async def test_a_change_to_what_is_sent_misses_exactly_the_affected_rounds( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + change: dict[str, Any], + expected_rounds: int, + ) -> None: + seen = _install_tool_fake(monkeypatch) + await run_orchestrator( + **self._kwargs(tmp_path, cache, Suite(examples=[_two_round_example()])) + ) + assert len(seen) == 6 + + changed = Suite(examples=[_two_round_example(**change)]) + await run_orchestrator(**self._kwargs(tmp_path, cache, changed)) + assert len(seen) == 6 + 2 * expected_rounds + + async def test_a_generation_config_warning_fires_once_per_call( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + caplog: pytest.LogCaptureFixture, + ) -> None: + # The config is translated once per work item and reused for the key + # and the dispatch; translating it twice doubled every warning. + _install_tool_fake(monkeypatch) + example = suite_example( + id="ex0", + inputs={"name": "A"}, + tools=[_ROUND_TOOL], + generation_config={"tool_choice": 42}, + ) + with caplog.at_level("WARNING", logger="evalshift_cli.runner.generation"): + await run_orchestrator(**self._kwargs(tmp_path, cache, Suite(examples=[example]))) + warnings = [r for r in caplog.records if "tool_choice not understood" in r.getMessage()] + assert len(warnings) == 2 # one per role + + async def test_a_max_tokens_override_is_a_miss( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + seen = _install_tool_fake(monkeypatch) + suite = Suite(examples=[_two_round_example()]) + await run_orchestrator(**self._kwargs(tmp_path, cache, suite)) + assert len(seen) == 6 + + capped = EvalShiftConfig( + prompts=[_round_config().prompts[0].model_copy(update={"max_tokens": 512})], + defaults=Defaults(concurrency=4, max_cost_usd=100.0), + ) + await run_orchestrator(**self._kwargs(tmp_path, cache, suite, config=capped)) + assert len(seen) == 12 + + async def test_a_row_without_a_trace_under_a_tool_key_is_redispatched( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + # A hit that cannot rebuild a ToolCompletionResult must not be served. + from sqlalchemy import text + + seen = _install_tool_fake(monkeypatch) + kwargs = self._kwargs(tmp_path, cache, Suite(examples=[_two_round_example()])) + await run_orchestrator(**kwargs) + assert await cache.count() == 6 + async with cache._engine.begin() as conn: + await conn.execute(text("UPDATE cached_calls SET trace_json = NULL")) + + second = await run_orchestrator(**kwargs) + assert len(seen) == 12 + assert (second.live_calls, second.cached_calls) == (2, 0) + + async def test_a_different_target_model_is_a_miss( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + seen = _install_tool_fake(monkeypatch) + await run_orchestrator(**self._kwargs(tmp_path, cache, _single_shot_tool_suite())) + second = await run_orchestrator( + **self._kwargs( + tmp_path, cache, _single_shot_tool_suite(), target_model="gemini-2.5-flash-lite" + ), + ) + assert len(seen) == 6 # only the new target's two examples went live + assert (second.live_calls, second.cached_calls) == (2, 2) + + async def test_each_sample_is_its_own_cached_tool_call( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + seen = _install_tool_fake(monkeypatch) + suite = Suite(examples=[suite_example(id="ex0", inputs={"name": "A"}, tools=[_ROUND_TOOL])]) + kwargs = self._kwargs(tmp_path, cache, suite, config=_tool_config(samples_per_example=2)) + + first = await run_orchestrator(**kwargs) + assert len(seen) == 4 # 2 roles x 2 samples, sample 1 not served from sample 0 + assert first.cached_calls == 0 + second = await run_orchestrator(**kwargs) + assert len(seen) == 4 + assert second.cached_calls == 4 + + async def test_tool_and_text_entries_never_collide( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + text_counter = _make_fake_client(monkeypatch) + seen = _install_tool_fake(monkeypatch) + plain = Suite(examples=[suite_example(id="ex0", inputs={"name": "A"})]) + with_tools = Suite( + examples=[suite_example(id="ex0", inputs={"name": "A"}, tools=[_ROUND_TOOL])] + ) + + await run_orchestrator(**self._kwargs(tmp_path, cache, plain)) + assert text_counter["calls"] == 2 + result = await run_orchestrator(**self._kwargs(tmp_path, cache, with_tools)) + assert len(seen) == 2 + assert result.cached_calls == 0 + assert all(r.trace is not None for r in iter_calls(result.run_dir)) + + async def test_resume_serves_pending_tool_items_from_the_cache( + self, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + cache: CacheStore, + ) -> None: + from datetime import datetime + + from evalshift_cli.runner import checkpoint as cp_mod + from evalshift_cli.runner.models import RunModels, RunState + + seen = _install_tool_fake(monkeypatch) + suite = Suite(examples=[_two_round_example()]) + kwargs = self._kwargs(tmp_path, cache, suite) + full = await run_orchestrator(**kwargs) + source_row = next(r for r in iter_calls(full.run_dir) if r.role == "source") + target_row = next(r for r in iter_calls(full.run_dir) if r.role == "target") + + # An interrupted run that finished only the source side. + run_dir = kwargs["runs_base"] / "r_20260601_dead00" + state = RunState( + run_id="r_20260601_dead00", + status="in_progress", + config_hash=cp_mod.compute_config_hash(kwargs["config"], str(kwargs["suite_path"])), + started_at=datetime(2026, 6, 1, tzinfo=UTC), + models=RunModels(source="gemini/gemini-2.5-flash", target="gemini/gemini-2.5-pro"), + prompt_ids=["agent"], + suite_path=str(kwargs["suite_path"]), + total_evaluations=2, + completed_evaluations=1, + ) + cp_mod.write_state(run_dir, state) + cp_mod.append_call(run_dir, source_row.model_copy(update={"run_id": state.run_id})) + + resumed = await run_orchestrator(**{**kwargs, "resume": True}) + assert resumed.run_dir == run_dir + assert len(seen) == 6 # the pending target item was served from cache + assert (resumed.live_calls, resumed.cached_calls) == (0, 1) + resumed_target = next(r for r in iter_calls(run_dir) if r.role == "target") + assert resumed_target.cached + assert resumed_target.model_dump( + exclude={"run_id", "cached", "cached_rounds"} + ) == target_row.model_dump(exclude={"run_id", "cached", "cached_rounds"}) diff --git a/tests/unit/test_reports.py b/tests/unit/test_reports.py index 48e7595..251e6e8 100644 --- a/tests/unit/test_reports.py +++ b/tests/unit/test_reports.py @@ -2006,6 +2006,22 @@ def test_run_deltas_are_none_when_the_source_side_measured_nothing(self) -> None assert _pct_delta(1.0, 2.0) == pytest.approx(100.0) assert _pct_delta(2.0, 1.0) == pytest.approx(-50.0) + def test_run_latency_delta_is_none_when_either_side_measured_nothing_live(self) -> None: + # A role whose every call was served (wholly or partly) from the cache + # has no live latency sample; its 0.0 mean is "unmeasured", so the + # header must not read it as a -100% latency change. + from evalshift_cli.reports.html import _run_latency_delta_pct + + def totals(src_live: float, tgt_live: float) -> dict[str, dict[str, float]]: + return { + "source": {"latency_ms_avg": 100.0 if src_live else 0.0, "live_calls": src_live}, + "target": {"latency_ms_avg": 150.0 if tgt_live else 0.0, "live_calls": tgt_live}, + } + + assert _run_latency_delta_pct(totals(2, 0)) is None + assert _run_latency_delta_pct(totals(0, 2)) is None + assert _run_latency_delta_pct(totals(2, 2)) == pytest.approx(50.0) + def test_verdict_panel_shows_the_outcome_split(self, tmp_path: Path) -> None: html = self._html(tmp_path, decision=True) assert "Equivalent 0.0%" in html @@ -2391,3 +2407,38 @@ def test_determinism_banner_suggests_repeated_sampling_only_when_n_is_one( html = render_html(build_report_payload(cwd / ".evalshift" / "runs" / run_id)) assert "Sampling is not controlled" in html assert "samples_per_example" not in html + + +def test_example_row_latency_is_incomparable_when_a_side_replayed_rounds() -> None: + from evalshift_cli.reports.json import _build_example_rows + from evalshift_cli.runner.models import Call + + def call(role: str, example_id: str, latency: int, cached_rounds: int = 0) -> Call: + return Call( + run_id="r_20260601_abc123", + prompt_id="p", + example_id=example_id, + model_id="m", + role=role, # type: ignore[arg-type] + latency_ms=latency, + cached_rounds=cached_rounds, + ) + + rows = _build_example_rows( + prompt_id="p", + calls=[ + call("source", "live", 100), + call("target", "live", 150), + call("source", "mixed", 100, cached_rounds=1), + call("target", "mixed", 150), + ], + records=[], + tags_by_example_id={}, + tool_evaluator_names=frozenset(), + examples_by_id={}, + ) + by_id = {r.example_id: r for r in rows} + assert by_id["live"].latency_comparable + assert by_id["live"].delta_latency_ms == 50 + assert not by_id["mixed"].latency_comparable + assert by_id["mixed"].delta_latency_ms == 0 diff --git a/tests/unit/test_runner_models.py b/tests/unit/test_runner_models.py index 2ce9f18..f4cd87c 100644 --- a/tests/unit/test_runner_models.py +++ b/tests/unit/test_runner_models.py @@ -349,3 +349,30 @@ def test_single_sample_run_is_returned_unchanged(self) -> None: self._call(example_id="a", role="target", sample_index=0), ] assert representative_calls(calls) == calls + + +class TestCallCachedRounds: + """``cached_rounds``: how many of a row's provider rounds were replayed from cache.""" + + _BASE = ( + '{"run_id": "r_20260601_abc123", "prompt_id": "p", "example_id": "e",' + ' "model_id": "m", "role": "source", "latency_ms": 40' + ) + + def test_rows_written_before_the_field_read_back_as_zero(self) -> None: + call = Call.model_validate_json(self._BASE + "}") + assert call.cached_rounds == 0 + assert not call.latency_replayed + + def test_a_fully_cached_row_replays_its_latency(self) -> None: + call = Call.model_validate_json(self._BASE + ', "cached": true}') + assert call.latency_replayed + + def test_a_partly_cached_row_is_not_cached_but_replays_latency(self) -> None: + call = Call.model_validate_json(self._BASE + ', "cached_rounds": 1}') + assert not call.cached + assert call.latency_replayed + + def test_negative_counts_are_rejected(self) -> None: + with pytest.raises(ValidationError): + Call.model_validate_json(self._BASE + ', "cached_rounds": -1}')