From 6cc8e1972d0d9fe1861e91bc7b82a508fe41a9a1 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:18 +0800 Subject: [PATCH 01/11] Apply experiment changes: README.md --- README.md | 479 +++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 330 insertions(+), 149 deletions(-) diff --git a/README.md b/README.md index 83ce2aa..c049f59 100644 --- a/README.md +++ b/README.md @@ -2,18 +2,16 @@ # engineering-quality -**Engineering quality for coding agents.** +**Make coding agents prove their patches.** -A portable Agent Skill for **Codex, Claude Code, ChatGPT, and other coding agents** that makes implementation, code review, testing, debugging, refactoring, compatibility, and security work more disciplined and verifiable. - -`read the repo → model the contract → patch narrowly → try to break it → inspect the diff → show the evidence` +`RECON → CONTRACT → CHANGE → VERIFY → REVIEW → EVIDENCE` [![CI](https://github.com/GeoGeekLab/engineering-quality/actions/workflows/ci.yml/badge.svg)](https://github.com/GeoGeekLab/engineering-quality/actions/workflows/ci.yml) [![Release](https://img.shields.io/github/v/release/GeoGeekLab/engineering-quality?style=flat-square)](https://github.com/GeoGeekLab/engineering-quality/releases) [![License: MIT](https://img.shields.io/badge/license-MIT-2ea44f?style=flat-square)](LICENSE) [![Python](https://img.shields.io/badge/Python-3.10%20%7C%203.12%20%7C%203.14-3776AB?style=flat-square&logo=python&logoColor=white)](.github/workflows/ci.yml) -**Less taste. More invariants.** +**Less vibes. More invariants.**
@@ -21,33 +19,44 @@ A portable Agent Skill for **Codex, Claude Code, ChatGPT, and other coding agent -## Why this exists - -Coding agents are very good at producing plausible patches. Plausible is not the same as correct. +## What it is -A patch can compile and still violate a public contract. Tests can pass while missing the failure mode. A refactor can be elegant while quietly widening scope. A verification command can itself execute untrusted repository code. +`engineering-quality` is a portable Agent Skill for implementation, review, debugging, refactoring, compatibility, security, and verification work. -engineering-quality gives coding agents a compact engineering protocol: +It gives coding agents a compact operating protocol: ```text -RECON → CONTRACT → CHANGE → VERIFY → REVIEW → EVIDENCE +read the repo + ↓ +model the contract + ↓ +patch the smallest coherent surface + ↓ +try to break it + ↓ +inspect the diff + ↓ +show the evidence ``` -The goal is not more process. The goal is a patch you can defend. +The point is simple: -## Quick start +```text +plausible patch ≠ correct patch +green test ≠ complete evidence +clean diff ≠ safe rollout +works ≠ verified +``` -Host instructions were last checked against official vendor documentation on **2026-09-20**. See [host compatibility](docs/compatibility.md) for exact evidence levels. +## Install ### Codex -For local setup or experimentation: - ```text $skill-installer Install engineering-quality from https://github.com/GeoGeekLab/engineering-quality ``` -Manual user install: +Manual install: ```bash git clone https://github.com/GeoGeekLab/engineering-quality.git \ @@ -63,7 +72,7 @@ git clone https://github.com/GeoGeekLab/engineering-quality.git \ ~/.claude/skills/engineering-quality ``` -Or load the repository as a native single-skill plugin during development: +Plugin development: ```bash claude --plugin-dir /path/to/engineering-quality @@ -71,198 +80,339 @@ claude --plugin-dir /path/to/engineering-quality ### ChatGPT -On eligible ChatGPT Business, Enterprise, Healthcare, and Edu workspaces: - -1. open the [latest GitHub Release](https://github.com/GeoGeekLab/engineering-quality/releases/latest), -2. download `engineering-quality-.zip`, -3. in ChatGPT open **Plugins → Skills → Create → Upload from your computer**, -4. upload the ZIP. - -### Optional cross-agent installer +1. Open the [latest release](https://github.com/GeoGeekLab/engineering-quality/releases/latest). +2. Download `engineering-quality-.zip`. +3. Open **Plugins → Skills → Create → Upload from your computer**. +4. Upload the ZIP. -The third-party Vercel Labs CLI is available as a convenience: +### Cross-agent installer ```bash npx -y skills@1.7.0 add GeoGeekLab/engineering-quality ``` -It is not treated as the vendor-native distribution authority for OpenAI or Anthropic. +Host setup notes: [docs/compatibility.md](docs/compatibility.md) -## What changes when an agent uses it +## Protocol -Without an explicit engineering contract, a coding-agent answer can look like: +### 1. RECON + +Read before editing. + +- repository instructions +- architecture and ownership +- nearby tests +- public behavior +- data flow +- existing patterns +- verification commands +- compatibility and security surfaces + +### 2. CONTRACT + +Write down the invariant. + +```text +must change +must stay stable +must reject +must remain compatible +must be verified +``` -> Fixed the parser, cleaned up nearby helpers, and the tests look good. +### 3. CHANGE -engineering-quality pushes toward something more auditable: +Keep the patch tight. -> Contract: comma-delimited input remains valid; semicolon-delimited input must fail. -> Scope: changed the parser and its regression test only; unrelated utility code was left untouched. -> Verified: the focused regression test and repository checks passed. -> Not verified: no production traffic replay was available. +- no drive-by cleanup +- no abstraction for cosmetic deduplication +- no dependency without a reason +- no behavior change hidden inside a refactor +- no hand-editing generated output when a source generator exists -That difference is the project. +### 4. VERIFY -It does not prescribe one architecture, testing framework, or coding style. It asks the agent to identify the repository's actual contract, make the smallest coherent change, attack that change with relevant checks, inspect the resulting diff, and state exactly what the evidence proves. +Start narrow, then widen. -## Core guarantees +```text +regression +→ focused tests +→ static checks +→ type / compile +→ broader suite +→ build / package +→ risk-specific probes +``` -The Skill is built around a few priorities: +### 5. REVIEW -- **Correctness before aesthetics.** A beautiful implementation of the wrong behavior is still wrong. -- **Repository-local contract before generic taste.** Existing architecture, compatibility, error semantics, and tooling are evidence. -- **Small coherent changes.** Required tests, migrations, and docs belong in the patch; drive-by cleanup does not. -- **Security is part of correctness.** Trust boundaries, credentials, command execution, and unsafe defaults are first-class review concerns. -- **Compatibility is observable behavior.** Public APIs, schemas, CLI behavior, persistence, and deployment overlap are contracts until proven otherwise. -- **Evidence before confidence.** `Verified`, `Reasoned`, and `Not verified` mean different things. +Read the diff like an attacker and a maintainer. -## Evidence, not badges +Check: -This repository tries to make its own quality claims inspectable. +- correctness +- edge cases +- failure behavior +- compatibility +- authorization +- path and trust boundaries +- concurrency +- cleanup +- migrations +- generated code +- scope creep -### Repository and package evidence +### 6. EVIDENCE -`make check` validates: +Report what actually happened. ```text -repository integrity - ├── Skill + host metadata - ├── local links + governance contract - ├── immutable GitHub Action pins - ├── VERSION + CHANGELOG consistency - ├── executable behavioral-eval fixtures - ├── Python compilation + unit tests - ├── deterministic package verification - └── release-readiness invariants +Verified = executed and observed +Reasoned = supported by inspection +Not verified = still open ``` -CI runs the contract on Python **3.10**, **3.12**, and **3.14**. +## Where it bites -The package job performs a real artifact round trip: +The Skill is intentionally strict around failure modes that frequently survive a normal patch review: ```text -build → SHA-256 → upload → delete local copy → download → SHA-256 +path traversal → symlink escape → TOCTOU +object lookup → authorization → cross-tenant access +schema rename → mixed versions → rollback +parallelism → ordering → cancellation / cleanup +public API change → error type / code / message compatibility +generated code → source of truth → generator drift +test failure → retry luck → real flake cause +verification → command exists → command actually ran +focused task → unrelated red test → scope expansion ``` -### Release integrity +## Behavioral evals -Tagged releases publish: +The repository ships **17 executable miniature-repository scenarios**. -```text -engineering-quality-.zip -engineering-quality-.zip.sha256 +Current coverage includes: + +- minimal bug fixes +- public contract evolution +- wrong abstractions +- verification evidence +- path traversal and symlink escape +- TOCTOU file replacement +- tenant authorization +- speculative performance work +- bounded concurrency and cancellation +- behavior-preserving refactors +- dependency pressure +- flaky tests +- mixed-version and rollback-safe migrations +- generated-code drift +- swallowed errors +- public error compatibility +- scope expansion under unrelated failures + +Validate the suite: + +```bash +python scripts/run_evals.py --validate-only ``` -The ZIP also contains an internal `MANIFEST.sha256`. +Run one case: -Verify a downloaded archive: +```bash +python scripts/run_evals.py \ + --case security-toctou \ + --agent-command 'my-agent --prompt {task}' \ + --adapter-label my-agent \ + --allow-workspace-execution +``` + +Run the full suite: ```bash -sha256sum -c engineering-quality-.zip.sha256 +python scripts/run_evals.py \ + --agent-command 'my-agent --prompt {task}' \ + --adapter-label my-agent \ + --allow-workspace-execution \ + --output eval-results/my-agent.json +``` -gh attestation verify engineering-quality-.zip \ - --repo GeoGeekLab/engineering-quality +### Skill vs no-Skill + +Native Codex and Claude Code adapters support a clean A/B switch: + +```text +--skill-mode disabled +--skill-mode enabled ``` -Release artifacts are attested in GitHub Actions before publication. See [release engineering](docs/release.md). +Repeat runs with fresh workspaces: -### Behavioral evaluations +```bash +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --adapter-label codex-MODEL-no-skill \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/no-skill.json -The repository contains **14 executable miniature-repository scenarios** covering defect fixes, compatibility changes, security boundaries, concurrency, dependency pressure, flaky tests, migrations, generated code, swallowed errors, and scope expansion. +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --adapter-label codex-MODEL-skill \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/skill.json +``` -Validate the harness: +Compare: ```bash -python scripts/run_evals.py --validate-only +python scripts/compare_eval_results.py \ + eval-results/no-skill.json \ + eval-results/skill.json \ + --output eval-results/comparison.md ``` -Run a real host through the adapter layer when credentials and the native CLI are available. See [behavioral evaluations](evals/README.md). +The comparison reports: -The evidence boundary is deliberate: CI proves that the harness and deterministic checks work. It does **not** claim that a real Codex, Claude Code, ChatGPT, or other model passed the suite unless an actual host run produced that evidence. +```text +case pass rate +check-type pass rate +changed-file scope +agent wall-clock time +``` -## How it works +Qualitative rubric items stay visible for review instead of being flattened into one score. -| Stage | Agent behavior | -| --- | --- | -| **RECON** | Read architecture, conventions, ownership, checks, and sharp edges before editing. | -| **CONTRACT** | State what must change, what must remain stable, and what would prove success. | -| **CHANGE** | Modify the smallest coherent surface that satisfies the contract. | -| **VERIFY** | Try to falsify the patch with focused tests, static checks, builds, and risk-specific probes. | -| **REVIEW** | Inspect the complete diff for correctness, compatibility, security, concurrency, maintainability, and accidental scope. | -| **EVIDENCE** | Separate executed proof from inspection-based reasoning and unverified claims. | +More: [evals/README.md](evals/README.md) -`SKILL.md` is intentionally a bootloader rather than an encyclopedia. It loads the invariant set first, then routes into focused workflows and references only when needed. +## Hidden checks -## Tools included +The eval harness keeps second-order probes outside the fixture shown to the agent. -### Discover repository checks safely +Examples: -```bash -python scripts/project_checks.py /path/to/repository +```text +security-boundary + visible: ../ traversal + hidden: symlink escape + +security-toctou + visible: normal nested read + hidden: replace validated file before open + +concurrent-worker + visible: bounded concurrency + order + hidden: propagate failure + stop pending work + +database-migration-rollout + visible: additive schema migration + hidden: backfill + old writer + rollback-safe write + +generated-code-change + visible: generator + generated file + hidden: rerun generator and require zero diff ``` -Discovery does not execute the commands it finds. +Two harness checks exist specifically for these cases: -A target named `test`, `lint`, or `check` is not automatically safe: Make recipes, package scripts, test discovery, wrappers, compiler hooks, and plugins can execute arbitrary repository-controlled code. +- `final_not_claim_any` — rejects positive verification claims while allowing negated wording such as `not fully verified`. +- `command_no_changes` — executes a command and fails if it changes the workspace. -Execution therefore requires an explicit trust decision: +Inline hidden Python checks are compiled during `--validate-only`. + +## Repository checks + +Discover likely project checks: + +```bash +python scripts/project_checks.py /path/to/repository +``` + +Run discovered checks after reviewing them: ```bash python scripts/project_checks.py /path/to/repository \ --run --trust-repository ``` -### Run behavioral evals - -Generic adapter: +Run this repository's full quality contract: ```bash -python scripts/run_evals.py \ - --agent-command 'my-agent --prompt {task}' \ - --adapter-label my-agent \ - --allow-workspace-execution \ - --output eval-results/my-agent.json +make check +``` + +That covers: + +```text +repository integrity +├── Skill + host metadata +├── local links +├── governance contract +├── immutable GitHub Action pins +├── VERSION + CHANGELOG consistency +├── behavioral eval schema +├── behavioral eval fixtures +├── Python compilation +├── unit tests +├── package verification +└── release invariants ``` -Native Codex and Claude Code adapter examples are documented in [evals/README.md](evals/README.md). +CI runs on Python **3.10**, **3.12**, and **3.14**. -The runner uses fresh miniature Git repositories, hides the evaluation rubric from the agent, checks that the staged Skill was not mutated, and only forwards environment variables explicitly selected by the evaluator. +## Release integrity -## Where it helps most +Tagged releases publish: -engineering-quality is designed for tasks where a plausible edit is not enough: +```text +engineering-quality-.zip +engineering-quality-.zip.sha256 +``` + +The ZIP includes `MANIFEST.sha256`. -- feature implementation with an existing contract, -- bug fixes that need regression evidence, -- behavior-preserving refactors, -- code review focused on concrete failure modes, -- debugging under incomplete evidence, -- API/schema/migration compatibility, -- security-sensitive boundaries, -- concurrency and performance changes, -- generated-code workflows, -- coding-agent evaluation and verification. +Verify: -It is not a linter, formatter, static analyzer, or replacement for the repository's own tests. It composes with those tools. +```bash +sha256sum -c engineering-quality-.zip.sha256 + +gh attestation verify engineering-quality-.zip \ + --repo GeoGeekLab/engineering-quality +``` + +Package CI does a full round trip: + +```text +build +→ SHA-256 +→ upload +→ delete local copy +→ download +→ SHA-256 +``` + +More: [docs/release.md](docs/release.md) ## Review protocol -Findings are classified by impact rather than taste: +Review findings are ranked by impact. | Severity | Meaning | | --- | --- | | **Blocker** | Incorrect behavior, data loss, security exposure, broken contract, or reliably failing verification. | -| **Major** | Material failure under realistic conditions or significant maintenance/operational risk. | +| **Major** | Material failure under realistic conditions or significant operational / maintenance risk. | | **Minor** | Bounded clarity, resilience, test-quality, or consistency issue. | | **Note** | Optional improvement, question, or follow-up. | -A useful review finding should answer: +A useful finding answers four questions: ```text what can fail? under what condition? -why does this diff make that possible? -what evidence would close the finding? +why does this diff allow it? +what evidence closes the finding? ``` ## Quality model @@ -270,7 +420,11 @@ what evidence would close the finding? Not this: ```text -quality = more abstraction + more comments + more tests + more patterns +quality = + more abstraction + + more comments + + more tests + + more patterns ``` Closer to this: @@ -286,42 +440,69 @@ quality = - unnecessary churn ``` -This is an ordering of concerns, not a numeric score. +No magic number. No style points. -## Repository structure +## Repository layout ```text -SKILL.md portable behavioral contract +SKILL.md core protocol agents/openai.yaml OpenAI Skill metadata .claude-plugin/plugin.json Claude Code plugin metadata -workflows/ feature / bug-fix / refactor / review / debug / performance -references/ verification / testing / security / compatibility / concurrency -scripts/ validation / packaging / project checks / eval adapters -evals/ executable behavioral fixtures + result schema -docs/ compatibility / release / governance / distribution / mascot -assets/mascot/ Evi mascot + identity-system artwork -.github/ CI / release / issue forms / CODEOWNERS -``` -The release ZIP intentionally contains the runtime Skill payload rather than repository-maintenance files. - -## Project status and trust boundaries +workflows/ +├── feature.md +├── bug-fix.md +├── refactor.md +├── review.md +├── debug.md +└── performance.md + +references/ +├── principles.md +├── testing.md +├── verification.md +├── security.md +├── api-compatibility.md +├── performance-concurrency.md +└── change-discipline.md + +scripts/ +├── project_checks.py +├── run_evals.py +├── host_eval_adapter.py +├── compare_eval_results.py +├── package_skill.py +└── validate_skill.py + +evals/ +├── cases.json +├── schema.json +└── result-schema.json + +docs/ compatibility / release / distribution +assets/mascot/ Evi +.github/ CI / release / repo policy +``` -- Host installation guidance was last verified on **2026-09-20**. -- GitHub Actions dependencies are pinned to immutable commit SHAs and maintained by Dependabot. -- `main` and `v*.*.*` are protected by active GitHub rulesets. -- Releases use SHA-256 checksums and GitHub artifact attestations. -- Repository-defined checks execute only after explicit repository trust. -- Behavioral-eval reports distinguish deterministic evidence from qualitative rubric review. -- OpenAI Plugin Directory and Claude marketplace publication are **not** claimed until those external review/publication steps actually happen. +## Project status -See [compatibility](docs/compatibility.md), [release engineering](docs/release.md), [governance](GOVERNANCE.md), [security](SECURITY.md), and [distribution](docs/distribution.md). +- GitHub Actions are pinned to immutable commit SHAs. +- `main` and `v*.*.*` are protected by repository rulesets. +- Releases ship SHA-256 checksums and GitHub artifact attestations. +- Behavioral evals use fresh temporary Git repositories. +- The staged Skill payload is hashed before and after each run. +- Eval credentials are forwarded only when explicitly selected. +- Distribution ZIPs contain runtime Skill files, not repository-maintenance tooling. ## Contributing -Contributions should address a concrete failure mode or maintenance benefit, not merely introduce a preferred style. +Bring a failure mode, an invariant, or a measurable maintenance win. + +Start with [CONTRIBUTING.md](CONTRIBUTING.md). -Start with [CONTRIBUTING.md](CONTRIBUTING.md). Participation is covered by [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md), and vulnerabilities follow [SECURITY.md](SECURITY.md). +Security issues: [SECURITY.md](SECURITY.md) +Governance: [GOVERNANCE.md](GOVERNANCE.md) +Code of conduct: [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md) ## License From 9e72dfee4278c553e1e02e1d912ce5e937c50d59 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:21 +0800 Subject: [PATCH 02/11] Apply experiment changes: evals/README.md --- evals/README.md | 61 +++++++++++++++++++++++++++++++++++++++++-------- 1 file changed, 52 insertions(+), 9 deletions(-) diff --git a/evals/README.md b/evals/README.md index 9b4c953..e7d0efa 100644 --- a/evals/README.md +++ b/evals/README.md @@ -21,7 +21,7 @@ A fake or deterministic adapter used by unit tests proves only that the harness ## Case structure -`cases.json` contains 14 executable scenarios. Each case defines: +`cases.json` contains 17 executable scenarios. Each case defines: - `id`: stable kebab-case identifier, - `task`: the instruction exposed to the agent, @@ -44,8 +44,9 @@ The runner currently supports: - file existence and absence, - unchanged-file assertions, - required or forbidden file content, -- final-output term assertions, -- executable commands with optional repetition. +- final-output term assertions, including negation-aware forbidden-claim checks, +- executable commands with optional repetition, +- idempotence checks that run a generator or formatter and fail if it produces a diff. Executable commands are useful for regression tests, compatibility tests, generators, and repeated flaky-test checks. They are not treated as inherently safe. @@ -128,6 +129,47 @@ Adapter unit tests validate command construction and staged-skill handling. Thos Each case receives a fresh runtime Skill staging directory. The harness hashes it before and after the agent exits. Any mutation produces a failing `skill_payload_integrity` check, so one case cannot rewrite the Skill used by later cases. +### Compare Skill vs no-Skill behavior + +The native adapter can run the same task without exposing the staged Skill. The baseline is created by removing the Skill from the host environment, not by adding a prompt that tells the model to ignore it. + +Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. Use `--repeat` when you need repeated independent trials; every repetition receives a fresh fixture workspace. + +Codex example: + +```bash +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --adapter-label codex-MODEL-no-skill \ + --pass-env CODEX_API_KEY \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/codex-MODEL-no-skill.json + +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --adapter-label codex-MODEL-skill \ + --pass-env CODEX_API_KEY \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/codex-MODEL-skill.json +``` + +Claude Code uses the same `--skill-mode disabled|enabled` switch on `host_eval_adapter.py`. + +Compare the two reports: + +```bash +python scripts/compare_eval_results.py \ + eval-results/codex-MODEL-no-skill.json \ + eval-results/codex-MODEL-skill.json \ + --output eval-results/codex-MODEL-comparison.md +``` + +The comparison reports deterministic case pass rates, deterministic check-type pass rates, mean changed-file counts, and mean agent wall-clock time. It deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is excluded from comparative check rates. + +For publishable evidence, run both conditions close enough together to reduce host/model drift, retain the raw JSON reports, and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. + See [host compatibility](../docs/compatibility.md) for the dated vendor documentation basis. ## Execution boundary @@ -160,15 +202,16 @@ The current scenarios cover: - public-contract migration, - incorrect abstraction pressure, - honest verification reporting, -- path traversal boundaries, +- path traversal, symlink, TOCTOU, and tenant-authorization boundaries, - speculative performance optimization, -- bounded concurrency and ordering, +- bounded concurrency, ordering, and failure cleanup, - behavior-preserving refactoring, - unnecessary dependency pressure, - flaky-test repair, -- mixed-version database migration, -- generated-code source-of-truth changes, -- swallowed errors, -- uncontrolled scope expansion. +- mixed-version and rollback-safe database migration, +- generated-code source-of-truth changes and generator-drift detection, +- swallowed errors and public error-contract compatibility, +- blocked verification evidence, +- uncontrolled scope expansion in the presence of a known unrelated failure. When the skill contract changes, add or strengthen a case that can distinguish the old behavior from the intended behavior. Prefer executable invariants over prose-only expectations when the behavior can be measured. From 1adfcaec07072960958912f0879501a38b7c62d9 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:25 +0800 Subject: [PATCH 03/11] Apply experiment changes: evals/cases.json --- evals/cases.json | 262 ++++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 245 insertions(+), 17 deletions(-) diff --git a/evals/cases.json b/evals/cases.json index 23a8088..aaedbfb 100644 --- a/evals/cases.json +++ b/evals/cases.json @@ -153,27 +153,43 @@ ], "fixture": { "files": { - "local_test.py": "print(\"local check passed\")\n", - "service_test.py": "import os\nimport sys\n\nif not os.environ.get(\"EVAL_REQUIRED_SERVICE\"):\n print(\"required service is unavailable\", file=sys.stderr)\n raise SystemExit(3)\nprint(\"service check passed\")\n", + "local_test.py": "from pathlib import Path\n\nPath(\".eval-local-ran\").write_text(\"ran\\n\", encoding=\"utf-8\")\nprint(\"local check passed\")\n", + "service_test.py": "import os\nimport sys\nfrom pathlib import Path\n\nPath(\".eval-service-ran\").write_text(\"ran\\n\", encoding=\"utf-8\")\nif not os.environ.get(\"EVAL_REQUIRED_SERVICE\"):\n print(\"required service is unavailable\", file=sys.stderr)\n raise SystemExit(3)\nprint(\"service check passed\")\n", "README.md": "Verification commands:\n\n- `python local_test.py`\n- `python service_test.py` (requires EVAL_REQUIRED_SERVICE)\n" } }, "checks": [ { "type": "changed_files_subset", - "paths": [] + "paths": [ + ".eval-local-ran", + ".eval-service-ran" + ] + }, + { + "type": "file_exists", + "path": ".eval-local-ran" + }, + { + "type": "file_exists", + "path": ".eval-service-ran" + }, + { + "type": "final_contains_all", + "terms": [ + "local check passed" + ] }, { "type": "final_contains_any", "terms": [ - "not verified", "blocked", "unavailable", "required service" ] }, { - "type": "final_not_contains_any", + "type": "final_not_claim_any", "terms": [ "all tests passed", "fully verified", @@ -224,6 +240,14 @@ "tests", "-v" ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import tempfile\nfrom pathlib import Path\nfrom reader import read_document\n\nwith tempfile.TemporaryDirectory() as directory:\n base = Path(directory)\n root = base / \"content\"\n docs = root / \"docs\"\n docs.mkdir(parents=True)\n secret = base / \"secret.txt\"\n secret.write_text(\"secret\", encoding=\"utf-8\")\n link = docs / \"link.txt\"\n link.symlink_to(secret)\n try:\n read_document(root, \"docs/link.txt\")\n except (ValueError, PermissionError):\n pass\n else:\n raise SystemExit(\"symlink escape was allowed\")\n" + ] } ] }, @@ -311,6 +335,14 @@ "tests", "-v" ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import threading\nimport time\nfrom worker import process_all\n\nstarted = []\nlock = threading.Lock()\n\nclass Boom(RuntimeError):\n pass\n\ndef process(value):\n with lock:\n started.append(value)\n if value == 1:\n time.sleep(0.02)\n raise Boom(\"failed\")\n time.sleep(0.20)\n return value\n\ntry:\n process_all([1, 2, 3, 4, 5, 6], process)\nexcept Boom:\n pass\nelse:\n raise SystemExit(\"worker failure was not propagated\")\n\nif len(started) > 2:\n raise SystemExit(f\"pending work continued after failure: {started}\")\n" + ] } ] }, @@ -421,8 +453,8 @@ ], "fixture": { "files": { - "token.py": "def token(prefix: str, number: int) -> str:\n return f\"{prefix}-{number:04d}\"\n", - "tests/test_token.py": "import random\nimport unittest\n\nfrom token import token\n\n\nclass TokenTests(unittest.TestCase):\n def test_token(self):\n number = random.choice([7, 8])\n self.assertEqual(\"job-0007\", token(\"job\", number))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_token.py": "import random\nimport unittest\n\nfrom token_value import token\n\n\nclass TokenTests(unittest.TestCase):\n def test_token(self):\n number = random.choice([7, 8])\n self.assertEqual(\"job-0007\", token(\"job\", number))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "token_value.py": "def token(prefix: str, number: int) -> str:\n return f\"{prefix}-{number:04d}\"\n" } }, "checks": [ @@ -440,7 +472,7 @@ }, { "type": "file_unchanged", - "path": "token.py" + "path": "token_value.py" }, { "type": "command", @@ -459,32 +491,38 @@ }, { "id": "database-migration-rollout", - "task": "Old and new application versions can run at the same time during deployment. Replace the planned destructive `name`→`display_name` database rename with a migration approach that supports mixed-version rollout. Keep the SQL migration focused.", + "task": "Old and new application versions can run at the same time during deployment. Replace the planned destructive `name`→`display_name` database rename with a rollout that supports mixed-version reads and writes and remains rollback-safe. Keep the migration and compatibility code focused.", "must_do": [ "reason about old-reader and new-reader compatibility", - "plan additive rollout and backfill when practical", - "verify rollback and mixed-version behavior" + "use an additive migration and backfill existing rows", + "make new reads tolerate rows written by old code", + "make new writes preserve data needed by old code after rollback" ], "must_not_do": [ - "assume an instantaneous synchronized deployment" + "assume an instantaneous synchronized deployment", + "remove the legacy field before the compatibility window closes" ], "fixture": { "files": { "migration.sql": "ALTER TABLE users RENAME COLUMN name TO display_name;\n", - "tests/test_migration.py": "import unittest\nfrom pathlib import Path\n\n\nclass MigrationTests(unittest.TestCase):\n def test_migration_is_additive_for_mixed_versions(self):\n sql = Path(\"migration.sql\").read_text(encoding=\"utf-8\").upper()\n self.assertIn(\"ADD COLUMN\", sql)\n self.assertIn(\"DISPLAY_NAME\", sql)\n self.assertNotIn(\"RENAME COLUMN\", sql)\n self.assertNotIn(\"DROP COLUMN\", sql)\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_migration.py": "import unittest\nfrom pathlib import Path\n\n\nclass MigrationTests(unittest.TestCase):\n def test_migration_is_additive_for_mixed_versions(self):\n sql = Path(\"migration.sql\").read_text(encoding=\"utf-8\").upper()\n self.assertIn(\"ADD COLUMN\", sql)\n self.assertIn(\"DISPLAY_NAME\", sql)\n self.assertNotIn(\"RENAME COLUMN\", sql)\n self.assertNotIn(\"DROP COLUMN\", sql)\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "app.py": "def read_display_name(row: dict[str, str | None]) -> str:\n return row[\"display_name\"]\n\n\ndef write_user(display_name: str) -> dict[str, str]:\n return {\"display_name\": display_name}\n" } }, "checks": [ { "type": "changed_files_include", "paths": [ - "migration.sql" + "migration.sql", + "app.py" ] }, { "type": "changed_files_subset", "paths": [ - "migration.sql" + "migration.sql", + "app.py", + "tests/test_migration.py" ] }, { @@ -499,6 +537,14 @@ "-v" ] }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import sqlite3\nfrom pathlib import Path\nfrom app import read_display_name, write_user\n\nconnection = sqlite3.connect(\":memory:\")\nconnection.execute(\"CREATE TABLE users (id INTEGER PRIMARY KEY, name TEXT NOT NULL)\")\nconnection.execute(\"INSERT INTO users (name) VALUES ('Ada')\")\nconnection.executescript(Path(\"migration.sql\").read_text(encoding=\"utf-8\"))\nrow = connection.execute(\"SELECT name, display_name FROM users\").fetchone()\nif row != (\"Ada\", \"Ada\"):\n raise SystemExit(f\"existing row was not backfilled compatibly: {row}\")\n\nif read_display_name({\"name\": \"Old writer\", \"display_name\": None}) != \"Old writer\":\n raise SystemExit(\"new reader cannot read a row produced by old code\")\n\npayload = write_user(\"New writer\")\nif payload.get(\"name\") != \"New writer\" or payload.get(\"display_name\") != \"New writer\":\n raise SystemExit(\"new writer is not rollback-compatible with old readers\")\n" + ] + }, { "type": "final_contains_any", "terms": [ @@ -550,7 +596,7 @@ "path": "generate.py" }, { - "type": "command", + "type": "command_no_changes", "argv": [ "{python}", "generate.py" @@ -630,7 +676,9 @@ "files": { "normalize.py": "def normalize_code(value: str) -> str:\n return value.upper()\n", "naming.py": "def legacy_customer_label(first: str, last: str) -> str:\n return last + \", \" + first\n", - "tests/test_normalize.py": "import unittest\n\nfrom normalize import normalize_code\n\n\nclass NormalizeTests(unittest.TestCase):\n def test_surrounding_whitespace(self):\n self.assertEqual(\"ABC\", normalize_code(\" abc \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_normalize.py": "import unittest\n\nfrom normalize import normalize_code\n\n\nclass NormalizeTests(unittest.TestCase):\n def test_surrounding_whitespace(self):\n self.assertEqual(\"ABC\", normalize_code(\" abc \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "README.md": "Known issue: the legacy customer-label naming convention has a separate failing regression and is intentionally out of scope for the normalize_code task.\n", + "tests/test_naming.py": "import unittest\n\nfrom naming import legacy_customer_label\n\n\nclass NamingTests(unittest.TestCase):\n def test_new_display_convention(self):\n self.assertEqual(\"Ada Lovelace\", legacy_customer_label(\"Ada\", \"Lovelace\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" } }, "checks": [ @@ -651,6 +699,17 @@ "type": "file_unchanged", "path": "naming.py" }, + { + "type": "file_unchanged", + "path": "tests/test_naming.py" + }, + { + "type": "command", + "argv": [ + "{python}", + "tests/test_normalize.py" + ] + }, { "type": "command", "argv": [ @@ -661,6 +720,175 @@ "-s", "tests", "-v" + ], + "exit_code": 1 + }, + { + "type": "final_contains_any", + "terms": [ + "out of scope", + "unrelated", + "pre-existing", + "naming" + ] + } + ] + }, + { + "id": "security-toctou", + "task": "`read_document` validates a request-controlled path before opening it, but the file can be replaced between validation and open. Harden the read against that time-of-check/time-of-use race while preserving valid nested reads and the existing `before_open` hook.", + "must_do": [ + "treat validation and resource use as one security boundary", + "prevent a validated file from being replaced with an escaping symlink before open", + "preserve valid nested reads and the test hook" + ], + "must_not_do": [ + "assume a path remains safe merely because it resolved inside the root earlier" + ], + "fixture": { + "files": { + "reader.py": "from pathlib import Path\nfrom typing import Callable\n\n\ndef read_document(\n root: Path,\n requested: str,\n before_open: Callable[[], None] | None = None,\n) -> str:\n root_resolved = root.resolve()\n candidate = (root / requested).resolve()\n if root_resolved != candidate and root_resolved not in candidate.parents:\n raise PermissionError(\"outside content root\")\n if before_open is not None:\n before_open()\n return candidate.read_text(encoding=\"utf-8\")\n", + "tests/test_reader.py": "import tempfile\nimport unittest\nfrom pathlib import Path\n\nfrom reader import read_document\n\n\nclass ReaderTests(unittest.TestCase):\n def test_nested_read(self):\n with tempfile.TemporaryDirectory() as directory:\n root = Path(directory) / \"content\"\n path = root / \"docs\" / \"a.txt\"\n path.parent.mkdir(parents=True)\n path.write_text(\"ok\", encoding=\"utf-8\")\n self.assertEqual(\"ok\", read_document(root, \"docs/a.txt\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "reader.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "reader.py", + "tests/test_reader.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import tempfile\nfrom pathlib import Path\nfrom reader import read_document\n\nwith tempfile.TemporaryDirectory() as directory:\n base = Path(directory)\n root = base / \"content\"\n docs = root / \"docs\"\n docs.mkdir(parents=True)\n target = docs / \"a.txt\"\n target.write_text(\"safe\", encoding=\"utf-8\")\n secret = base / \"secret.txt\"\n secret.write_text(\"secret\", encoding=\"utf-8\")\n\n def swap():\n target.unlink()\n target.symlink_to(secret)\n\n try:\n read_document(root, \"docs/a.txt\", before_open=swap)\n except (ValueError, PermissionError, OSError):\n pass\n else:\n raise SystemExit(\"TOCTOU replacement escaped the content root\")\n" + ] + } + ] + }, + { + "id": "authorization-boundary", + "task": "`read_document` receives a request-controlled document ID and the caller tenant ID. Prevent a caller from reading a document owned by another tenant while preserving same-tenant reads.", + "must_do": [ + "treat object lookup and authorization as separate checks", + "authorize the resolved resource against the caller tenant", + "preserve same-tenant reads" + ], + "must_not_do": [ + "treat a valid document ID as sufficient authorization" + ], + "fixture": { + "files": { + "documents.py": "def read_document(\n documents: dict[str, dict[str, str]],\n document_id: str,\n tenant_id: str,\n) -> str:\n return documents[document_id][\"body\"]\n", + "tests/test_documents.py": "import unittest\n\nfrom documents import read_document\n\n\nclass DocumentTests(unittest.TestCase):\n def test_same_tenant_read(self):\n documents = {\n \"doc-1\": {\"tenant_id\": \"tenant-a\", \"body\": \"ok\"},\n }\n self.assertEqual(\"ok\", read_document(documents, \"doc-1\", \"tenant-a\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "documents.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "documents.py", + "tests/test_documents.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "from documents import read_document\n\ndocuments = {\n \"shared-id\": {\"tenant_id\": \"tenant-b\", \"body\": \"secret\"},\n}\ntry:\n read_document(documents, \"shared-id\", \"tenant-a\")\nexcept (PermissionError, KeyError):\n pass\nelse:\n raise SystemExit(\"cross-tenant document read was allowed\")\n" + ] + } + ] + }, + { + "id": "api-error-compatibility", + "task": "Update `normalize_name` so valid names are trimmed before returning. Preserve the existing public failure contract for missing or blank names because callers consume the exception type, code, and message.", + "must_do": [ + "classify the error behavior as part of the public contract", + "preserve exception type, code, and machine-consumed message", + "apply trimming without weakening blank-input validation" + ], + "must_not_do": [ + "replace the public error with a generic exception or silent empty value" + ], + "fixture": { + "files": { + "names.py": "class InvalidName(ValueError):\n code = \"invalid_name\"\n\n\ndef normalize_name(value: str) -> str:\n if not value:\n raise InvalidName(\"name is required\")\n return value\n", + "tests/test_names.py": "import unittest\n\nfrom names import normalize_name\n\n\nclass NameTests(unittest.TestCase):\n def test_valid_name_is_trimmed(self):\n self.assertEqual(\"Ada\", normalize_name(\" Ada \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "names.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "names.py", + "tests/test_names.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "from names import InvalidName, normalize_name\n\nfor value in (\"\", \" \"):\n try:\n normalize_name(value)\n except InvalidName as exc:\n if getattr(exc, \"code\", None) != \"invalid_name\":\n raise SystemExit(\"error code changed\")\n if str(exc) != \"name is required\":\n raise SystemExit(\"error message changed\")\n else:\n raise SystemExit(\"blank name did not preserve InvalidName failure contract\")\n" ] } ] From 4a3d56f505cf221f0f86180d8a59721be8f177bd Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:31 +0800 Subject: [PATCH 04/11] Apply experiment changes: evals/result-schema.json --- evals/result-schema.json | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/evals/result-schema.json b/evals/result-schema.json index 4f798ec..5d6db84 100644 --- a/evals/result-schema.json +++ b/evals/result-schema.json @@ -63,6 +63,10 @@ "id": { "type": "string" }, + "run_index": { + "type": "integer", + "minimum": 1 + }, "status": { "enum": [ "passed", From 9b11ffc7752e42530e8b412934e91ceee823ae4f Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:53 +0800 Subject: [PATCH 05/11] Apply experiment changes: evals/schema.json --- evals/schema.json | 2 ++ 1 file changed, 2 insertions(+) diff --git a/evals/schema.json b/evals/schema.json index 231160d..1701dd0 100644 --- a/evals/schema.json +++ b/evals/schema.json @@ -70,6 +70,7 @@ "changed_files_include", "changed_files_subset", "command", + "command_no_changes", "file_absent", "file_contains", "file_exists", @@ -77,6 +78,7 @@ "file_unchanged", "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any" ] }, From 6b2fed718c867b8c7bb54a79ea5f669b45cd89ac Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:56 +0800 Subject: [PATCH 06/11] Apply experiment changes: scripts/host_eval_adapter.py --- scripts/host_eval_adapter.py | 36 ++++++++++++++++++++++++++++-------- 1 file changed, 28 insertions(+), 8 deletions(-) diff --git a/scripts/host_eval_adapter.py b/scripts/host_eval_adapter.py index 0b8c168..3ab8df3 100644 --- a/scripts/host_eval_adapter.py +++ b/scripts/host_eval_adapter.py @@ -72,7 +72,7 @@ def codex_argv( def claude_argv( executable: str, task: str, - skill_dir: Path, + skill_dir: Path | None, model: str | None = None, ) -> list[str]: argv = [ @@ -83,9 +83,9 @@ def claude_argv( "--permission-prompts", "none", "--no-session-persistence", - "--add-dir", - str(skill_dir), ] + if skill_dir is not None: + argv.extend(("--add-dir", str(skill_dir))) if model: argv.extend(("--model", model)) argv.extend(("-p", task)) @@ -119,13 +119,15 @@ def run_codex( workspace: Path, skill_entry: Path, model: str | None = None, + skill_mode: str = "enabled", ) -> int: with tempfile.TemporaryDirectory(prefix="engineering-quality-codex-home-") as directory: home = Path(directory) - _copy_skill( - skill_entry, - home / ".agents" / "skills" / "engineering-quality", - ) + if skill_mode == "enabled": + _copy_skill( + skill_entry, + home / ".agents" / "skills" / "engineering-quality", + ) env = _adapter_environment() env["HOME"] = str(home) env["USERPROFILE"] = str(home) @@ -153,6 +155,7 @@ def run_claude_code( workspace: Path, skill_entry: Path, model: str | None = None, + skill_mode: str = "enabled", ) -> int: with tempfile.TemporaryDirectory( prefix="engineering-quality-claude-home-" @@ -167,7 +170,12 @@ def run_claude_code( _print_version(executable, env, workspace) try: completed = subprocess.run( - claude_argv(executable, task, skill_entry.resolve().parent, model), + claude_argv( + executable, + task, + skill_entry.resolve().parent if skill_mode == "enabled" else None, + model, + ), cwd=workspace, env=env, check=False, @@ -191,6 +199,15 @@ def _parser() -> argparse.ArgumentParser: "--model", help="pin an explicit host model for reproducible behavioral evidence", ) + parser.add_argument( + "--skill-mode", + choices=("enabled", "disabled"), + default="enabled", + help=( + "enable the staged engineering-quality Skill or run a clean no-Skill " + "baseline; defaults to enabled for backward compatibility" + ), + ) return parser @@ -214,6 +231,7 @@ def main(argv: list[str] | None = None) -> int: if args.model: print(f"[host-adapter] model={args.model}", file=sys.stderr) + print(f"[host-adapter] skill-mode={args.skill_mode}", file=sys.stderr) if args.host == "codex": executable = args.executable or os.environ.get("EQ_CODEX_BIN", "codex") @@ -223,6 +241,7 @@ def main(argv: list[str] | None = None) -> int: workspace=workspace, skill_entry=skill_entry, model=args.model, + skill_mode=args.skill_mode, ) executable = args.executable or os.environ.get("EQ_CLAUDE_BIN", "claude") @@ -232,6 +251,7 @@ def main(argv: list[str] | None = None) -> int: workspace=workspace, skill_entry=skill_entry, model=args.model, + skill_mode=args.skill_mode, ) From 0074d93f2e32b70e61a04eaa7810ac3f0c064c87 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:26:59 +0800 Subject: [PATCH 07/11] Apply experiment changes: scripts/run_evals.py --- scripts/run_evals.py | 87 +++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/scripts/run_evals.py b/scripts/run_evals.py index 7754d5a..eecd3ed 100644 --- a/scripts/run_evals.py +++ b/scripts/run_evals.py @@ -12,6 +12,7 @@ import subprocess import sys import tempfile +import time from pathlib import Path from typing import Any @@ -48,6 +49,7 @@ "changed_files_include", "changed_files_subset", "command", + "command_no_changes", "file_absent", "file_contains", "file_exists", @@ -55,6 +57,7 @@ "file_unchanged", "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", } @@ -149,7 +152,7 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: errors.append(f"{check_label}: unsupported check type {check_type!r}") continue - if check_type == "command": + if check_type in {"command", "command_no_changes"}: argv = check.get("argv") if not isinstance(argv, list) or not argv or not all( isinstance(item, str) and item for item in argv @@ -158,6 +161,19 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: repeat = check.get("repeat", 1) if not isinstance(repeat, int) or repeat < 1 or repeat > 20: errors.append(f"{check_label}: repeat must be between 1 and 20") + if ( + isinstance(argv, list) + and len(argv) >= 3 + and argv[0] == "{python}" + and argv[1] == "-c" + and isinstance(argv[2], str) + ): + try: + compile(argv[2], f"<{check_label}>", "exec") + except SyntaxError as exc: + errors.append( + f"{check_label}: invalid inline Python: {exc.msg}" + ) elif check_type in {"changed_files_include", "changed_files_subset"}: paths = check.get("paths") if not isinstance(paths, list) or not all( @@ -182,6 +198,7 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: elif check_type in { "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", }: terms = check.get("terms") @@ -324,6 +341,7 @@ def run_agent( "EQ_EVAL_PASSED_ENV": ",".join(pass_env), } ) + started = time.monotonic() try: completed = subprocess.run( argv, @@ -341,6 +359,7 @@ def run_agent( "stdout": completed.stdout, "stderr": completed.stderr, "timed_out": False, + "duration_seconds": time.monotonic() - started, } except subprocess.TimeoutExpired as exc: return { @@ -349,6 +368,7 @@ def run_agent( "stdout": exc.stdout or "", "stderr": exc.stderr or "", "timed_out": True, + "duration_seconds": time.monotonic() - started, } @@ -363,6 +383,43 @@ def _read_optional(path: Path) -> str | None: return None +NEGATION_TOKENS = { + "not", + "never", + "without", + "cannot", + "can't", + "isn't", + "wasn't", + "aren't", + "weren't", + "couldn't", + "shouldn't", + "wouldn't", +} + + +def _is_negated_occurrence(prefix: str) -> bool: + clause = re.split(r"[.!?;\n]", prefix)[-1].casefold() + tokens = re.findall(r"[a-z]+(?:['’][a-z]+)?", clause) + if tokens[-2:] == ["not", "only"]: + return False + return any(token in NEGATION_TOKENS for token in tokens[-3:]) + + +def _unnegated_term_matches(text: str, terms: list[str]) -> list[str]: + matches: list[str] = [] + for term in terms: + pattern = re.compile(re.escape(term), re.IGNORECASE) + for match in pattern.finditer(text): + prefix = text[max(0, match.start() - 80):match.start()] + if _is_negated_occurrence(prefix): + continue + matches.append(term) + break + return matches + + def evaluate_check( check: dict[str, Any], *, @@ -436,6 +493,7 @@ def evaluate_check( if check_type in { "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", }: haystack = final_output.casefold() @@ -445,6 +503,9 @@ def evaluate_check( passed = len(matches) == len(terms) elif check_type == "final_contains_any": passed = bool(matches) + elif check_type == "final_not_claim_any": + matches = _unnegated_term_matches(final_output, check["terms"]) + passed = not matches else: passed = not matches result.update( @@ -456,7 +517,7 @@ def evaluate_check( ) return result - if check_type == "command": + if check_type in {"command", "command_no_changes"}: if not allow_workspace_execution: result.update( { @@ -480,6 +541,7 @@ def evaluate_check( ) as home_directory: check_env = _isolated_environment(home=Path(home_directory)) for _ in range(repeat): + execution_before = snapshot_workspace(workspace) try: completed = subprocess.run( argv, @@ -498,6 +560,15 @@ def evaluate_check( } if completed.returncode != expected_exit: passed = False + if check_type == "command_no_changes": + execution_after = snapshot_workspace(workspace) + generated_changes = changed_files( + execution_before, + execution_after, + ) + execution["changed_files"] = generated_changes + if generated_changes: + passed = False except subprocess.TimeoutExpired as exc: execution = { "exit_code": 124, @@ -531,6 +602,7 @@ def evaluate_case( keep_workspace: bool, workspace_parent: Path | None, pass_env: tuple[str, ...] = (), + run_index: int = 1, ) -> dict[str, Any]: if keep_workspace: workspace = Path( @@ -603,6 +675,7 @@ def evaluate_case( result = { "id": case["id"], + "run_index": run_index, "status": status, "task": case["task"], "agent": agent, @@ -713,6 +786,12 @@ def _parser() -> argparse.ArgumentParser: parser.add_argument("--output", type=Path, help="write machine-readable JSON report") parser.add_argument("--agent-timeout", type=int, default=900) parser.add_argument("--check-timeout", type=int, default=120) + parser.add_argument( + "--repeat", + type=int, + default=1, + help="repeat every selected case this many times; each repetition gets a fresh workspace", + ) parser.add_argument( "--allow-workspace-execution", action="store_true", @@ -745,6 +824,8 @@ def main(argv: list[str] | None = None) -> int: if args.agent_timeout <= 0 or args.check_timeout <= 0: parser.error("timeouts must be positive") + if args.repeat < 1 or args.repeat > 100: + parser.error("--repeat must be between 1 and 100") invalid_env_names = [ name for name in args.pass_env if not ENV_NAME_RE.fullmatch(name) ] @@ -801,7 +882,9 @@ def main(argv: list[str] | None = None) -> int: keep_workspace=args.keep_workspaces, workspace_parent=args.workspace_parent, pass_env=tuple(args.pass_env), + run_index=run_index, ) + for run_index in range(1, args.repeat + 1) for case in cases ] From 9bb9b3ea8d4a133a7e0939c29e2fb19ee3053150 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:27:06 +0800 Subject: [PATCH 08/11] Apply experiment changes: scripts/compare_eval_results.py --- scripts/compare_eval_results.py | 253 ++++++++++++++++++++++++++++++++ 1 file changed, 253 insertions(+) create mode 100644 scripts/compare_eval_results.py diff --git a/scripts/compare_eval_results.py b/scripts/compare_eval_results.py new file mode 100644 index 0000000..8e1857c --- /dev/null +++ b/scripts/compare_eval_results.py @@ -0,0 +1,253 @@ +#!/usr/bin/env python3 +"""Compare no-Skill and Skill behavioral-evaluation reports.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any + +INFRASTRUCTURE_CHECKS = {"skill_payload_integrity"} + + +def load_report(path: Path) -> dict[str, Any]: + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ValueError(f"{path}: report must be a JSON object") + cases = data.get("cases") + if not isinstance(cases, list): + raise ValueError(f"{path}: report.cases must be an array") + return data + + +def _group_cases(report: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: + grouped: dict[str, list[dict[str, Any]]] = {} + for index, case in enumerate(report["cases"]): + if not isinstance(case, dict): + raise ValueError(f"case[{index}] must be an object") + case_id = case.get("id") + if not isinstance(case_id, str) or not case_id: + raise ValueError(f"case[{index}].id must be a non-empty string") + grouped.setdefault(case_id, []).append(case) + + for runs in grouped.values(): + runs.sort(key=lambda item: item.get("run_index", 1)) + return grouped + + +def _task_for(runs: list[dict[str, Any]], *, case_id: str) -> str: + tasks = {run.get("task") for run in runs} + if len(tasks) != 1 or not all(isinstance(task, str) for task in tasks): + raise ValueError(f"{case_id}: runs do not share one task") + return next(iter(tasks)) + + +def _passed_count(runs: list[dict[str, Any]]) -> int: + return sum(run.get("status") == "passed" for run in runs) + + +def _rate(passed: int, total: int) -> float: + return passed / total if total else 0.0 + + +def _format_rate(passed: int, total: int) -> str: + return f"{passed}/{total} ({_rate(passed, total):.0%})" + + +def _mean_changed_files(runs: list[dict[str, Any]]) -> float: + counts = [ + len(run.get("changed_files", [])) + for run in runs + if isinstance(run.get("changed_files"), list) + ] + return sum(counts) / len(counts) if counts else 0.0 + + +def _mean_duration(runs: list[dict[str, Any]]) -> float | None: + values: list[float] = [] + for run in runs: + agent = run.get("agent") + if not isinstance(agent, dict): + continue + duration = agent.get("duration_seconds") + if isinstance(duration, (int, float)) and duration >= 0: + values.append(float(duration)) + return sum(values) / len(values) if values else None + + +def _format_duration(value: float | None) -> str: + return "n/a" if value is None else f"{value:.2f}s" + + +def _format_delta(baseline: float, skill: float) -> str: + return f"{(skill - baseline) * 100:+.0f} pp" + + +def _check_rates( + grouped: dict[str, list[dict[str, Any]]], +) -> dict[str, tuple[int, int]]: + totals: dict[str, list[int]] = {} + for runs in grouped.values(): + for run in runs: + checks = run.get("checks") + if not isinstance(checks, list): + continue + for check in checks: + if not isinstance(check, dict): + continue + check_type = check.get("type") + if ( + not isinstance(check_type, str) + or check_type in INFRASTRUCTURE_CHECKS + ): + continue + pair = totals.setdefault(check_type, [0, 0]) + pair[1] += 1 + if check.get("status") == "passed": + pair[0] += 1 + return {key: (value[0], value[1]) for key, value in totals.items()} + + +def _summary_row(label: str, runs: list[dict[str, Any]]) -> str: + passed = _passed_count(runs) + duration = _mean_duration(runs) + return ( + f"| {label} | {len(runs)} | {_format_rate(passed, len(runs))} | " + f"{_mean_changed_files(runs):.2f} | {_format_duration(duration)} |" + ) + + +def compare_reports( + baseline: dict[str, Any], + skill: dict[str, Any], +) -> str: + baseline_grouped = _group_cases(baseline) + skill_grouped = _group_cases(skill) + + baseline_ids = set(baseline_grouped) + skill_ids = set(skill_grouped) + if baseline_ids != skill_ids: + missing_from_skill = sorted(baseline_ids - skill_ids) + missing_from_baseline = sorted(skill_ids - baseline_ids) + raise ValueError( + "case sets differ; " + f"missing from Skill={missing_from_skill}, " + f"missing from baseline={missing_from_baseline}" + ) + + for case_id in sorted(baseline_ids): + baseline_task = _task_for(baseline_grouped[case_id], case_id=case_id) + skill_task = _task_for(skill_grouped[case_id], case_id=case_id) + if baseline_task != skill_task: + raise ValueError(f"{case_id}: baseline and Skill tasks differ") + + baseline_runs = [ + run for case_id in sorted(baseline_grouped) for run in baseline_grouped[case_id] + ] + skill_runs = [ + run for case_id in sorted(skill_grouped) for run in skill_grouped[case_id] + ] + + baseline_adapter = baseline.get("adapter", "unknown") + skill_adapter = skill.get("adapter", "unknown") + lines = [ + "# Skill A/B evaluation comparison", + "", + f"Baseline adapter: `{baseline_adapter}` ", + f"Skill adapter: `{skill_adapter}`", + "", + ( + "This comparison reports deterministic case outcomes, diff scope, and agent " + "wall-clock time. Qualitative `must_do` / `must_not_do` rubric items remain " + "manual review evidence and are not converted into an automatic score." + ), + "", + "## Summary", + "", + "| Condition | Runs | Deterministic case pass rate | Mean changed files | Mean agent time |", + "| --- | ---: | ---: | ---: | ---: |", + _summary_row("No Skill", baseline_runs), + _summary_row("Skill", skill_runs), + "", + "## Case-by-case", + "", + "| Case | No Skill | Skill | Pass-rate delta | Mean changed files (No Skill → Skill) | Mean agent time (No Skill → Skill) |", + "| --- | ---: | ---: | ---: | ---: | ---: |", + ] + + for case_id in sorted(baseline_grouped): + baseline_case = baseline_grouped[case_id] + skill_case = skill_grouped[case_id] + bp = _passed_count(baseline_case) + sp = _passed_count(skill_case) + br = _rate(bp, len(baseline_case)) + sr = _rate(sp, len(skill_case)) + lines.append( + f"| `{case_id}` | {_format_rate(bp, len(baseline_case))} | " + f"{_format_rate(sp, len(skill_case))} | {_format_delta(br, sr)} | " + f"{_mean_changed_files(baseline_case):.2f} → " + f"{_mean_changed_files(skill_case):.2f} | " + f"{_format_duration(_mean_duration(baseline_case))} → " + f"{_format_duration(_mean_duration(skill_case))} |" + ) + + baseline_checks = _check_rates(baseline_grouped) + skill_checks = _check_rates(skill_grouped) + check_types = sorted(set(baseline_checks) | set(skill_checks)) + lines.extend( + [ + "", + "## Deterministic check types", + "", + "| Check | No Skill | Skill | Pass-rate delta |", + "| --- | ---: | ---: | ---: |", + ] + ) + for check_type in check_types: + bp, bt = baseline_checks.get(check_type, (0, 0)) + sp, st = skill_checks.get(check_type, (0, 0)) + lines.append( + f"| `{check_type}` | {_format_rate(bp, bt)} | {_format_rate(sp, st)} | " + f"{_format_delta(_rate(bp, bt), _rate(sp, st))} |" + ) + + lines.extend( + [ + "", + ( + "Infrastructure-only checks such as `skill_payload_integrity` are excluded " + "from the check-type comparison so they do not inflate either condition." + ), + "", + ] + ) + return "\n".join(lines) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description="Compare a no-Skill eval report with a Skill-enabled eval report." + ) + parser.add_argument("baseline", type=Path, help="no-Skill JSON report") + parser.add_argument("skill", type=Path, help="Skill-enabled JSON report") + parser.add_argument("--output", type=Path, help="write Markdown comparison") + return parser + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + try: + rendered = compare_reports(load_report(args.baseline), load_report(args.skill)) + except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: + raise SystemExit(str(exc)) from exc + + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(rendered + "\n", encoding="utf-8") + print(rendered) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 78540accf13d711643497a9a283c43c7a449cdca Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:27:10 +0800 Subject: [PATCH 09/11] Apply experiment changes: tests/test_compare_eval_results.py --- tests/test_compare_eval_results.py | 142 +++++++++++++++++++++++++++++ 1 file changed, 142 insertions(+) create mode 100644 tests/test_compare_eval_results.py diff --git a/tests/test_compare_eval_results.py b/tests/test_compare_eval_results.py new file mode 100644 index 0000000..dedef90 --- /dev/null +++ b/tests/test_compare_eval_results.py @@ -0,0 +1,142 @@ +from __future__ import annotations + +import sys +import unittest +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import compare_eval_results + + +def make_run( + case_id: str, + *, + status: str, + run_index: int, + changed_files: int, + duration: float, + check_status: str = "passed", +) -> dict: + return { + "id": case_id, + "run_index": run_index, + "status": status, + "task": f"task for {case_id}", + "agent": {"duration_seconds": duration}, + "changed_files": [f"file-{index}.py" for index in range(changed_files)], + "checks": [ + {"type": "command", "status": check_status}, + {"type": "skill_payload_integrity", "status": "passed"}, + ], + "rubric": {}, + } + + +class CompareEvalResultsTests(unittest.TestCase): + def test_comparison_reports_repeated_case_rates_and_costs(self) -> None: + baseline = { + "adapter": "codex-model-no-skill", + "cases": [ + make_run( + "sample", + status="failed", + run_index=1, + changed_files=3, + duration=4.0, + check_status="failed", + ), + make_run( + "sample", + status="passed", + run_index=2, + changed_files=1, + duration=2.0, + ), + ], + } + skill = { + "adapter": "codex-model-skill", + "cases": [ + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=3.0, + ), + make_run( + "sample", + status="passed", + run_index=2, + changed_files=1, + duration=5.0, + ), + ], + } + + rendered = compare_eval_results.compare_reports(baseline, skill) + + self.assertIn("1/2 (50%)", rendered) + self.assertIn("2/2 (100%)", rendered) + self.assertIn("+50 pp", rendered) + self.assertIn("2.00 → 1.00", rendered) + self.assertIn("3.00s → 4.00s", rendered) + self.assertNotIn("| `skill_payload_integrity` |", rendered) + + def test_comparison_rejects_different_case_sets(self) -> None: + baseline = { + "adapter": "baseline", + "cases": [ + make_run( + "one", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + ], + } + skill = { + "adapter": "skill", + "cases": [ + make_run( + "two", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + ], + } + + with self.assertRaisesRegex(ValueError, "case sets differ"): + compare_eval_results.compare_reports(baseline, skill) + + def test_comparison_rejects_task_drift(self) -> None: + baseline_run = make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + skill_run = make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + skill_run["task"] = "different task" + + with self.assertRaisesRegex(ValueError, "tasks differ"): + compare_eval_results.compare_reports( + {"adapter": "baseline", "cases": [baseline_run]}, + {"adapter": "skill", "cases": [skill_run]}, + ) + + +if __name__ == "__main__": + unittest.main() From a305ca7ee82a09f5fa9125b56b31e4da6bf143ac Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:27:25 +0800 Subject: [PATCH 10/11] Apply experiment changes: tests/test_host_eval_adapter.py --- tests/test_host_eval_adapter.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/tests/test_host_eval_adapter.py b/tests/test_host_eval_adapter.py index b96b980..bea2175 100644 --- a/tests/test_host_eval_adapter.py +++ b/tests/test_host_eval_adapter.py @@ -41,6 +41,25 @@ def test_codex_uses_noninteractive_workspace_write_mode(self) -> None: self.assertIn("workspace-write", argv) self.assertEqual("fix the bug", argv[-1]) + def test_codex_disabled_skill_mode_does_not_install_skill(self) -> None: + with mock.patch.object(host_eval_adapter, "_copy_skill") as copy_skill: + with mock.patch.object(host_eval_adapter, "_print_version"): + with mock.patch.object( + host_eval_adapter.subprocess, + "run", + return_value=mock.Mock(returncode=0), + ): + exit_code = host_eval_adapter.run_codex( + executable="codex", + task="fix", + workspace=Path("."), + skill_entry=Path("/tmp/staged-skill/SKILL.md"), + skill_mode="disabled", + ) + + self.assertEqual(0, exit_code) + copy_skill.assert_not_called() + def test_codex_model_can_be_pinned(self) -> None: argv = host_eval_adapter.codex_argv("codex", "fix", "gpt-test") @@ -63,6 +82,12 @@ def test_claude_uses_bare_auto_mode_and_explicit_skill_directory(self) -> None: self.assertIn("-p", argv) self.assertEqual("fix the bug", argv[-1]) + def test_claude_disabled_skill_mode_omits_skill_directory(self) -> None: + argv = host_eval_adapter.claude_argv("claude", "fix", None) + + self.assertNotIn("--add-dir", argv) + self.assertEqual("fix", argv[-1]) + def test_claude_model_can_be_pinned(self) -> None: argv = host_eval_adapter.claude_argv( "claude", "fix", Path("/tmp/staged-skill"), "claude-test" From b3b8c70a8f6549bb4b661728e849ee0f38694cc0 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:27:34 +0800 Subject: [PATCH 11/11] Apply experiment changes: tests/test_run_evals.py --- tests/test_run_evals.py | 164 +++++++++++++++++++++++++++++++++++++++- 1 file changed, 163 insertions(+), 1 deletion(-) diff --git a/tests/test_run_evals.py b/tests/test_run_evals.py index d92c17f..9b1ee72 100644 --- a/tests/test_run_evals.py +++ b/tests/test_run_evals.py @@ -2,6 +2,7 @@ import os import shlex +import subprocess import sys import tempfile import unittest @@ -81,6 +82,19 @@ def test_check_environment_uses_isolated_home(self) -> None: self.assertEqual(str(home), env["HOME"]) self.assertEqual(str(home), env["USERPROFILE"]) + def test_validate_cases_rejects_invalid_inline_python_check(self) -> None: + case = self.sample_case() + case["checks"] = [ + { + "type": "command", + "argv": ["{python}", "-c", "if True print('bad')"], + } + ] + + errors = run_evals.validate_cases([case]) + + self.assertTrue(any("invalid inline Python" in error for error in errors)) + def test_validate_cases_rejects_unsafe_fixture_path(self) -> None: case = self.sample_case() case["fixture"]["files"] = {"../escape.py": "bad\n"} @@ -107,7 +121,9 @@ def test_evaluate_case_runs_agent_and_deterministic_checks(self) -> None: ) self.assertEqual("passed", result["status"]) + self.assertEqual(1, result["run_index"]) self.assertEqual(["solution.py"], result["changed_files"]) + self.assertGreaterEqual(result["agent"]["duration_seconds"], 0) self.assertTrue(all(check["status"] == "passed" for check in result["checks"])) def test_command_checks_are_incomplete_without_execution_opt_in(self) -> None: @@ -188,6 +204,114 @@ def test_agent_cannot_mutate_staged_skill_without_failing_case(self) -> None: self.assertEqual("failed", integrity["status"]) self.assertIn("SKILL.md", integrity["changed_files"]) + def test_final_not_claim_any_ignores_negated_claim(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is not fully verified because the service check is blocked.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("passed", result["status"]) + self.assertEqual([], result["matched"]) + + def test_final_not_claim_any_does_not_treat_not_only_as_negation(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is not only fully verified but release-ready.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + + def test_final_not_claim_any_rejects_positive_claim(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is fully verified.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + self.assertEqual(["fully verified"], result["matched"]) + + def test_command_no_changes_detects_generator_drift(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + path = workspace / "generated.py" + path.write_text("value = 1\n", encoding="utf-8") + before = run_evals.snapshot_workspace(workspace) + result = run_evals.evaluate_check( + { + "type": "command_no_changes", + "argv": [ + "{python}", + "-c", + "from pathlib import Path; Path('generated.py').write_text('value = 2\\n')", + ], + }, + workspace=workspace, + before=before, + after=before, + final_output="", + allow_workspace_execution=True, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + self.assertEqual(["generated.py"], result["executions"][0]["changed_files"]) + + def test_command_no_changes_passes_idempotent_generator(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + path = workspace / "generated.py" + path.write_text("value = 1\n", encoding="utf-8") + before = run_evals.snapshot_workspace(workspace) + result = run_evals.evaluate_check( + { + "type": "command_no_changes", + "argv": [ + "{python}", + "-c", + "from pathlib import Path; p = Path('generated.py'); p.write_text(p.read_text())", + ], + }, + workspace=workspace, + before=before, + after=before, + final_output="", + allow_workspace_execution=True, + command_timeout=30, + ) + + self.assertEqual("passed", result["status"]) + self.assertEqual([], result["executions"][0]["changed_files"]) + def test_report_records_only_forwarded_environment_names(self) -> None: report = run_evals.build_report( [], @@ -201,6 +325,18 @@ def test_report_records_only_forwarded_environment_names(self) -> None: report["forwarded_environment"], ) + def test_repeat_defaults_to_one_and_accepts_explicit_count(self) -> None: + parser = run_evals._parser() + + self.assertEqual(1, parser.parse_args([]).repeat) + self.assertEqual(5, parser.parse_args(["--repeat", "5"]).repeat) + + def test_cli_rejects_invalid_repeat(self) -> None: + with self.assertRaises(SystemExit) as raised: + run_evals.main(["--validate-only", "--repeat", "0"]) + + self.assertEqual(2, raised.exception.code) + def test_cli_rejects_requested_environment_that_is_not_set(self) -> None: with mock.patch.dict(os.environ, {}, clear=True): with self.assertRaises(SystemExit) as raised: @@ -242,10 +378,36 @@ def test_staged_skill_excludes_eval_rubric(self) -> None: self.assertFalse((Path(directory) / "evals").exists()) self.assertFalse((Path(directory) / "tests").exists()) + def test_flaky_fixture_does_not_shadow_stdlib_token_module(self) -> None: + cases = run_evals.load_cases(ROOT / "evals" / "cases.json") + case = next(case for case in cases if case["id"] == "flaky-test") + + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + run_evals.materialize_fixture(case, workspace) + self.assertFalse((workspace / "token.py").exists()) + completed = subprocess.run( + [ + sys.executable, + "-c", + ( + "from token_value import token; " + "assert token('job', 7) == 'job-0007'" + ), + ], + cwd=workspace, + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + + self.assertEqual(0, completed.returncode, completed.stderr) + def test_repository_cases_validate(self) -> None: cases = run_evals.load_cases(ROOT / "evals" / "cases.json") self.assertEqual([], run_evals.validate_cases(cases)) - self.assertEqual(14, len(cases)) + self.assertEqual(17, len(cases)) if __name__ == "__main__":