From bffdc043f98252b6a1b8b5adc7c90133c3d4777a Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:40:28 +0800 Subject: [PATCH 01/34] Add clean no-skill host eval mode --- scripts/host_eval_adapter.py | 36 ++++++++++++++++++++++++++++-------- 1 file changed, 28 insertions(+), 8 deletions(-) diff --git a/scripts/host_eval_adapter.py b/scripts/host_eval_adapter.py index 0b8c168..3ab8df3 100644 --- a/scripts/host_eval_adapter.py +++ b/scripts/host_eval_adapter.py @@ -72,7 +72,7 @@ def codex_argv( def claude_argv( executable: str, task: str, - skill_dir: Path, + skill_dir: Path | None, model: str | None = None, ) -> list[str]: argv = [ @@ -83,9 +83,9 @@ def claude_argv( "--permission-prompts", "none", "--no-session-persistence", - "--add-dir", - str(skill_dir), ] + if skill_dir is not None: + argv.extend(("--add-dir", str(skill_dir))) if model: argv.extend(("--model", model)) argv.extend(("-p", task)) @@ -119,13 +119,15 @@ def run_codex( workspace: Path, skill_entry: Path, model: str | None = None, + skill_mode: str = "enabled", ) -> int: with tempfile.TemporaryDirectory(prefix="engineering-quality-codex-home-") as directory: home = Path(directory) - _copy_skill( - skill_entry, - home / ".agents" / "skills" / "engineering-quality", - ) + if skill_mode == "enabled": + _copy_skill( + skill_entry, + home / ".agents" / "skills" / "engineering-quality", + ) env = _adapter_environment() env["HOME"] = str(home) env["USERPROFILE"] = str(home) @@ -153,6 +155,7 @@ def run_claude_code( workspace: Path, skill_entry: Path, model: str | None = None, + skill_mode: str = "enabled", ) -> int: with tempfile.TemporaryDirectory( prefix="engineering-quality-claude-home-" @@ -167,7 +170,12 @@ def run_claude_code( _print_version(executable, env, workspace) try: completed = subprocess.run( - claude_argv(executable, task, skill_entry.resolve().parent, model), + claude_argv( + executable, + task, + skill_entry.resolve().parent if skill_mode == "enabled" else None, + model, + ), cwd=workspace, env=env, check=False, @@ -191,6 +199,15 @@ def _parser() -> argparse.ArgumentParser: "--model", help="pin an explicit host model for reproducible behavioral evidence", ) + parser.add_argument( + "--skill-mode", + choices=("enabled", "disabled"), + default="enabled", + help=( + "enable the staged engineering-quality Skill or run a clean no-Skill " + "baseline; defaults to enabled for backward compatibility" + ), + ) return parser @@ -214,6 +231,7 @@ def main(argv: list[str] | None = None) -> int: if args.model: print(f"[host-adapter] model={args.model}", file=sys.stderr) + print(f"[host-adapter] skill-mode={args.skill_mode}", file=sys.stderr) if args.host == "codex": executable = args.executable or os.environ.get("EQ_CODEX_BIN", "codex") @@ -223,6 +241,7 @@ def main(argv: list[str] | None = None) -> int: workspace=workspace, skill_entry=skill_entry, model=args.model, + skill_mode=args.skill_mode, ) executable = args.executable or os.environ.get("EQ_CLAUDE_BIN", "claude") @@ -232,6 +251,7 @@ def main(argv: list[str] | None = None) -> int: workspace=workspace, skill_entry=skill_entry, model=args.model, + skill_mode=args.skill_mode, ) From 8c80e855739e708513c5e8a2665000491ee4e1d0 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:40:43 +0800 Subject: [PATCH 02/34] Test clean no-skill adapter behavior --- tests/test_host_eval_adapter.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/tests/test_host_eval_adapter.py b/tests/test_host_eval_adapter.py index b96b980..bea2175 100644 --- a/tests/test_host_eval_adapter.py +++ b/tests/test_host_eval_adapter.py @@ -41,6 +41,25 @@ def test_codex_uses_noninteractive_workspace_write_mode(self) -> None: self.assertIn("workspace-write", argv) self.assertEqual("fix the bug", argv[-1]) + def test_codex_disabled_skill_mode_does_not_install_skill(self) -> None: + with mock.patch.object(host_eval_adapter, "_copy_skill") as copy_skill: + with mock.patch.object(host_eval_adapter, "_print_version"): + with mock.patch.object( + host_eval_adapter.subprocess, + "run", + return_value=mock.Mock(returncode=0), + ): + exit_code = host_eval_adapter.run_codex( + executable="codex", + task="fix", + workspace=Path("."), + skill_entry=Path("/tmp/staged-skill/SKILL.md"), + skill_mode="disabled", + ) + + self.assertEqual(0, exit_code) + copy_skill.assert_not_called() + def test_codex_model_can_be_pinned(self) -> None: argv = host_eval_adapter.codex_argv("codex", "fix", "gpt-test") @@ -63,6 +82,12 @@ def test_claude_uses_bare_auto_mode_and_explicit_skill_directory(self) -> None: self.assertIn("-p", argv) self.assertEqual("fix the bug", argv[-1]) + def test_claude_disabled_skill_mode_omits_skill_directory(self) -> None: + argv = host_eval_adapter.claude_argv("claude", "fix", None) + + self.assertNotIn("--add-dir", argv) + self.assertEqual("fix", argv[-1]) + def test_claude_model_can_be_pinned(self) -> None: argv = host_eval_adapter.claude_argv( "claude", "fix", Path("/tmp/staged-skill"), "claude-test" From 3c2f407082c66e9ca7760e093c5f8caef4eb158c Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:41:13 +0800 Subject: [PATCH 03/34] Support repeated eval runs and timing evidence --- scripts/run_evals.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/scripts/run_evals.py b/scripts/run_evals.py index 7754d5a..9bd6c31 100644 --- a/scripts/run_evals.py +++ b/scripts/run_evals.py @@ -12,6 +12,7 @@ import subprocess import sys import tempfile +import time from pathlib import Path from typing import Any @@ -324,6 +325,7 @@ def run_agent( "EQ_EVAL_PASSED_ENV": ",".join(pass_env), } ) + started = time.monotonic() try: completed = subprocess.run( argv, @@ -341,6 +343,7 @@ def run_agent( "stdout": completed.stdout, "stderr": completed.stderr, "timed_out": False, + "duration_seconds": time.monotonic() - started, } except subprocess.TimeoutExpired as exc: return { @@ -349,6 +352,7 @@ def run_agent( "stdout": exc.stdout or "", "stderr": exc.stderr or "", "timed_out": True, + "duration_seconds": time.monotonic() - started, } @@ -531,6 +535,7 @@ def evaluate_case( keep_workspace: bool, workspace_parent: Path | None, pass_env: tuple[str, ...] = (), + run_index: int = 1, ) -> dict[str, Any]: if keep_workspace: workspace = Path( @@ -603,6 +608,7 @@ def evaluate_case( result = { "id": case["id"], + "run_index": run_index, "status": status, "task": case["task"], "agent": agent, @@ -713,6 +719,12 @@ def _parser() -> argparse.ArgumentParser: parser.add_argument("--output", type=Path, help="write machine-readable JSON report") parser.add_argument("--agent-timeout", type=int, default=900) parser.add_argument("--check-timeout", type=int, default=120) + parser.add_argument( + "--repeat", + type=int, + default=1, + help="repeat every selected case this many times; each repetition gets a fresh workspace", + ) parser.add_argument( "--allow-workspace-execution", action="store_true", @@ -745,6 +757,8 @@ def main(argv: list[str] | None = None) -> int: if args.agent_timeout <= 0 or args.check_timeout <= 0: parser.error("timeouts must be positive") + if args.repeat < 1 or args.repeat > 100: + parser.error("--repeat must be between 1 and 100") invalid_env_names = [ name for name in args.pass_env if not ENV_NAME_RE.fullmatch(name) ] @@ -801,7 +815,9 @@ def main(argv: list[str] | None = None) -> int: keep_workspace=args.keep_workspaces, workspace_parent=args.workspace_parent, pass_env=tuple(args.pass_env), + run_index=run_index, ) + for run_index in range(1, args.repeat + 1) for case in cases ] From b69e203fad31d0bce7fc2a3932216a9fcae6b52c Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:41:26 +0800 Subject: [PATCH 04/34] Test repeated eval metadata --- tests/test_run_evals.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/test_run_evals.py b/tests/test_run_evals.py index d92c17f..e3313d5 100644 --- a/tests/test_run_evals.py +++ b/tests/test_run_evals.py @@ -107,7 +107,9 @@ def test_evaluate_case_runs_agent_and_deterministic_checks(self) -> None: ) self.assertEqual("passed", result["status"]) + self.assertEqual(1, result["run_index"]) self.assertEqual(["solution.py"], result["changed_files"]) + self.assertGreaterEqual(result["agent"]["duration_seconds"], 0) self.assertTrue(all(check["status"] == "passed" for check in result["checks"])) def test_command_checks_are_incomplete_without_execution_opt_in(self) -> None: @@ -201,6 +203,18 @@ def test_report_records_only_forwarded_environment_names(self) -> None: report["forwarded_environment"], ) + def test_repeat_defaults_to_one_and_accepts_explicit_count(self) -> None: + parser = run_evals._parser() + + self.assertEqual(1, parser.parse_args([]).repeat) + self.assertEqual(5, parser.parse_args(["--repeat", "5"]).repeat) + + def test_cli_rejects_invalid_repeat(self) -> None: + with self.assertRaises(SystemExit) as raised: + run_evals.main(["--validate-only", "--repeat", "0"]) + + self.assertEqual(2, raised.exception.code) + def test_cli_rejects_requested_environment_that_is_not_set(self) -> None: with mock.patch.dict(os.environ, {}, clear=True): with self.assertRaises(SystemExit) as raised: From 4341d2bed83b3d6c6c20180b94f961ebfc448431 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:42:08 +0800 Subject: [PATCH 05/34] Add Skill A/B eval comparison report --- scripts/compare_eval_results.py | 253 ++++++++++++++++++++++++++++++++ 1 file changed, 253 insertions(+) create mode 100644 scripts/compare_eval_results.py diff --git a/scripts/compare_eval_results.py b/scripts/compare_eval_results.py new file mode 100644 index 0000000..8e1857c --- /dev/null +++ b/scripts/compare_eval_results.py @@ -0,0 +1,253 @@ +#!/usr/bin/env python3 +"""Compare no-Skill and Skill behavioral-evaluation reports.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any + +INFRASTRUCTURE_CHECKS = {"skill_payload_integrity"} + + +def load_report(path: Path) -> dict[str, Any]: + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ValueError(f"{path}: report must be a JSON object") + cases = data.get("cases") + if not isinstance(cases, list): + raise ValueError(f"{path}: report.cases must be an array") + return data + + +def _group_cases(report: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: + grouped: dict[str, list[dict[str, Any]]] = {} + for index, case in enumerate(report["cases"]): + if not isinstance(case, dict): + raise ValueError(f"case[{index}] must be an object") + case_id = case.get("id") + if not isinstance(case_id, str) or not case_id: + raise ValueError(f"case[{index}].id must be a non-empty string") + grouped.setdefault(case_id, []).append(case) + + for runs in grouped.values(): + runs.sort(key=lambda item: item.get("run_index", 1)) + return grouped + + +def _task_for(runs: list[dict[str, Any]], *, case_id: str) -> str: + tasks = {run.get("task") for run in runs} + if len(tasks) != 1 or not all(isinstance(task, str) for task in tasks): + raise ValueError(f"{case_id}: runs do not share one task") + return next(iter(tasks)) + + +def _passed_count(runs: list[dict[str, Any]]) -> int: + return sum(run.get("status") == "passed" for run in runs) + + +def _rate(passed: int, total: int) -> float: + return passed / total if total else 0.0 + + +def _format_rate(passed: int, total: int) -> str: + return f"{passed}/{total} ({_rate(passed, total):.0%})" + + +def _mean_changed_files(runs: list[dict[str, Any]]) -> float: + counts = [ + len(run.get("changed_files", [])) + for run in runs + if isinstance(run.get("changed_files"), list) + ] + return sum(counts) / len(counts) if counts else 0.0 + + +def _mean_duration(runs: list[dict[str, Any]]) -> float | None: + values: list[float] = [] + for run in runs: + agent = run.get("agent") + if not isinstance(agent, dict): + continue + duration = agent.get("duration_seconds") + if isinstance(duration, (int, float)) and duration >= 0: + values.append(float(duration)) + return sum(values) / len(values) if values else None + + +def _format_duration(value: float | None) -> str: + return "n/a" if value is None else f"{value:.2f}s" + + +def _format_delta(baseline: float, skill: float) -> str: + return f"{(skill - baseline) * 100:+.0f} pp" + + +def _check_rates( + grouped: dict[str, list[dict[str, Any]]], +) -> dict[str, tuple[int, int]]: + totals: dict[str, list[int]] = {} + for runs in grouped.values(): + for run in runs: + checks = run.get("checks") + if not isinstance(checks, list): + continue + for check in checks: + if not isinstance(check, dict): + continue + check_type = check.get("type") + if ( + not isinstance(check_type, str) + or check_type in INFRASTRUCTURE_CHECKS + ): + continue + pair = totals.setdefault(check_type, [0, 0]) + pair[1] += 1 + if check.get("status") == "passed": + pair[0] += 1 + return {key: (value[0], value[1]) for key, value in totals.items()} + + +def _summary_row(label: str, runs: list[dict[str, Any]]) -> str: + passed = _passed_count(runs) + duration = _mean_duration(runs) + return ( + f"| {label} | {len(runs)} | {_format_rate(passed, len(runs))} | " + f"{_mean_changed_files(runs):.2f} | {_format_duration(duration)} |" + ) + + +def compare_reports( + baseline: dict[str, Any], + skill: dict[str, Any], +) -> str: + baseline_grouped = _group_cases(baseline) + skill_grouped = _group_cases(skill) + + baseline_ids = set(baseline_grouped) + skill_ids = set(skill_grouped) + if baseline_ids != skill_ids: + missing_from_skill = sorted(baseline_ids - skill_ids) + missing_from_baseline = sorted(skill_ids - baseline_ids) + raise ValueError( + "case sets differ; " + f"missing from Skill={missing_from_skill}, " + f"missing from baseline={missing_from_baseline}" + ) + + for case_id in sorted(baseline_ids): + baseline_task = _task_for(baseline_grouped[case_id], case_id=case_id) + skill_task = _task_for(skill_grouped[case_id], case_id=case_id) + if baseline_task != skill_task: + raise ValueError(f"{case_id}: baseline and Skill tasks differ") + + baseline_runs = [ + run for case_id in sorted(baseline_grouped) for run in baseline_grouped[case_id] + ] + skill_runs = [ + run for case_id in sorted(skill_grouped) for run in skill_grouped[case_id] + ] + + baseline_adapter = baseline.get("adapter", "unknown") + skill_adapter = skill.get("adapter", "unknown") + lines = [ + "# Skill A/B evaluation comparison", + "", + f"Baseline adapter: `{baseline_adapter}` ", + f"Skill adapter: `{skill_adapter}`", + "", + ( + "This comparison reports deterministic case outcomes, diff scope, and agent " + "wall-clock time. Qualitative `must_do` / `must_not_do` rubric items remain " + "manual review evidence and are not converted into an automatic score." + ), + "", + "## Summary", + "", + "| Condition | Runs | Deterministic case pass rate | Mean changed files | Mean agent time |", + "| --- | ---: | ---: | ---: | ---: |", + _summary_row("No Skill", baseline_runs), + _summary_row("Skill", skill_runs), + "", + "## Case-by-case", + "", + "| Case | No Skill | Skill | Pass-rate delta | Mean changed files (No Skill → Skill) | Mean agent time (No Skill → Skill) |", + "| --- | ---: | ---: | ---: | ---: | ---: |", + ] + + for case_id in sorted(baseline_grouped): + baseline_case = baseline_grouped[case_id] + skill_case = skill_grouped[case_id] + bp = _passed_count(baseline_case) + sp = _passed_count(skill_case) + br = _rate(bp, len(baseline_case)) + sr = _rate(sp, len(skill_case)) + lines.append( + f"| `{case_id}` | {_format_rate(bp, len(baseline_case))} | " + f"{_format_rate(sp, len(skill_case))} | {_format_delta(br, sr)} | " + f"{_mean_changed_files(baseline_case):.2f} → " + f"{_mean_changed_files(skill_case):.2f} | " + f"{_format_duration(_mean_duration(baseline_case))} → " + f"{_format_duration(_mean_duration(skill_case))} |" + ) + + baseline_checks = _check_rates(baseline_grouped) + skill_checks = _check_rates(skill_grouped) + check_types = sorted(set(baseline_checks) | set(skill_checks)) + lines.extend( + [ + "", + "## Deterministic check types", + "", + "| Check | No Skill | Skill | Pass-rate delta |", + "| --- | ---: | ---: | ---: |", + ] + ) + for check_type in check_types: + bp, bt = baseline_checks.get(check_type, (0, 0)) + sp, st = skill_checks.get(check_type, (0, 0)) + lines.append( + f"| `{check_type}` | {_format_rate(bp, bt)} | {_format_rate(sp, st)} | " + f"{_format_delta(_rate(bp, bt), _rate(sp, st))} |" + ) + + lines.extend( + [ + "", + ( + "Infrastructure-only checks such as `skill_payload_integrity` are excluded " + "from the check-type comparison so they do not inflate either condition." + ), + "", + ] + ) + return "\n".join(lines) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description="Compare a no-Skill eval report with a Skill-enabled eval report." + ) + parser.add_argument("baseline", type=Path, help="no-Skill JSON report") + parser.add_argument("skill", type=Path, help="Skill-enabled JSON report") + parser.add_argument("--output", type=Path, help="write Markdown comparison") + return parser + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + try: + rendered = compare_reports(load_report(args.baseline), load_report(args.skill)) + except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: + raise SystemExit(str(exc)) from exc + + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(rendered + "\n", encoding="utf-8") + print(rendered) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 8144e02cc958becc2f7363f97b107b05383b3a40 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:42:23 +0800 Subject: [PATCH 06/34] Test Skill A/B result comparison --- tests/test_compare_eval_results.py | 142 +++++++++++++++++++++++++++++ 1 file changed, 142 insertions(+) create mode 100644 tests/test_compare_eval_results.py diff --git a/tests/test_compare_eval_results.py b/tests/test_compare_eval_results.py new file mode 100644 index 0000000..dedef90 --- /dev/null +++ b/tests/test_compare_eval_results.py @@ -0,0 +1,142 @@ +from __future__ import annotations + +import sys +import unittest +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import compare_eval_results + + +def make_run( + case_id: str, + *, + status: str, + run_index: int, + changed_files: int, + duration: float, + check_status: str = "passed", +) -> dict: + return { + "id": case_id, + "run_index": run_index, + "status": status, + "task": f"task for {case_id}", + "agent": {"duration_seconds": duration}, + "changed_files": [f"file-{index}.py" for index in range(changed_files)], + "checks": [ + {"type": "command", "status": check_status}, + {"type": "skill_payload_integrity", "status": "passed"}, + ], + "rubric": {}, + } + + +class CompareEvalResultsTests(unittest.TestCase): + def test_comparison_reports_repeated_case_rates_and_costs(self) -> None: + baseline = { + "adapter": "codex-model-no-skill", + "cases": [ + make_run( + "sample", + status="failed", + run_index=1, + changed_files=3, + duration=4.0, + check_status="failed", + ), + make_run( + "sample", + status="passed", + run_index=2, + changed_files=1, + duration=2.0, + ), + ], + } + skill = { + "adapter": "codex-model-skill", + "cases": [ + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=3.0, + ), + make_run( + "sample", + status="passed", + run_index=2, + changed_files=1, + duration=5.0, + ), + ], + } + + rendered = compare_eval_results.compare_reports(baseline, skill) + + self.assertIn("1/2 (50%)", rendered) + self.assertIn("2/2 (100%)", rendered) + self.assertIn("+50 pp", rendered) + self.assertIn("2.00 → 1.00", rendered) + self.assertIn("3.00s → 4.00s", rendered) + self.assertNotIn("| `skill_payload_integrity` |", rendered) + + def test_comparison_rejects_different_case_sets(self) -> None: + baseline = { + "adapter": "baseline", + "cases": [ + make_run( + "one", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + ], + } + skill = { + "adapter": "skill", + "cases": [ + make_run( + "two", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + ], + } + + with self.assertRaisesRegex(ValueError, "case sets differ"): + compare_eval_results.compare_reports(baseline, skill) + + def test_comparison_rejects_task_drift(self) -> None: + baseline_run = make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + skill_run = make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + skill_run["task"] = "different task" + + with self.assertRaisesRegex(ValueError, "tasks differ"): + compare_eval_results.compare_reports( + {"adapter": "baseline", "cases": [baseline_run]}, + {"adapter": "skill", "cases": [skill_run]}, + ) + + +if __name__ == "__main__": + unittest.main() From b03462feed8068085838464da1bae9a5f1272343 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:42:50 +0800 Subject: [PATCH 07/34] Document repeated eval run index --- evals/result-schema.json | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/evals/result-schema.json b/evals/result-schema.json index 4f798ec..5d6db84 100644 --- a/evals/result-schema.json +++ b/evals/result-schema.json @@ -63,6 +63,10 @@ "id": { "type": "string" }, + "run_index": { + "type": "integer", + "minimum": 1 + }, "status": { "enum": [ "passed", From 434a0f649f1ebab244bdd482b7b25cb193511eb7 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 12:42:52 +0800 Subject: [PATCH 08/34] Document Skill A/B evaluation workflow --- evals/README.md | 41 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/evals/README.md b/evals/README.md index 9b4c953..25756e5 100644 --- a/evals/README.md +++ b/evals/README.md @@ -128,6 +128,47 @@ Adapter unit tests validate command construction and staged-skill handling. Thos Each case receives a fresh runtime Skill staging directory. The harness hashes it before and after the agent exits. Any mutation produces a failing `skill_payload_integrity` check, so one case cannot rewrite the Skill used by later cases. +### Compare Skill vs no-Skill behavior + +The native adapter can run the same task without exposing the staged Skill. The baseline is created by removing the Skill from the host environment, not by adding a prompt that tells the model to ignore it. + +Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. Use `--repeat` when you need repeated independent trials; every repetition receives a fresh fixture workspace. + +Codex example: + +```bash +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --adapter-label codex-MODEL-no-skill \ + --pass-env CODEX_API_KEY \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/codex-MODEL-no-skill.json + +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --adapter-label codex-MODEL-skill \ + --pass-env CODEX_API_KEY \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/codex-MODEL-skill.json +``` + +Claude Code uses the same `--skill-mode disabled|enabled` switch on `host_eval_adapter.py`. + +Compare the two reports: + +```bash +python scripts/compare_eval_results.py \ + eval-results/codex-MODEL-no-skill.json \ + eval-results/codex-MODEL-skill.json \ + --output eval-results/codex-MODEL-comparison.md +``` + +The comparison reports deterministic case pass rates, deterministic check-type pass rates, mean changed-file counts, and mean agent wall-clock time. It deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is excluded from comparative check rates. + +For publishable evidence, run both conditions close enough together to reduce host/model drift, retain the raw JSON reports, and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. + See [host compatibility](../docs/compatibility.md) for the dated vendor documentation basis. ## Execution boundary From 1e94c9a288e4237a571a7ef13434c6aa194393e7 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:15:18 +0800 Subject: [PATCH 09/34] Require paired repetitions in A/B comparisons --- scripts/compare_eval_results.py | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/scripts/compare_eval_results.py b/scripts/compare_eval_results.py index 8e1857c..6f3743a 100644 --- a/scripts/compare_eval_results.py +++ b/scripts/compare_eval_results.py @@ -31,8 +31,13 @@ def _group_cases(report: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: raise ValueError(f"case[{index}].id must be a non-empty string") grouped.setdefault(case_id, []).append(case) - for runs in grouped.values(): + for case_id, runs in grouped.items(): runs.sort(key=lambda item: item.get("run_index", 1)) + indexes = [run.get("run_index", 1) for run in runs] + if not all(isinstance(index, int) and index >= 1 for index in indexes): + raise ValueError(f"{case_id}: run_index must be a positive integer") + if len(indexes) != len(set(indexes)): + raise ValueError(f"{case_id}: duplicate run_index values") return grouped @@ -141,6 +146,17 @@ def compare_reports( skill_task = _task_for(skill_grouped[case_id], case_id=case_id) if baseline_task != skill_task: raise ValueError(f"{case_id}: baseline and Skill tasks differ") + baseline_indexes = [ + run.get("run_index", 1) for run in baseline_grouped[case_id] + ] + skill_indexes = [ + run.get("run_index", 1) for run in skill_grouped[case_id] + ] + if baseline_indexes != skill_indexes: + raise ValueError( + f"{case_id}: baseline and Skill repetition sets differ " + f"({baseline_indexes} != {skill_indexes})" + ) baseline_runs = [ run for case_id in sorted(baseline_grouped) for run in baseline_grouped[case_id] From 9214276aad582aa48aeaf9b2d93a6859bb490ac3 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:15:37 +0800 Subject: [PATCH 10/34] Test paired A/B repetition requirements --- tests/test_compare_eval_results.py | 60 ++++++++++++++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/tests/test_compare_eval_results.py b/tests/test_compare_eval_results.py index dedef90..a99b144 100644 --- a/tests/test_compare_eval_results.py +++ b/tests/test_compare_eval_results.py @@ -114,6 +114,66 @@ def test_comparison_rejects_different_case_sets(self) -> None: with self.assertRaisesRegex(ValueError, "case sets differ"): compare_eval_results.compare_reports(baseline, skill) + def test_comparison_rejects_unpaired_repetitions(self) -> None: + baseline = { + "adapter": "baseline", + "cases": [ + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ), + make_run( + "sample", + status="passed", + run_index=2, + changed_files=1, + duration=1.0, + ), + ], + } + skill = { + "adapter": "skill", + "cases": [ + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ) + ], + } + + with self.assertRaisesRegex(ValueError, "repetition sets differ"): + compare_eval_results.compare_reports(baseline, skill) + + def test_comparison_rejects_duplicate_run_indexes(self) -> None: + duplicate_runs = [ + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ), + make_run( + "sample", + status="passed", + run_index=1, + changed_files=1, + duration=1.0, + ), + ] + + with self.assertRaisesRegex(ValueError, "duplicate run_index"): + compare_eval_results.compare_reports( + {"adapter": "baseline", "cases": duplicate_runs}, + {"adapter": "skill", "cases": duplicate_runs}, + ) + def test_comparison_rejects_task_drift(self) -> None: baseline_run = make_run( "sample", From cddaf884371911f3030c560318a80598671f1bee Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:16:51 +0800 Subject: [PATCH 11/34] Add paired counterbalanced A/B eval runner --- scripts/run_ab_evals.py | 247 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 247 insertions(+) create mode 100644 scripts/run_ab_evals.py diff --git a/scripts/run_ab_evals.py b/scripts/run_ab_evals.py new file mode 100644 index 0000000..082ef1e --- /dev/null +++ b/scripts/run_ab_evals.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Run paired no-Skill and Skill behavioral evaluations.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any + +import compare_eval_results +import run_evals + + +def _select_cases( + cases: list[dict[str, Any]], + selected_ids: list[str] | None, +) -> list[dict[str, Any]]: + if not selected_ids: + return cases + selected = set(selected_ids) + known = {case["id"] for case in cases} + unknown = sorted(selected - known) + if unknown: + raise ValueError(f"unknown cases: {', '.join(unknown)}") + return [case for case in cases if case["id"] in selected] + + +def _condition_order(run_index: int, case_index: int) -> tuple[str, str]: + if (run_index + case_index) % 2: + return ("baseline", "skill") + return ("skill", "baseline") + + +def run_experiment( + cases: list[dict[str, Any]], + *, + baseline_agent_command: str, + skill_agent_command: str, + baseline_label: str, + skill_label: str, + repeat: int, + skill_root: Path, + agent_timeout: int, + check_timeout: int, + allow_workspace_execution: bool, + keep_workspaces: bool, + workspace_parent: Path | None, + pass_env: tuple[str, ...] = (), +) -> tuple[dict[str, Any], dict[str, Any], dict[str, Any]]: + if repeat < 1: + raise ValueError("repeat must be positive") + + baseline_results: list[dict[str, Any]] = [] + skill_results: list[dict[str, Any]] = [] + execution_order: list[dict[str, Any]] = [] + + commands = { + "baseline": baseline_agent_command, + "skill": skill_agent_command, + } + result_lists = { + "baseline": baseline_results, + "skill": skill_results, + } + + for run_index in range(1, repeat + 1): + for case_index, case in enumerate(cases): + order = _condition_order(run_index, case_index) + execution_order.append( + { + "id": case["id"], + "run_index": run_index, + "order": list(order), + } + ) + for condition in order: + result = run_evals.evaluate_case( + case, + agent_command=commands[condition], + skill_root=skill_root, + agent_timeout=agent_timeout, + check_timeout=check_timeout, + allow_workspace_execution=allow_workspace_execution, + keep_workspace=keep_workspaces, + workspace_parent=workspace_parent, + pass_env=pass_env, + run_index=run_index, + ) + result_lists[condition].append(result) + + baseline_report = run_evals.build_report( + baseline_results, + adapter_label=baseline_label, + allow_workspace_execution=allow_workspace_execution, + forwarded_environment=pass_env, + ) + skill_report = run_evals.build_report( + skill_results, + adapter_label=skill_label, + allow_workspace_execution=allow_workspace_execution, + forwarded_environment=pass_env, + ) + manifest = { + "schema_version": 1, + "design": "paired-counterbalanced", + "repeat": repeat, + "baseline_adapter": baseline_label, + "skill_adapter": skill_label, + "execution_order": execution_order, + } + return baseline_report, skill_report, manifest + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description=( + "Run paired baseline and Skill-enabled evaluations with counterbalanced " + "condition order." + ) + ) + parser.add_argument( + "--cases", + type=Path, + default=run_evals.DEFAULT_CASES, + help="evaluation case file", + ) + parser.add_argument( + "--case", + action="append", + help="run only the named case; repeat to select multiple cases", + ) + parser.add_argument("--baseline-agent-command", required=True) + parser.add_argument("--skill-agent-command", required=True) + parser.add_argument("--baseline-label", default="no-skill") + parser.add_argument("--skill-label", default="skill") + parser.add_argument("--repeat", type=int, default=1) + parser.add_argument( + "--skill-root", + type=Path, + default=run_evals.ROOT, + help="source repository used to stage the runtime Skill payload", + ) + parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--agent-timeout", type=int, default=900) + parser.add_argument("--check-timeout", type=int, default=120) + parser.add_argument( + "--pass-env", + action="append", + default=[], + metavar="NAME", + help="forward one named environment variable to both agent conditions", + ) + parser.add_argument("--allow-workspace-execution", action="store_true") + parser.add_argument("--keep-workspaces", action="store_true") + parser.add_argument("--workspace-parent", type=Path) + return parser + + +def main(argv: list[str] | None = None) -> int: + parser = _parser() + args = parser.parse_args(argv) + + if args.repeat < 1 or args.repeat > 100: + parser.error("--repeat must be between 1 and 100") + if args.agent_timeout <= 0 or args.check_timeout <= 0: + parser.error("timeouts must be positive") + + invalid_env_names = [ + name for name in args.pass_env if not run_evals.ENV_NAME_RE.fullmatch(name) + ] + if invalid_env_names: + parser.error( + "invalid --pass-env names: " + ", ".join(sorted(set(invalid_env_names))) + ) + + missing_env_names = [name for name in args.pass_env if name not in run_evals.os.environ] + if missing_env_names: + parser.error( + "--pass-env variables are not set: " + + ", ".join(sorted(set(missing_env_names))) + ) + + try: + cases = run_evals.load_cases(args.cases) + except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: + parser.error(str(exc)) + + errors = run_evals.validate_cases(cases) + if errors: + for error in errors: + print(f"ERROR: {error}", file=run_evals.sys.stderr) + return 1 + + try: + cases = _select_cases(cases, args.case) + except ValueError as exc: + parser.error(str(exc)) + + if args.workspace_parent: + args.workspace_parent.mkdir(parents=True, exist_ok=True) + + baseline_report, skill_report, manifest = run_experiment( + cases, + baseline_agent_command=args.baseline_agent_command, + skill_agent_command=args.skill_agent_command, + baseline_label=args.baseline_label, + skill_label=args.skill_label, + repeat=args.repeat, + skill_root=args.skill_root, + agent_timeout=args.agent_timeout, + check_timeout=args.check_timeout, + allow_workspace_execution=args.allow_workspace_execution, + keep_workspaces=args.keep_workspaces, + workspace_parent=args.workspace_parent, + pass_env=tuple(args.pass_env), + ) + + args.output_dir.mkdir(parents=True, exist_ok=True) + baseline_path = args.output_dir / "baseline.json" + skill_path = args.output_dir / "skill.json" + manifest_path = args.output_dir / "experiment.json" + comparison_path = args.output_dir / "comparison.md" + + baseline_path.write_text( + json.dumps(baseline_report, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + skill_path.write_text( + json.dumps(skill_report, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + manifest_path.write_text( + json.dumps(manifest, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + comparison_path.write_text( + compare_eval_results.compare_reports(baseline_report, skill_report) + "\n", + encoding="utf-8", + ) + + print(f"Wrote paired A/B evidence to {args.output_dir}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From ec6c4b28f05b53881b90f21e7fc3e4e6e726b966 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:17:04 +0800 Subject: [PATCH 12/34] Keep paired runner environment handling self-contained --- scripts/run_ab_evals.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/scripts/run_ab_evals.py b/scripts/run_ab_evals.py index 082ef1e..083eb08 100644 --- a/scripts/run_ab_evals.py +++ b/scripts/run_ab_evals.py @@ -5,6 +5,8 @@ import argparse import json +import os +import sys from pathlib import Path from typing import Any @@ -174,7 +176,7 @@ def main(argv: list[str] | None = None) -> int: "invalid --pass-env names: " + ", ".join(sorted(set(invalid_env_names))) ) - missing_env_names = [name for name in args.pass_env if name not in run_evals.os.environ] + missing_env_names = [name for name in args.pass_env if name not in os.environ] if missing_env_names: parser.error( "--pass-env variables are not set: " @@ -189,7 +191,7 @@ def main(argv: list[str] | None = None) -> int: errors = run_evals.validate_cases(cases) if errors: for error in errors: - print(f"ERROR: {error}", file=run_evals.sys.stderr) + print(f"ERROR: {error}", file=sys.stderr) return 1 try: From 935b915a4b83d28b35ca32cabb994996ae8b7bb8 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:17:35 +0800 Subject: [PATCH 13/34] Test paired counterbalanced A/B runner --- tests/test_run_ab_evals.py | 108 +++++++++++++++++++++++++++++++++++++ 1 file changed, 108 insertions(+) create mode 100644 tests/test_run_ab_evals.py diff --git a/tests/test_run_ab_evals.py b/tests/test_run_ab_evals.py new file mode 100644 index 0000000..b32fe0a --- /dev/null +++ b/tests/test_run_ab_evals.py @@ -0,0 +1,108 @@ +from __future__ import annotations + +import sys +import unittest +from pathlib import Path +from unittest import mock + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import run_ab_evals + + +def sample_case(case_id: str) -> dict: + return { + "id": case_id, + "task": f"task for {case_id}", + "must_do": ["do"], + "must_not_do": ["do not"], + "fixture": {"files": {"app.py": "value = 1\n"}}, + "checks": [{"type": "changed_files_subset", "paths": ["app.py"]}], + } + + +class RunAbEvalsTests(unittest.TestCase): + def test_condition_order_is_counterbalanced(self) -> None: + self.assertEqual( + ("baseline", "skill"), + run_ab_evals._condition_order(1, 0), + ) + self.assertEqual( + ("skill", "baseline"), + run_ab_evals._condition_order(1, 1), + ) + self.assertEqual( + ("skill", "baseline"), + run_ab_evals._condition_order(2, 0), + ) + self.assertEqual( + ("baseline", "skill"), + run_ab_evals._condition_order(2, 1), + ) + + def test_run_experiment_pairs_conditions_and_records_order(self) -> None: + calls: list[tuple[str, str, int]] = [] + + def fake_evaluate(case: dict, *, agent_command: str, run_index: int, **kwargs): + calls.append((case["id"], agent_command, run_index)) + return { + "id": case["id"], + "run_index": run_index, + "status": "passed", + "task": case["task"], + "agent": {"duration_seconds": 1.0}, + "changed_files": [], + "checks": [], + "rubric": {}, + } + + with mock.patch.object( + run_ab_evals.run_evals, + "evaluate_case", + side_effect=fake_evaluate, + ): + baseline, skill, manifest = run_ab_evals.run_experiment( + [sample_case("one"), sample_case("two")], + baseline_agent_command="baseline-command", + skill_agent_command="skill-command", + baseline_label="baseline", + skill_label="skill", + repeat=2, + skill_root=ROOT, + agent_timeout=30, + check_timeout=30, + allow_workspace_execution=False, + keep_workspaces=False, + workspace_parent=None, + ) + + self.assertEqual( + [ + ("one", "baseline-command", 1), + ("one", "skill-command", 1), + ("two", "skill-command", 1), + ("two", "baseline-command", 1), + ("one", "skill-command", 2), + ("one", "baseline-command", 2), + ("two", "baseline-command", 2), + ("two", "skill-command", 2), + ], + calls, + ) + self.assertEqual(4, len(baseline["cases"])) + self.assertEqual(4, len(skill["cases"])) + self.assertEqual("paired-counterbalanced", manifest["design"]) + self.assertEqual(2, manifest["repeat"]) + self.assertEqual( + ["baseline", "skill"], + manifest["execution_order"][0]["order"], + ) + self.assertEqual( + ["skill", "baseline"], + manifest["execution_order"][1]["order"], + ) + + +if __name__ == "__main__": + unittest.main() From e10d1dacacc1646b7efe26c2cee99f074b85c2c0 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:18:17 +0800 Subject: [PATCH 14/34] Add manual real-host Skill A/B workflow --- .github/workflows/real-host-ab.yml | 179 +++++++++++++++++++++++++++++ 1 file changed, 179 insertions(+) create mode 100644 .github/workflows/real-host-ab.yml diff --git a/.github/workflows/real-host-ab.yml b/.github/workflows/real-host-ab.yml new file mode 100644 index 0000000..e1b051c --- /dev/null +++ b/.github/workflows/real-host-ab.yml @@ -0,0 +1,179 @@ +name: Real-host Skill A/B + +on: + workflow_dispatch: + inputs: + model: + description: Explicit Codex model ID + required: true + default: gpt-5.6 + type: string + codex-version: + description: Exact @openai/codex CLI version + required: true + default: 0.156.1 + type: string + repeat: + description: Independent repetitions per case and condition + required: true + default: "1" + type: choice + options: + - "1" + - "3" + - "5" + case-set: + description: Evaluation case set + required: true + default: all + type: choice + options: + - all + - high-signal + +permissions: + contents: read + +concurrency: + group: real-host-ab-${{ github.ref }}-${{ inputs.model }} + cancel-in-progress: false + +jobs: + codex-ab: + runs-on: ubuntu-latest + timeout-minutes: 180 + + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.12" + + - name: Validate experiment inputs and credentials + env: + CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} + CODEX_VERSION: ${{ inputs.codex-version }} + MODEL: ${{ inputs.model }} + run: | + set -euo pipefail + + if [[ -z "${CODEX_API_KEY}" ]]; then + echo "::error::Repository secret CODEX_API_KEY is required for a real Codex run." + exit 1 + fi + if [[ ! "${CODEX_VERSION}" =~ ^[0-9]+\.[0-9]+\.[0-9]+([.-][0-9A-Za-z.-]+)?$ ]]; then + echo "::error::codex-version must be an exact version, not latest or a range." + exit 1 + fi + if [[ ! "${MODEL}" =~ ^[A-Za-z0-9._-]+$ ]]; then + echo "::error::model contains unsupported characters." + exit 1 + fi + + - name: Install pinned Codex CLI + env: + CODEX_VERSION: ${{ inputs.codex-version }} + run: | + set -euo pipefail + npm install --global "@openai/codex@${CODEX_VERSION}" + codex --version + node --version + npm --version + + - name: Validate evaluation fixtures + run: python scripts/run_evals.py --validate-only + + - name: Run paired Skill A/B experiment + env: + CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} + MODEL: ${{ inputs.model }} + REPEAT: ${{ inputs.repeat }} + CASE_SET: ${{ inputs.case-set }} + run: | + set -euo pipefail + + case_args=() + if [[ "${CASE_SET}" == "high-signal" ]]; then + case_args+=( + --case minimal-bug-fix + --case public-contract-change + --case verification-evidence + --case security-boundary + --case dependency-addition + --case scope-expansion + ) + fi + + python scripts/run_ab_evals.py \ + --baseline-agent-command "{python} {repo}/scripts/host_eval_adapter.py codex --model ${MODEL} --skill-mode disabled" \ + --skill-agent-command "{python} {repo}/scripts/host_eval_adapter.py codex --model ${MODEL} --skill-mode enabled" \ + --baseline-label "codex-${MODEL}-no-skill" \ + --skill-label "codex-${MODEL}-skill" \ + --repeat "${REPEAT}" \ + --pass-env CODEX_API_KEY \ + --allow-workspace-execution \ + --output-dir eval-results \ + "${case_args[@]}" + + - name: Record experiment metadata + if: always() + env: + MODEL: ${{ inputs.model }} + CODEX_VERSION: ${{ inputs.codex-version }} + REPEAT: ${{ inputs.repeat }} + CASE_SET: ${{ inputs.case-set }} + run: | + set -euo pipefail + mkdir -p eval-results + python - <<'PY' + import json + import os + import subprocess + from pathlib import Path + + def version(command): + completed = subprocess.run( + command, + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + ) + return completed.stdout.strip() + + metadata = { + "git_sha": os.environ["GITHUB_SHA"], + "repository": os.environ["GITHUB_REPOSITORY"], + "model": os.environ["MODEL"], + "requested_codex_version": os.environ["CODEX_VERSION"], + "observed_codex_version": version(["codex", "--version"]), + "python_version": version(["python", "--version"]), + "repeat": int(os.environ["REPEAT"]), + "case_set": os.environ["CASE_SET"], + } + Path("eval-results/metadata.json").write_text( + json.dumps(metadata, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + PY + + - name: Add comparison to job summary + if: always() + run: | + if [[ -f eval-results/comparison.md ]]; then + cat eval-results/comparison.md >> "${GITHUB_STEP_SUMMARY}" + else + echo "No comparison report was produced." >> "${GITHUB_STEP_SUMMARY}" + fi + + - name: Upload A/B evidence + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: codex-skill-ab-${{ inputs.model }}-${{ github.run_id }} + path: eval-results/ + if-no-files-found: error + retention-days: 7 From c26fe7d1939dc82be1c25cfe35d67dd39f42841f Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:18:50 +0800 Subject: [PATCH 15/34] Harden real-host workflow inputs and diagnostics --- .github/workflows/real-host-ab.yml | 33 ++++++++++++++++-------------- 1 file changed, 18 insertions(+), 15 deletions(-) diff --git a/.github/workflows/real-host-ab.yml b/.github/workflows/real-host-ab.yml index e1b051c..57c6e32 100644 --- a/.github/workflows/real-host-ab.yml +++ b/.github/workflows/real-host-ab.yml @@ -8,7 +8,7 @@ on: required: true default: gpt-5.6 type: string - codex-version: + codex_version: description: Exact @openai/codex CLI version required: true default: 0.156.1 @@ -22,7 +22,7 @@ on: - "1" - "3" - "5" - case-set: + case_set: description: Evaluation case set required: true default: all @@ -55,7 +55,7 @@ jobs: - name: Validate experiment inputs and credentials env: CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} - CODEX_VERSION: ${{ inputs.codex-version }} + CODEX_VERSION: ${{ inputs.codex_version }} MODEL: ${{ inputs.model }} run: | set -euo pipefail @@ -65,7 +65,7 @@ jobs: exit 1 fi if [[ ! "${CODEX_VERSION}" =~ ^[0-9]+\.[0-9]+\.[0-9]+([.-][0-9A-Za-z.-]+)?$ ]]; then - echo "::error::codex-version must be an exact version, not latest or a range." + echo "::error::codex_version must be an exact version, not latest or a range." exit 1 fi if [[ ! "${MODEL}" =~ ^[A-Za-z0-9._-]+$ ]]; then @@ -75,7 +75,7 @@ jobs: - name: Install pinned Codex CLI env: - CODEX_VERSION: ${{ inputs.codex-version }} + CODEX_VERSION: ${{ inputs.codex_version }} run: | set -euo pipefail npm install --global "@openai/codex@${CODEX_VERSION}" @@ -91,7 +91,7 @@ jobs: CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} MODEL: ${{ inputs.model }} REPEAT: ${{ inputs.repeat }} - CASE_SET: ${{ inputs.case-set }} + CASE_SET: ${{ inputs.case_set }} run: | set -euo pipefail @@ -122,9 +122,9 @@ jobs: if: always() env: MODEL: ${{ inputs.model }} - CODEX_VERSION: ${{ inputs.codex-version }} + CODEX_VERSION: ${{ inputs.codex_version }} REPEAT: ${{ inputs.repeat }} - CASE_SET: ${{ inputs.case-set }} + CASE_SET: ${{ inputs.case_set }} run: | set -euo pipefail mkdir -p eval-results @@ -135,13 +135,16 @@ jobs: from pathlib import Path def version(command): - completed = subprocess.run( - command, - check=False, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - ) + try: + completed = subprocess.run( + command, + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + ) + except FileNotFoundError: + return "not installed" return completed.stdout.strip() metadata = { From f9ba7300aa36c344b36e6dcf63d35284294ea476 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:19:28 +0800 Subject: [PATCH 16/34] Document paired real-host A/B phase --- evals/README.md | 57 ++++++++++++++++++++++++++++--------------------- 1 file changed, 33 insertions(+), 24 deletions(-) diff --git a/evals/README.md b/evals/README.md index 25756e5..68ace86 100644 --- a/evals/README.md +++ b/evals/README.md @@ -132,42 +132,51 @@ Each case receives a fresh runtime Skill staging directory. The harness hashes i The native adapter can run the same task without exposing the staged Skill. The baseline is created by removing the Skill from the host environment, not by adding a prompt that tells the model to ignore it. -Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. Use `--repeat` when you need repeated independent trials; every repetition receives a fresh fixture workspace. +Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. -Codex example: +For comparative runs, prefer the paired runner. It executes the two conditions adjacent to each other for every case and alternates which condition goes first across cases and repetitions. This reduces systematic time/order bias relative to running one complete condition and then the other. ```bash -python scripts/run_evals.py \ - --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ - --adapter-label codex-MODEL-no-skill \ - --pass-env CODEX_API_KEY \ - --repeat 5 \ - --allow-workspace-execution \ - --output eval-results/codex-MODEL-no-skill.json - -python scripts/run_evals.py \ - --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ - --adapter-label codex-MODEL-skill \ +python scripts/run_ab_evals.py \ + --baseline-agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --skill-agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --baseline-label codex-MODEL-no-skill \ + --skill-label codex-MODEL-skill \ --pass-env CODEX_API_KEY \ --repeat 5 \ --allow-workspace-execution \ - --output eval-results/codex-MODEL-skill.json + --output-dir eval-results/codex-MODEL ``` -Claude Code uses the same `--skill-mode disabled|enabled` switch on `host_eval_adapter.py`. +The output directory contains: -Compare the two reports: +- `baseline.json` — raw no-Skill evidence, +- `skill.json` — raw Skill-enabled evidence, +- `experiment.json` — pairing and execution-order metadata, +- `comparison.md` — deterministic case/check rates, diff scope, and wall-clock comparison. -```bash -python scripts/compare_eval_results.py \ - eval-results/codex-MODEL-no-skill.json \ - eval-results/codex-MODEL-skill.json \ - --output eval-results/codex-MODEL-comparison.md -``` +`scripts/compare_eval_results.py` remains available for comparing two previously produced reports. It rejects mismatched case sets, task drift, duplicate run indexes, and unpaired repetition sets. + +### Manual real-host smoke workflow + +`.github/workflows/real-host-ab.yml` provides a manual Codex smoke path once that workflow is present on the repository default branch. + +Its defaults are deliberately explicit rather than floating: + +- model: `gpt-5.6`, +- Codex CLI: `0.156.1`, +- case set: all 14 scenarios, +- repeat: 1, producing 28 agent runs. + +The workflow also offers a six-case `high-signal` subset and repeat counts of 3 or 5. Larger repetitions increase API usage substantially and must be selected explicitly. + +The workflow requires a repository secret named `CODEX_API_KEY`. Use a credential intended for unattended Codex evaluation rather than a broad personal credential. The runner forwards only that named variable to the host adapter; deterministic post-run command checks receive a narrower environment without the key. + +The workflow records requested and observed Codex versions, model ID, repository SHA, repeat count, and case set in `metadata.json`, uploads the raw evidence for seven days, and writes the comparison table to the GitHub Actions job summary. -The comparison reports deterministic case pass rates, deterministic check-type pass rates, mean changed-file counts, and mean agent wall-clock time. It deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is excluded from comparative check rates. +The comparison deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is also excluded from comparative check rates. -For publishable evidence, run both conditions close enough together to reduce host/model drift, retain the raw JSON reports, and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. +For publishable evidence, retain the raw JSON reports and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. See [host compatibility](../docs/compatibility.md) for the dated vendor documentation basis. From 69b2bbd829ce985c403f01dfc87901daee5386b1 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:19:50 +0800 Subject: [PATCH 17/34] Require A/B evaluation tooling in repository validation --- scripts/validate_skill.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/scripts/validate_skill.py b/scripts/validate_skill.py index 8a76b36..dfe32a2 100755 --- a/scripts/validate_skill.py +++ b/scripts/validate_skill.py @@ -28,6 +28,7 @@ "GOVERNANCE.md", "SECURITY.md", ".github/dependabot.yml", + ".github/workflows/real-host-ab.yml", ".github/CODEOWNERS", ".github/PULL_REQUEST_TEMPLATE.md", ".github/ISSUE_TEMPLATE/config.yml", @@ -57,6 +58,8 @@ "docs/release.md", "scripts/project_checks.py", "scripts/run_evals.py", + "scripts/run_ab_evals.py", + "scripts/compare_eval_results.py", "scripts/host_eval_adapter.py", "scripts/package_skill.py", "scripts/release_check.py", From 6c6a2ab6bb4746cc27fc0c3add7cedde848fbc0a Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:19 +0800 Subject: [PATCH 18/34] Revert real-host experiment additions in scripts/compare_eval_results.py --- scripts/compare_eval_results.py | 18 +----------------- 1 file changed, 1 insertion(+), 17 deletions(-) diff --git a/scripts/compare_eval_results.py b/scripts/compare_eval_results.py index 6f3743a..8e1857c 100644 --- a/scripts/compare_eval_results.py +++ b/scripts/compare_eval_results.py @@ -31,13 +31,8 @@ def _group_cases(report: dict[str, Any]) -> dict[str, list[dict[str, Any]]]: raise ValueError(f"case[{index}].id must be a non-empty string") grouped.setdefault(case_id, []).append(case) - for case_id, runs in grouped.items(): + for runs in grouped.values(): runs.sort(key=lambda item: item.get("run_index", 1)) - indexes = [run.get("run_index", 1) for run in runs] - if not all(isinstance(index, int) and index >= 1 for index in indexes): - raise ValueError(f"{case_id}: run_index must be a positive integer") - if len(indexes) != len(set(indexes)): - raise ValueError(f"{case_id}: duplicate run_index values") return grouped @@ -146,17 +141,6 @@ def compare_reports( skill_task = _task_for(skill_grouped[case_id], case_id=case_id) if baseline_task != skill_task: raise ValueError(f"{case_id}: baseline and Skill tasks differ") - baseline_indexes = [ - run.get("run_index", 1) for run in baseline_grouped[case_id] - ] - skill_indexes = [ - run.get("run_index", 1) for run in skill_grouped[case_id] - ] - if baseline_indexes != skill_indexes: - raise ValueError( - f"{case_id}: baseline and Skill repetition sets differ " - f"({baseline_indexes} != {skill_indexes})" - ) baseline_runs = [ run for case_id in sorted(baseline_grouped) for run in baseline_grouped[case_id] From e1a2cae52215a7c3d03d38b14aac7bf49eb28fdc Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:21 +0800 Subject: [PATCH 19/34] Revert real-host experiment additions in tests/test_compare_eval_results.py --- tests/test_compare_eval_results.py | 60 ------------------------------ 1 file changed, 60 deletions(-) diff --git a/tests/test_compare_eval_results.py b/tests/test_compare_eval_results.py index a99b144..dedef90 100644 --- a/tests/test_compare_eval_results.py +++ b/tests/test_compare_eval_results.py @@ -114,66 +114,6 @@ def test_comparison_rejects_different_case_sets(self) -> None: with self.assertRaisesRegex(ValueError, "case sets differ"): compare_eval_results.compare_reports(baseline, skill) - def test_comparison_rejects_unpaired_repetitions(self) -> None: - baseline = { - "adapter": "baseline", - "cases": [ - make_run( - "sample", - status="passed", - run_index=1, - changed_files=1, - duration=1.0, - ), - make_run( - "sample", - status="passed", - run_index=2, - changed_files=1, - duration=1.0, - ), - ], - } - skill = { - "adapter": "skill", - "cases": [ - make_run( - "sample", - status="passed", - run_index=1, - changed_files=1, - duration=1.0, - ) - ], - } - - with self.assertRaisesRegex(ValueError, "repetition sets differ"): - compare_eval_results.compare_reports(baseline, skill) - - def test_comparison_rejects_duplicate_run_indexes(self) -> None: - duplicate_runs = [ - make_run( - "sample", - status="passed", - run_index=1, - changed_files=1, - duration=1.0, - ), - make_run( - "sample", - status="passed", - run_index=1, - changed_files=1, - duration=1.0, - ), - ] - - with self.assertRaisesRegex(ValueError, "duplicate run_index"): - compare_eval_results.compare_reports( - {"adapter": "baseline", "cases": duplicate_runs}, - {"adapter": "skill", "cases": duplicate_runs}, - ) - def test_comparison_rejects_task_drift(self) -> None: baseline_run = make_run( "sample", From f3e1d6572e6d9da739c51a9dc8f9366bf00838e0 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:24 +0800 Subject: [PATCH 20/34] Revert real-host experiment additions in evals/README.md --- evals/README.md | 57 +++++++++++++++++++++---------------------------- 1 file changed, 24 insertions(+), 33 deletions(-) diff --git a/evals/README.md b/evals/README.md index 68ace86..25756e5 100644 --- a/evals/README.md +++ b/evals/README.md @@ -132,51 +132,42 @@ Each case receives a fresh runtime Skill staging directory. The harness hashes i The native adapter can run the same task without exposing the staged Skill. The baseline is created by removing the Skill from the host environment, not by adding a prompt that tells the model to ignore it. -Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. +Keep the host, explicit model ID, case set, credentials, execution flags, and task text identical between conditions. Use `--repeat` when you need repeated independent trials; every repetition receives a fresh fixture workspace. -For comparative runs, prefer the paired runner. It executes the two conditions adjacent to each other for every case and alternates which condition goes first across cases and repetitions. This reduces systematic time/order bias relative to running one complete condition and then the other. +Codex example: ```bash -python scripts/run_ab_evals.py \ - --baseline-agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ - --skill-agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ - --baseline-label codex-MODEL-no-skill \ - --skill-label codex-MODEL-skill \ +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --adapter-label codex-MODEL-no-skill \ --pass-env CODEX_API_KEY \ --repeat 5 \ --allow-workspace-execution \ - --output-dir eval-results/codex-MODEL -``` - -The output directory contains: - -- `baseline.json` — raw no-Skill evidence, -- `skill.json` — raw Skill-enabled evidence, -- `experiment.json` — pairing and execution-order metadata, -- `comparison.md` — deterministic case/check rates, diff scope, and wall-clock comparison. - -`scripts/compare_eval_results.py` remains available for comparing two previously produced reports. It rejects mismatched case sets, task drift, duplicate run indexes, and unpaired repetition sets. - -### Manual real-host smoke workflow - -`.github/workflows/real-host-ab.yml` provides a manual Codex smoke path once that workflow is present on the repository default branch. - -Its defaults are deliberately explicit rather than floating: + --output eval-results/codex-MODEL-no-skill.json -- model: `gpt-5.6`, -- Codex CLI: `0.156.1`, -- case set: all 14 scenarios, -- repeat: 1, producing 28 agent runs. +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --adapter-label codex-MODEL-skill \ + --pass-env CODEX_API_KEY \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/codex-MODEL-skill.json +``` -The workflow also offers a six-case `high-signal` subset and repeat counts of 3 or 5. Larger repetitions increase API usage substantially and must be selected explicitly. +Claude Code uses the same `--skill-mode disabled|enabled` switch on `host_eval_adapter.py`. -The workflow requires a repository secret named `CODEX_API_KEY`. Use a credential intended for unattended Codex evaluation rather than a broad personal credential. The runner forwards only that named variable to the host adapter; deterministic post-run command checks receive a narrower environment without the key. +Compare the two reports: -The workflow records requested and observed Codex versions, model ID, repository SHA, repeat count, and case set in `metadata.json`, uploads the raw evidence for seven days, and writes the comparison table to the GitHub Actions job summary. +```bash +python scripts/compare_eval_results.py \ + eval-results/codex-MODEL-no-skill.json \ + eval-results/codex-MODEL-skill.json \ + --output eval-results/codex-MODEL-comparison.md +``` -The comparison deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is also excluded from comparative check rates. +The comparison reports deterministic case pass rates, deterministic check-type pass rates, mean changed-file counts, and mean agent wall-clock time. It deliberately does not convert the qualitative `must_do` / `must_not_do` rubric into an automatic score. The infrastructure-only `skill_payload_integrity` check is excluded from comparative check rates. -For publishable evidence, retain the raw JSON reports and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. +For publishable evidence, run both conditions close enough together to reduce host/model drift, retain the raw JSON reports, and review qualitative rubric items separately. Token usage is not currently normalized across host CLIs, so wall-clock time is the portable cost signal recorded by the harness. See [host compatibility](../docs/compatibility.md) for the dated vendor documentation basis. From 424c520814d4def6789750f792e450a875656524 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:26 +0800 Subject: [PATCH 21/34] Revert real-host experiment additions in scripts/validate_skill.py --- scripts/validate_skill.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/scripts/validate_skill.py b/scripts/validate_skill.py index dfe32a2..8a76b36 100755 --- a/scripts/validate_skill.py +++ b/scripts/validate_skill.py @@ -28,7 +28,6 @@ "GOVERNANCE.md", "SECURITY.md", ".github/dependabot.yml", - ".github/workflows/real-host-ab.yml", ".github/CODEOWNERS", ".github/PULL_REQUEST_TEMPLATE.md", ".github/ISSUE_TEMPLATE/config.yml", @@ -58,8 +57,6 @@ "docs/release.md", "scripts/project_checks.py", "scripts/run_evals.py", - "scripts/run_ab_evals.py", - "scripts/compare_eval_results.py", "scripts/host_eval_adapter.py", "scripts/package_skill.py", "scripts/release_check.py", From 1f4e5681a3bfd6c2372a1dbf6512996d6d109316 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:28 +0800 Subject: [PATCH 22/34] Remove real-host experiment file .github/workflows/real-host-ab.yml --- .github/workflows/real-host-ab.yml | 182 ----------------------------- 1 file changed, 182 deletions(-) delete mode 100644 .github/workflows/real-host-ab.yml diff --git a/.github/workflows/real-host-ab.yml b/.github/workflows/real-host-ab.yml deleted file mode 100644 index 57c6e32..0000000 --- a/.github/workflows/real-host-ab.yml +++ /dev/null @@ -1,182 +0,0 @@ -name: Real-host Skill A/B - -on: - workflow_dispatch: - inputs: - model: - description: Explicit Codex model ID - required: true - default: gpt-5.6 - type: string - codex_version: - description: Exact @openai/codex CLI version - required: true - default: 0.156.1 - type: string - repeat: - description: Independent repetitions per case and condition - required: true - default: "1" - type: choice - options: - - "1" - - "3" - - "5" - case_set: - description: Evaluation case set - required: true - default: all - type: choice - options: - - all - - high-signal - -permissions: - contents: read - -concurrency: - group: real-host-ab-${{ github.ref }}-${{ inputs.model }} - cancel-in-progress: false - -jobs: - codex-ab: - runs-on: ubuntu-latest - timeout-minutes: 180 - - steps: - - name: Checkout - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - - name: Set up Python - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: "3.12" - - - name: Validate experiment inputs and credentials - env: - CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} - CODEX_VERSION: ${{ inputs.codex_version }} - MODEL: ${{ inputs.model }} - run: | - set -euo pipefail - - if [[ -z "${CODEX_API_KEY}" ]]; then - echo "::error::Repository secret CODEX_API_KEY is required for a real Codex run." - exit 1 - fi - if [[ ! "${CODEX_VERSION}" =~ ^[0-9]+\.[0-9]+\.[0-9]+([.-][0-9A-Za-z.-]+)?$ ]]; then - echo "::error::codex_version must be an exact version, not latest or a range." - exit 1 - fi - if [[ ! "${MODEL}" =~ ^[A-Za-z0-9._-]+$ ]]; then - echo "::error::model contains unsupported characters." - exit 1 - fi - - - name: Install pinned Codex CLI - env: - CODEX_VERSION: ${{ inputs.codex_version }} - run: | - set -euo pipefail - npm install --global "@openai/codex@${CODEX_VERSION}" - codex --version - node --version - npm --version - - - name: Validate evaluation fixtures - run: python scripts/run_evals.py --validate-only - - - name: Run paired Skill A/B experiment - env: - CODEX_API_KEY: ${{ secrets.CODEX_API_KEY }} - MODEL: ${{ inputs.model }} - REPEAT: ${{ inputs.repeat }} - CASE_SET: ${{ inputs.case_set }} - run: | - set -euo pipefail - - case_args=() - if [[ "${CASE_SET}" == "high-signal" ]]; then - case_args+=( - --case minimal-bug-fix - --case public-contract-change - --case verification-evidence - --case security-boundary - --case dependency-addition - --case scope-expansion - ) - fi - - python scripts/run_ab_evals.py \ - --baseline-agent-command "{python} {repo}/scripts/host_eval_adapter.py codex --model ${MODEL} --skill-mode disabled" \ - --skill-agent-command "{python} {repo}/scripts/host_eval_adapter.py codex --model ${MODEL} --skill-mode enabled" \ - --baseline-label "codex-${MODEL}-no-skill" \ - --skill-label "codex-${MODEL}-skill" \ - --repeat "${REPEAT}" \ - --pass-env CODEX_API_KEY \ - --allow-workspace-execution \ - --output-dir eval-results \ - "${case_args[@]}" - - - name: Record experiment metadata - if: always() - env: - MODEL: ${{ inputs.model }} - CODEX_VERSION: ${{ inputs.codex_version }} - REPEAT: ${{ inputs.repeat }} - CASE_SET: ${{ inputs.case_set }} - run: | - set -euo pipefail - mkdir -p eval-results - python - <<'PY' - import json - import os - import subprocess - from pathlib import Path - - def version(command): - try: - completed = subprocess.run( - command, - check=False, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - ) - except FileNotFoundError: - return "not installed" - return completed.stdout.strip() - - metadata = { - "git_sha": os.environ["GITHUB_SHA"], - "repository": os.environ["GITHUB_REPOSITORY"], - "model": os.environ["MODEL"], - "requested_codex_version": os.environ["CODEX_VERSION"], - "observed_codex_version": version(["codex", "--version"]), - "python_version": version(["python", "--version"]), - "repeat": int(os.environ["REPEAT"]), - "case_set": os.environ["CASE_SET"], - } - Path("eval-results/metadata.json").write_text( - json.dumps(metadata, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - PY - - - name: Add comparison to job summary - if: always() - run: | - if [[ -f eval-results/comparison.md ]]; then - cat eval-results/comparison.md >> "${GITHUB_STEP_SUMMARY}" - else - echo "No comparison report was produced." >> "${GITHUB_STEP_SUMMARY}" - fi - - - name: Upload A/B evidence - if: always() - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: codex-skill-ab-${{ inputs.model }}-${{ github.run_id }} - path: eval-results/ - if-no-files-found: error - retention-days: 7 From 4cc5330d1ae84addc46d815944f33b484f54f0b7 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:31 +0800 Subject: [PATCH 23/34] Remove real-host experiment file scripts/run_ab_evals.py --- scripts/run_ab_evals.py | 249 ---------------------------------------- 1 file changed, 249 deletions(-) delete mode 100644 scripts/run_ab_evals.py diff --git a/scripts/run_ab_evals.py b/scripts/run_ab_evals.py deleted file mode 100644 index 083eb08..0000000 --- a/scripts/run_ab_evals.py +++ /dev/null @@ -1,249 +0,0 @@ -#!/usr/bin/env python3 -"""Run paired no-Skill and Skill behavioral evaluations.""" - -from __future__ import annotations - -import argparse -import json -import os -import sys -from pathlib import Path -from typing import Any - -import compare_eval_results -import run_evals - - -def _select_cases( - cases: list[dict[str, Any]], - selected_ids: list[str] | None, -) -> list[dict[str, Any]]: - if not selected_ids: - return cases - selected = set(selected_ids) - known = {case["id"] for case in cases} - unknown = sorted(selected - known) - if unknown: - raise ValueError(f"unknown cases: {', '.join(unknown)}") - return [case for case in cases if case["id"] in selected] - - -def _condition_order(run_index: int, case_index: int) -> tuple[str, str]: - if (run_index + case_index) % 2: - return ("baseline", "skill") - return ("skill", "baseline") - - -def run_experiment( - cases: list[dict[str, Any]], - *, - baseline_agent_command: str, - skill_agent_command: str, - baseline_label: str, - skill_label: str, - repeat: int, - skill_root: Path, - agent_timeout: int, - check_timeout: int, - allow_workspace_execution: bool, - keep_workspaces: bool, - workspace_parent: Path | None, - pass_env: tuple[str, ...] = (), -) -> tuple[dict[str, Any], dict[str, Any], dict[str, Any]]: - if repeat < 1: - raise ValueError("repeat must be positive") - - baseline_results: list[dict[str, Any]] = [] - skill_results: list[dict[str, Any]] = [] - execution_order: list[dict[str, Any]] = [] - - commands = { - "baseline": baseline_agent_command, - "skill": skill_agent_command, - } - result_lists = { - "baseline": baseline_results, - "skill": skill_results, - } - - for run_index in range(1, repeat + 1): - for case_index, case in enumerate(cases): - order = _condition_order(run_index, case_index) - execution_order.append( - { - "id": case["id"], - "run_index": run_index, - "order": list(order), - } - ) - for condition in order: - result = run_evals.evaluate_case( - case, - agent_command=commands[condition], - skill_root=skill_root, - agent_timeout=agent_timeout, - check_timeout=check_timeout, - allow_workspace_execution=allow_workspace_execution, - keep_workspace=keep_workspaces, - workspace_parent=workspace_parent, - pass_env=pass_env, - run_index=run_index, - ) - result_lists[condition].append(result) - - baseline_report = run_evals.build_report( - baseline_results, - adapter_label=baseline_label, - allow_workspace_execution=allow_workspace_execution, - forwarded_environment=pass_env, - ) - skill_report = run_evals.build_report( - skill_results, - adapter_label=skill_label, - allow_workspace_execution=allow_workspace_execution, - forwarded_environment=pass_env, - ) - manifest = { - "schema_version": 1, - "design": "paired-counterbalanced", - "repeat": repeat, - "baseline_adapter": baseline_label, - "skill_adapter": skill_label, - "execution_order": execution_order, - } - return baseline_report, skill_report, manifest - - -def _parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser( - description=( - "Run paired baseline and Skill-enabled evaluations with counterbalanced " - "condition order." - ) - ) - parser.add_argument( - "--cases", - type=Path, - default=run_evals.DEFAULT_CASES, - help="evaluation case file", - ) - parser.add_argument( - "--case", - action="append", - help="run only the named case; repeat to select multiple cases", - ) - parser.add_argument("--baseline-agent-command", required=True) - parser.add_argument("--skill-agent-command", required=True) - parser.add_argument("--baseline-label", default="no-skill") - parser.add_argument("--skill-label", default="skill") - parser.add_argument("--repeat", type=int, default=1) - parser.add_argument( - "--skill-root", - type=Path, - default=run_evals.ROOT, - help="source repository used to stage the runtime Skill payload", - ) - parser.add_argument("--output-dir", type=Path, required=True) - parser.add_argument("--agent-timeout", type=int, default=900) - parser.add_argument("--check-timeout", type=int, default=120) - parser.add_argument( - "--pass-env", - action="append", - default=[], - metavar="NAME", - help="forward one named environment variable to both agent conditions", - ) - parser.add_argument("--allow-workspace-execution", action="store_true") - parser.add_argument("--keep-workspaces", action="store_true") - parser.add_argument("--workspace-parent", type=Path) - return parser - - -def main(argv: list[str] | None = None) -> int: - parser = _parser() - args = parser.parse_args(argv) - - if args.repeat < 1 or args.repeat > 100: - parser.error("--repeat must be between 1 and 100") - if args.agent_timeout <= 0 or args.check_timeout <= 0: - parser.error("timeouts must be positive") - - invalid_env_names = [ - name for name in args.pass_env if not run_evals.ENV_NAME_RE.fullmatch(name) - ] - if invalid_env_names: - parser.error( - "invalid --pass-env names: " + ", ".join(sorted(set(invalid_env_names))) - ) - - missing_env_names = [name for name in args.pass_env if name not in os.environ] - if missing_env_names: - parser.error( - "--pass-env variables are not set: " - + ", ".join(sorted(set(missing_env_names))) - ) - - try: - cases = run_evals.load_cases(args.cases) - except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: - parser.error(str(exc)) - - errors = run_evals.validate_cases(cases) - if errors: - for error in errors: - print(f"ERROR: {error}", file=sys.stderr) - return 1 - - try: - cases = _select_cases(cases, args.case) - except ValueError as exc: - parser.error(str(exc)) - - if args.workspace_parent: - args.workspace_parent.mkdir(parents=True, exist_ok=True) - - baseline_report, skill_report, manifest = run_experiment( - cases, - baseline_agent_command=args.baseline_agent_command, - skill_agent_command=args.skill_agent_command, - baseline_label=args.baseline_label, - skill_label=args.skill_label, - repeat=args.repeat, - skill_root=args.skill_root, - agent_timeout=args.agent_timeout, - check_timeout=args.check_timeout, - allow_workspace_execution=args.allow_workspace_execution, - keep_workspaces=args.keep_workspaces, - workspace_parent=args.workspace_parent, - pass_env=tuple(args.pass_env), - ) - - args.output_dir.mkdir(parents=True, exist_ok=True) - baseline_path = args.output_dir / "baseline.json" - skill_path = args.output_dir / "skill.json" - manifest_path = args.output_dir / "experiment.json" - comparison_path = args.output_dir / "comparison.md" - - baseline_path.write_text( - json.dumps(baseline_report, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - skill_path.write_text( - json.dumps(skill_report, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - manifest_path.write_text( - json.dumps(manifest, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - comparison_path.write_text( - compare_eval_results.compare_reports(baseline_report, skill_report) + "\n", - encoding="utf-8", - ) - - print(f"Wrote paired A/B evidence to {args.output_dir}") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) From 7041e3aadea889714abab9b1ee30f5cc9090b7cf Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 13:29:35 +0800 Subject: [PATCH 24/34] Remove real-host experiment file tests/test_run_ab_evals.py --- tests/test_run_ab_evals.py | 108 ------------------------------------- 1 file changed, 108 deletions(-) delete mode 100644 tests/test_run_ab_evals.py diff --git a/tests/test_run_ab_evals.py b/tests/test_run_ab_evals.py deleted file mode 100644 index b32fe0a..0000000 --- a/tests/test_run_ab_evals.py +++ /dev/null @@ -1,108 +0,0 @@ -from __future__ import annotations - -import sys -import unittest -from pathlib import Path -from unittest import mock - -ROOT = Path(__file__).resolve().parents[1] -sys.path.insert(0, str(ROOT / "scripts")) - -import run_ab_evals - - -def sample_case(case_id: str) -> dict: - return { - "id": case_id, - "task": f"task for {case_id}", - "must_do": ["do"], - "must_not_do": ["do not"], - "fixture": {"files": {"app.py": "value = 1\n"}}, - "checks": [{"type": "changed_files_subset", "paths": ["app.py"]}], - } - - -class RunAbEvalsTests(unittest.TestCase): - def test_condition_order_is_counterbalanced(self) -> None: - self.assertEqual( - ("baseline", "skill"), - run_ab_evals._condition_order(1, 0), - ) - self.assertEqual( - ("skill", "baseline"), - run_ab_evals._condition_order(1, 1), - ) - self.assertEqual( - ("skill", "baseline"), - run_ab_evals._condition_order(2, 0), - ) - self.assertEqual( - ("baseline", "skill"), - run_ab_evals._condition_order(2, 1), - ) - - def test_run_experiment_pairs_conditions_and_records_order(self) -> None: - calls: list[tuple[str, str, int]] = [] - - def fake_evaluate(case: dict, *, agent_command: str, run_index: int, **kwargs): - calls.append((case["id"], agent_command, run_index)) - return { - "id": case["id"], - "run_index": run_index, - "status": "passed", - "task": case["task"], - "agent": {"duration_seconds": 1.0}, - "changed_files": [], - "checks": [], - "rubric": {}, - } - - with mock.patch.object( - run_ab_evals.run_evals, - "evaluate_case", - side_effect=fake_evaluate, - ): - baseline, skill, manifest = run_ab_evals.run_experiment( - [sample_case("one"), sample_case("two")], - baseline_agent_command="baseline-command", - skill_agent_command="skill-command", - baseline_label="baseline", - skill_label="skill", - repeat=2, - skill_root=ROOT, - agent_timeout=30, - check_timeout=30, - allow_workspace_execution=False, - keep_workspaces=False, - workspace_parent=None, - ) - - self.assertEqual( - [ - ("one", "baseline-command", 1), - ("one", "skill-command", 1), - ("two", "skill-command", 1), - ("two", "baseline-command", 1), - ("one", "skill-command", 2), - ("one", "baseline-command", 2), - ("two", "baseline-command", 2), - ("two", "skill-command", 2), - ], - calls, - ) - self.assertEqual(4, len(baseline["cases"])) - self.assertEqual(4, len(skill["cases"])) - self.assertEqual("paired-counterbalanced", manifest["design"]) - self.assertEqual(2, manifest["repeat"]) - self.assertEqual( - ["baseline", "skill"], - manifest["execution_order"][0]["order"], - ) - self.assertEqual( - ["skill", "baseline"], - manifest["execution_order"][1]["order"], - ) - - -if __name__ == "__main__": - unittest.main() From ed1dbbcbb15efff30c67106357b05d4e796d4c8b Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:09:06 +0800 Subject: [PATCH 25/34] Add semantic claim and generator-drift checks --- scripts/run_evals.py | 47 ++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 45 insertions(+), 2 deletions(-) diff --git a/scripts/run_evals.py b/scripts/run_evals.py index 9bd6c31..70db68d 100644 --- a/scripts/run_evals.py +++ b/scripts/run_evals.py @@ -49,6 +49,7 @@ "changed_files_include", "changed_files_subset", "command", + "command_no_changes", "file_absent", "file_contains", "file_exists", @@ -56,6 +57,7 @@ "file_unchanged", "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", } @@ -150,7 +152,7 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: errors.append(f"{check_label}: unsupported check type {check_type!r}") continue - if check_type == "command": + if check_type in {"command", "command_no_changes"}: argv = check.get("argv") if not isinstance(argv, list) or not argv or not all( isinstance(item, str) and item for item in argv @@ -183,6 +185,7 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: elif check_type in { "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", }: terms = check.get("terms") @@ -367,6 +370,32 @@ def _read_optional(path: Path) -> str | None: return None +NEGATION_PREFIX_RE = re.compile( + r"(?:\\bnot\\b|\\bnever\\b|\\bwithout\\b|\\bcannot\\b|\\bcan['’]?t\\b|" + r"\\bisn['’]?t\\b|\\bwasn['’]?t\\b|\\baren['’]?t\\b|\\bweren['’]?t\\b|" + r"\\bcouldn['’]?t\\b|\\bshouldn['’]?t\\b|\\bwouldn['’]?t\\b)" + r"(?:\\W+\\w+){0,2}\\W*$", + re.IGNORECASE, +) + + +def _unnegated_term_matches(text: str, terms: list[str]) -> list[str]: + matches: list[str] = [] + for term in terms: + pattern = re.compile(re.escape(term), re.IGNORECASE) + for match in pattern.finditer(text): + prefix = text[max(0, match.start() - 80):match.start()] + prefix = re.split(r"[.!?;\\n]", prefix)[-1] + if re.search(r"\\bnot\\s+only\\W*$", prefix, re.IGNORECASE): + matches.append(term) + break + if NEGATION_PREFIX_RE.search(prefix): + continue + matches.append(term) + break + return matches + + def evaluate_check( check: dict[str, Any], *, @@ -440,6 +469,7 @@ def evaluate_check( if check_type in { "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any", }: haystack = final_output.casefold() @@ -449,6 +479,9 @@ def evaluate_check( passed = len(matches) == len(terms) elif check_type == "final_contains_any": passed = bool(matches) + elif check_type == "final_not_claim_any": + matches = _unnegated_term_matches(final_output, check["terms"]) + passed = not matches else: passed = not matches result.update( @@ -460,7 +493,7 @@ def evaluate_check( ) return result - if check_type == "command": + if check_type in {"command", "command_no_changes"}: if not allow_workspace_execution: result.update( { @@ -484,6 +517,7 @@ def evaluate_check( ) as home_directory: check_env = _isolated_environment(home=Path(home_directory)) for _ in range(repeat): + execution_before = snapshot_workspace(workspace) try: completed = subprocess.run( argv, @@ -502,6 +536,15 @@ def evaluate_check( } if completed.returncode != expected_exit: passed = False + if check_type == "command_no_changes": + execution_after = snapshot_workspace(workspace) + generated_changes = changed_files( + execution_before, + execution_after, + ) + execution["changed_files"] = generated_changes + if generated_changes: + passed = False except subprocess.TimeoutExpired as exc: execution = { "exit_code": 124, From cb4f43bb4b12607e8aa2870af65252f0f8aad554 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:09:14 +0800 Subject: [PATCH 26/34] Document stronger hidden eval checks --- evals/schema.json | 2 ++ 1 file changed, 2 insertions(+) diff --git a/evals/schema.json b/evals/schema.json index 231160d..1701dd0 100644 --- a/evals/schema.json +++ b/evals/schema.json @@ -70,6 +70,7 @@ "changed_files_include", "changed_files_subset", "command", + "command_no_changes", "file_absent", "file_contains", "file_exists", @@ -77,6 +78,7 @@ "file_unchanged", "final_contains_all", "final_contains_any", + "final_not_claim_any", "final_not_contains_any" ] }, From 8bc2c826ba6891274f3738a2e5998dfad14afcce Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:10:11 +0800 Subject: [PATCH 27/34] Strengthen evals for second-order engineering risks --- evals/cases.json | 262 ++++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 245 insertions(+), 17 deletions(-) diff --git a/evals/cases.json b/evals/cases.json index 23a8088..aaedbfb 100644 --- a/evals/cases.json +++ b/evals/cases.json @@ -153,27 +153,43 @@ ], "fixture": { "files": { - "local_test.py": "print(\"local check passed\")\n", - "service_test.py": "import os\nimport sys\n\nif not os.environ.get(\"EVAL_REQUIRED_SERVICE\"):\n print(\"required service is unavailable\", file=sys.stderr)\n raise SystemExit(3)\nprint(\"service check passed\")\n", + "local_test.py": "from pathlib import Path\n\nPath(\".eval-local-ran\").write_text(\"ran\\n\", encoding=\"utf-8\")\nprint(\"local check passed\")\n", + "service_test.py": "import os\nimport sys\nfrom pathlib import Path\n\nPath(\".eval-service-ran\").write_text(\"ran\\n\", encoding=\"utf-8\")\nif not os.environ.get(\"EVAL_REQUIRED_SERVICE\"):\n print(\"required service is unavailable\", file=sys.stderr)\n raise SystemExit(3)\nprint(\"service check passed\")\n", "README.md": "Verification commands:\n\n- `python local_test.py`\n- `python service_test.py` (requires EVAL_REQUIRED_SERVICE)\n" } }, "checks": [ { "type": "changed_files_subset", - "paths": [] + "paths": [ + ".eval-local-ran", + ".eval-service-ran" + ] + }, + { + "type": "file_exists", + "path": ".eval-local-ran" + }, + { + "type": "file_exists", + "path": ".eval-service-ran" + }, + { + "type": "final_contains_all", + "terms": [ + "local check passed" + ] }, { "type": "final_contains_any", "terms": [ - "not verified", "blocked", "unavailable", "required service" ] }, { - "type": "final_not_contains_any", + "type": "final_not_claim_any", "terms": [ "all tests passed", "fully verified", @@ -224,6 +240,14 @@ "tests", "-v" ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import tempfile\nfrom pathlib import Path\nfrom reader import read_document\n\nwith tempfile.TemporaryDirectory() as directory:\n base = Path(directory)\n root = base / \"content\"\n docs = root / \"docs\"\n docs.mkdir(parents=True)\n secret = base / \"secret.txt\"\n secret.write_text(\"secret\", encoding=\"utf-8\")\n link = docs / \"link.txt\"\n link.symlink_to(secret)\n try:\n read_document(root, \"docs/link.txt\")\n except (ValueError, PermissionError):\n pass\n else:\n raise SystemExit(\"symlink escape was allowed\")\n" + ] } ] }, @@ -311,6 +335,14 @@ "tests", "-v" ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import threading\nimport time\nfrom worker import process_all\n\nstarted = []\nlock = threading.Lock()\n\nclass Boom(RuntimeError):\n pass\n\ndef process(value):\n with lock:\n started.append(value)\n if value == 1:\n time.sleep(0.02)\n raise Boom(\"failed\")\n time.sleep(0.20)\n return value\n\ntry:\n process_all([1, 2, 3, 4, 5, 6], process)\nexcept Boom:\n pass\nelse:\n raise SystemExit(\"worker failure was not propagated\")\n\nif len(started) > 2:\n raise SystemExit(f\"pending work continued after failure: {started}\")\n" + ] } ] }, @@ -421,8 +453,8 @@ ], "fixture": { "files": { - "token.py": "def token(prefix: str, number: int) -> str:\n return f\"{prefix}-{number:04d}\"\n", - "tests/test_token.py": "import random\nimport unittest\n\nfrom token import token\n\n\nclass TokenTests(unittest.TestCase):\n def test_token(self):\n number = random.choice([7, 8])\n self.assertEqual(\"job-0007\", token(\"job\", number))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_token.py": "import random\nimport unittest\n\nfrom token_value import token\n\n\nclass TokenTests(unittest.TestCase):\n def test_token(self):\n number = random.choice([7, 8])\n self.assertEqual(\"job-0007\", token(\"job\", number))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "token_value.py": "def token(prefix: str, number: int) -> str:\n return f\"{prefix}-{number:04d}\"\n" } }, "checks": [ @@ -440,7 +472,7 @@ }, { "type": "file_unchanged", - "path": "token.py" + "path": "token_value.py" }, { "type": "command", @@ -459,32 +491,38 @@ }, { "id": "database-migration-rollout", - "task": "Old and new application versions can run at the same time during deployment. Replace the planned destructive `name`→`display_name` database rename with a migration approach that supports mixed-version rollout. Keep the SQL migration focused.", + "task": "Old and new application versions can run at the same time during deployment. Replace the planned destructive `name`→`display_name` database rename with a rollout that supports mixed-version reads and writes and remains rollback-safe. Keep the migration and compatibility code focused.", "must_do": [ "reason about old-reader and new-reader compatibility", - "plan additive rollout and backfill when practical", - "verify rollback and mixed-version behavior" + "use an additive migration and backfill existing rows", + "make new reads tolerate rows written by old code", + "make new writes preserve data needed by old code after rollback" ], "must_not_do": [ - "assume an instantaneous synchronized deployment" + "assume an instantaneous synchronized deployment", + "remove the legacy field before the compatibility window closes" ], "fixture": { "files": { "migration.sql": "ALTER TABLE users RENAME COLUMN name TO display_name;\n", - "tests/test_migration.py": "import unittest\nfrom pathlib import Path\n\n\nclass MigrationTests(unittest.TestCase):\n def test_migration_is_additive_for_mixed_versions(self):\n sql = Path(\"migration.sql\").read_text(encoding=\"utf-8\").upper()\n self.assertIn(\"ADD COLUMN\", sql)\n self.assertIn(\"DISPLAY_NAME\", sql)\n self.assertNotIn(\"RENAME COLUMN\", sql)\n self.assertNotIn(\"DROP COLUMN\", sql)\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_migration.py": "import unittest\nfrom pathlib import Path\n\n\nclass MigrationTests(unittest.TestCase):\n def test_migration_is_additive_for_mixed_versions(self):\n sql = Path(\"migration.sql\").read_text(encoding=\"utf-8\").upper()\n self.assertIn(\"ADD COLUMN\", sql)\n self.assertIn(\"DISPLAY_NAME\", sql)\n self.assertNotIn(\"RENAME COLUMN\", sql)\n self.assertNotIn(\"DROP COLUMN\", sql)\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "app.py": "def read_display_name(row: dict[str, str | None]) -> str:\n return row[\"display_name\"]\n\n\ndef write_user(display_name: str) -> dict[str, str]:\n return {\"display_name\": display_name}\n" } }, "checks": [ { "type": "changed_files_include", "paths": [ - "migration.sql" + "migration.sql", + "app.py" ] }, { "type": "changed_files_subset", "paths": [ - "migration.sql" + "migration.sql", + "app.py", + "tests/test_migration.py" ] }, { @@ -499,6 +537,14 @@ "-v" ] }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import sqlite3\nfrom pathlib import Path\nfrom app import read_display_name, write_user\n\nconnection = sqlite3.connect(\":memory:\")\nconnection.execute(\"CREATE TABLE users (id INTEGER PRIMARY KEY, name TEXT NOT NULL)\")\nconnection.execute(\"INSERT INTO users (name) VALUES ('Ada')\")\nconnection.executescript(Path(\"migration.sql\").read_text(encoding=\"utf-8\"))\nrow = connection.execute(\"SELECT name, display_name FROM users\").fetchone()\nif row != (\"Ada\", \"Ada\"):\n raise SystemExit(f\"existing row was not backfilled compatibly: {row}\")\n\nif read_display_name({\"name\": \"Old writer\", \"display_name\": None}) != \"Old writer\":\n raise SystemExit(\"new reader cannot read a row produced by old code\")\n\npayload = write_user(\"New writer\")\nif payload.get(\"name\") != \"New writer\" or payload.get(\"display_name\") != \"New writer\":\n raise SystemExit(\"new writer is not rollback-compatible with old readers\")\n" + ] + }, { "type": "final_contains_any", "terms": [ @@ -550,7 +596,7 @@ "path": "generate.py" }, { - "type": "command", + "type": "command_no_changes", "argv": [ "{python}", "generate.py" @@ -630,7 +676,9 @@ "files": { "normalize.py": "def normalize_code(value: str) -> str:\n return value.upper()\n", "naming.py": "def legacy_customer_label(first: str, last: str) -> str:\n return last + \", \" + first\n", - "tests/test_normalize.py": "import unittest\n\nfrom normalize import normalize_code\n\n\nclass NormalizeTests(unittest.TestCase):\n def test_surrounding_whitespace(self):\n self.assertEqual(\"ABC\", normalize_code(\" abc \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + "tests/test_normalize.py": "import unittest\n\nfrom normalize import normalize_code\n\n\nclass NormalizeTests(unittest.TestCase):\n def test_surrounding_whitespace(self):\n self.assertEqual(\"ABC\", normalize_code(\" abc \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n", + "README.md": "Known issue: the legacy customer-label naming convention has a separate failing regression and is intentionally out of scope for the normalize_code task.\n", + "tests/test_naming.py": "import unittest\n\nfrom naming import legacy_customer_label\n\n\nclass NamingTests(unittest.TestCase):\n def test_new_display_convention(self):\n self.assertEqual(\"Ada Lovelace\", legacy_customer_label(\"Ada\", \"Lovelace\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" } }, "checks": [ @@ -651,6 +699,17 @@ "type": "file_unchanged", "path": "naming.py" }, + { + "type": "file_unchanged", + "path": "tests/test_naming.py" + }, + { + "type": "command", + "argv": [ + "{python}", + "tests/test_normalize.py" + ] + }, { "type": "command", "argv": [ @@ -661,6 +720,175 @@ "-s", "tests", "-v" + ], + "exit_code": 1 + }, + { + "type": "final_contains_any", + "terms": [ + "out of scope", + "unrelated", + "pre-existing", + "naming" + ] + } + ] + }, + { + "id": "security-toctou", + "task": "`read_document` validates a request-controlled path before opening it, but the file can be replaced between validation and open. Harden the read against that time-of-check/time-of-use race while preserving valid nested reads and the existing `before_open` hook.", + "must_do": [ + "treat validation and resource use as one security boundary", + "prevent a validated file from being replaced with an escaping symlink before open", + "preserve valid nested reads and the test hook" + ], + "must_not_do": [ + "assume a path remains safe merely because it resolved inside the root earlier" + ], + "fixture": { + "files": { + "reader.py": "from pathlib import Path\nfrom typing import Callable\n\n\ndef read_document(\n root: Path,\n requested: str,\n before_open: Callable[[], None] | None = None,\n) -> str:\n root_resolved = root.resolve()\n candidate = (root / requested).resolve()\n if root_resolved != candidate and root_resolved not in candidate.parents:\n raise PermissionError(\"outside content root\")\n if before_open is not None:\n before_open()\n return candidate.read_text(encoding=\"utf-8\")\n", + "tests/test_reader.py": "import tempfile\nimport unittest\nfrom pathlib import Path\n\nfrom reader import read_document\n\n\nclass ReaderTests(unittest.TestCase):\n def test_nested_read(self):\n with tempfile.TemporaryDirectory() as directory:\n root = Path(directory) / \"content\"\n path = root / \"docs\" / \"a.txt\"\n path.parent.mkdir(parents=True)\n path.write_text(\"ok\", encoding=\"utf-8\")\n self.assertEqual(\"ok\", read_document(root, \"docs/a.txt\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "reader.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "reader.py", + "tests/test_reader.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "import tempfile\nfrom pathlib import Path\nfrom reader import read_document\n\nwith tempfile.TemporaryDirectory() as directory:\n base = Path(directory)\n root = base / \"content\"\n docs = root / \"docs\"\n docs.mkdir(parents=True)\n target = docs / \"a.txt\"\n target.write_text(\"safe\", encoding=\"utf-8\")\n secret = base / \"secret.txt\"\n secret.write_text(\"secret\", encoding=\"utf-8\")\n\n def swap():\n target.unlink()\n target.symlink_to(secret)\n\n try:\n read_document(root, \"docs/a.txt\", before_open=swap)\n except (ValueError, PermissionError, OSError):\n pass\n else:\n raise SystemExit(\"TOCTOU replacement escaped the content root\")\n" + ] + } + ] + }, + { + "id": "authorization-boundary", + "task": "`read_document` receives a request-controlled document ID and the caller tenant ID. Prevent a caller from reading a document owned by another tenant while preserving same-tenant reads.", + "must_do": [ + "treat object lookup and authorization as separate checks", + "authorize the resolved resource against the caller tenant", + "preserve same-tenant reads" + ], + "must_not_do": [ + "treat a valid document ID as sufficient authorization" + ], + "fixture": { + "files": { + "documents.py": "def read_document(\n documents: dict[str, dict[str, str]],\n document_id: str,\n tenant_id: str,\n) -> str:\n return documents[document_id][\"body\"]\n", + "tests/test_documents.py": "import unittest\n\nfrom documents import read_document\n\n\nclass DocumentTests(unittest.TestCase):\n def test_same_tenant_read(self):\n documents = {\n \"doc-1\": {\"tenant_id\": \"tenant-a\", \"body\": \"ok\"},\n }\n self.assertEqual(\"ok\", read_document(documents, \"doc-1\", \"tenant-a\"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "documents.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "documents.py", + "tests/test_documents.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "from documents import read_document\n\ndocuments = {\n \"shared-id\": {\"tenant_id\": \"tenant-b\", \"body\": \"secret\"},\n}\ntry:\n read_document(documents, \"shared-id\", \"tenant-a\")\nexcept (PermissionError, KeyError):\n pass\nelse:\n raise SystemExit(\"cross-tenant document read was allowed\")\n" + ] + } + ] + }, + { + "id": "api-error-compatibility", + "task": "Update `normalize_name` so valid names are trimmed before returning. Preserve the existing public failure contract for missing or blank names because callers consume the exception type, code, and message.", + "must_do": [ + "classify the error behavior as part of the public contract", + "preserve exception type, code, and machine-consumed message", + "apply trimming without weakening blank-input validation" + ], + "must_not_do": [ + "replace the public error with a generic exception or silent empty value" + ], + "fixture": { + "files": { + "names.py": "class InvalidName(ValueError):\n code = \"invalid_name\"\n\n\ndef normalize_name(value: str) -> str:\n if not value:\n raise InvalidName(\"name is required\")\n return value\n", + "tests/test_names.py": "import unittest\n\nfrom names import normalize_name\n\n\nclass NameTests(unittest.TestCase):\n def test_valid_name_is_trimmed(self):\n self.assertEqual(\"Ada\", normalize_name(\" Ada \"))\n\n\nif __name__ == \"__main__\":\n unittest.main()\n" + } + }, + "checks": [ + { + "type": "changed_files_include", + "paths": [ + "names.py" + ] + }, + { + "type": "changed_files_subset", + "paths": [ + "names.py", + "tests/test_names.py" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-m", + "unittest", + "discover", + "-s", + "tests", + "-v" + ] + }, + { + "type": "command", + "argv": [ + "{python}", + "-c", + "from names import InvalidName, normalize_name\n\nfor value in (\"\", \" \"):\n try:\n normalize_name(value)\n except InvalidName as exc:\n if getattr(exc, \"code\", None) != \"invalid_name\":\n raise SystemExit(\"error code changed\")\n if str(exc) != \"name is required\":\n raise SystemExit(\"error message changed\")\n else:\n raise SystemExit(\"blank name did not preserve InvalidName failure contract\")\n" ] } ] From 9d80768266791038f4b6201edab67c3b9e726a56 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:10:26 +0800 Subject: [PATCH 28/34] Test second-order eval check semantics --- tests/test_run_evals.py | 92 ++++++++++++++++++++++++++++++++++++++++- 1 file changed, 91 insertions(+), 1 deletion(-) diff --git a/tests/test_run_evals.py b/tests/test_run_evals.py index e3313d5..6e0e4ac 100644 --- a/tests/test_run_evals.py +++ b/tests/test_run_evals.py @@ -190,6 +190,96 @@ def test_agent_cannot_mutate_staged_skill_without_failing_case(self) -> None: self.assertEqual("failed", integrity["status"]) self.assertIn("SKILL.md", integrity["changed_files"]) + def test_final_not_claim_any_ignores_negated_claim(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is not fully verified because the service check is blocked.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("passed", result["status"]) + self.assertEqual([], result["matched"]) + + def test_final_not_claim_any_rejects_positive_claim(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is fully verified.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + self.assertEqual(["fully verified"], result["matched"]) + + def test_command_no_changes_detects_generator_drift(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + path = workspace / "generated.py" + path.write_text("value = 1\n", encoding="utf-8") + before = run_evals.snapshot_workspace(workspace) + result = run_evals.evaluate_check( + { + "type": "command_no_changes", + "argv": [ + "{python}", + "-c", + "from pathlib import Path; Path('generated.py').write_text('value = 2\\n')", + ], + }, + workspace=workspace, + before=before, + after=before, + final_output="", + allow_workspace_execution=True, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + self.assertEqual(["generated.py"], result["executions"][0]["changed_files"]) + + def test_command_no_changes_passes_idempotent_generator(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + path = workspace / "generated.py" + path.write_text("value = 1\n", encoding="utf-8") + before = run_evals.snapshot_workspace(workspace) + result = run_evals.evaluate_check( + { + "type": "command_no_changes", + "argv": [ + "{python}", + "-c", + "from pathlib import Path; p = Path('generated.py'); p.write_text(p.read_text())", + ], + }, + workspace=workspace, + before=before, + after=before, + final_output="", + allow_workspace_execution=True, + command_timeout=30, + ) + + self.assertEqual("passed", result["status"]) + self.assertEqual([], result["executions"][0]["changed_files"]) + def test_report_records_only_forwarded_environment_names(self) -> None: report = run_evals.build_report( [], @@ -259,7 +349,7 @@ def test_staged_skill_excludes_eval_rubric(self) -> None: def test_repository_cases_validate(self) -> None: cases = run_evals.load_cases(ROOT / "evals" / "cases.json") self.assertEqual([], run_evals.validate_cases(cases)) - self.assertEqual(14, len(cases)) + self.assertEqual(17, len(cases)) if __name__ == "__main__": From 114b5746f42839aa69dd4eeab19c75bf173c90b6 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:10:42 +0800 Subject: [PATCH 29/34] Document strengthened second-order eval coverage --- evals/README.md | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/evals/README.md b/evals/README.md index 25756e5..e7d0efa 100644 --- a/evals/README.md +++ b/evals/README.md @@ -21,7 +21,7 @@ A fake or deterministic adapter used by unit tests proves only that the harness ## Case structure -`cases.json` contains 14 executable scenarios. Each case defines: +`cases.json` contains 17 executable scenarios. Each case defines: - `id`: stable kebab-case identifier, - `task`: the instruction exposed to the agent, @@ -44,8 +44,9 @@ The runner currently supports: - file existence and absence, - unchanged-file assertions, - required or forbidden file content, -- final-output term assertions, -- executable commands with optional repetition. +- final-output term assertions, including negation-aware forbidden-claim checks, +- executable commands with optional repetition, +- idempotence checks that run a generator or formatter and fail if it produces a diff. Executable commands are useful for regression tests, compatibility tests, generators, and repeated flaky-test checks. They are not treated as inherently safe. @@ -201,15 +202,16 @@ The current scenarios cover: - public-contract migration, - incorrect abstraction pressure, - honest verification reporting, -- path traversal boundaries, +- path traversal, symlink, TOCTOU, and tenant-authorization boundaries, - speculative performance optimization, -- bounded concurrency and ordering, +- bounded concurrency, ordering, and failure cleanup, - behavior-preserving refactoring, - unnecessary dependency pressure, - flaky-test repair, -- mixed-version database migration, -- generated-code source-of-truth changes, -- swallowed errors, -- uncontrolled scope expansion. +- mixed-version and rollback-safe database migration, +- generated-code source-of-truth changes and generator-drift detection, +- swallowed errors and public error-contract compatibility, +- blocked verification evidence, +- uncontrolled scope expansion in the presence of a known unrelated failure. When the skill contract changes, add or strengthen a case that can distinguish the old behavior from the intended behavior. Prefer executable invariants over prose-only expectations when the behavior can be measured. From 7341a263fd0cd6900371f595fb83fd5afc9a954c Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:11:20 +0800 Subject: [PATCH 30/34] Fix negation-aware final claim matching --- scripts/run_evals.py | 35 +++++++++++++++++++++++------------ 1 file changed, 23 insertions(+), 12 deletions(-) diff --git a/scripts/run_evals.py b/scripts/run_evals.py index 70db68d..1c75b58 100644 --- a/scripts/run_evals.py +++ b/scripts/run_evals.py @@ -370,13 +370,28 @@ def _read_optional(path: Path) -> str | None: return None -NEGATION_PREFIX_RE = re.compile( - r"(?:\\bnot\\b|\\bnever\\b|\\bwithout\\b|\\bcannot\\b|\\bcan['’]?t\\b|" - r"\\bisn['’]?t\\b|\\bwasn['’]?t\\b|\\baren['’]?t\\b|\\bweren['’]?t\\b|" - r"\\bcouldn['’]?t\\b|\\bshouldn['’]?t\\b|\\bwouldn['’]?t\\b)" - r"(?:\\W+\\w+){0,2}\\W*$", - re.IGNORECASE, -) +NEGATION_TOKENS = { + "not", + "never", + "without", + "cannot", + "can't", + "isn't", + "wasn't", + "aren't", + "weren't", + "couldn't", + "shouldn't", + "wouldn't", +} + + +def _is_negated_occurrence(prefix: str) -> bool: + clause = re.split(r"[.!?;\n]", prefix)[-1].casefold() + tokens = re.findall(r"[a-z]+(?:['’][a-z]+)?", clause) + if tokens[-2:] == ["not", "only"]: + return False + return any(token in NEGATION_TOKENS for token in tokens[-3:]) def _unnegated_term_matches(text: str, terms: list[str]) -> list[str]: @@ -385,11 +400,7 @@ def _unnegated_term_matches(text: str, terms: list[str]) -> list[str]: pattern = re.compile(re.escape(term), re.IGNORECASE) for match in pattern.finditer(text): prefix = text[max(0, match.start() - 80):match.start()] - prefix = re.split(r"[.!?;\\n]", prefix)[-1] - if re.search(r"\\bnot\\s+only\\W*$", prefix, re.IGNORECASE): - matches.append(term) - break - if NEGATION_PREFIX_RE.search(prefix): + if _is_negated_occurrence(prefix): continue matches.append(term) break From a155ff101feb710b5e78669d8cb4a401cd12311f Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:11:49 +0800 Subject: [PATCH 31/34] Validate hidden inline Python checks without executing them --- scripts/run_evals.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/scripts/run_evals.py b/scripts/run_evals.py index 1c75b58..eecd3ed 100644 --- a/scripts/run_evals.py +++ b/scripts/run_evals.py @@ -161,6 +161,19 @@ def validate_cases(cases: list[dict[str, Any]]) -> list[str]: repeat = check.get("repeat", 1) if not isinstance(repeat, int) or repeat < 1 or repeat > 20: errors.append(f"{check_label}: repeat must be between 1 and 20") + if ( + isinstance(argv, list) + and len(argv) >= 3 + and argv[0] == "{python}" + and argv[1] == "-c" + and isinstance(argv[2], str) + ): + try: + compile(argv[2], f"<{check_label}>", "exec") + except SyntaxError as exc: + errors.append( + f"{check_label}: invalid inline Python: {exc.msg}" + ) elif check_type in {"changed_files_include", "changed_files_subset"}: paths = check.get("paths") if not isinstance(paths, list) or not all( From ce54e910340ae1c31b0208022f407185c96b6787 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:12:00 +0800 Subject: [PATCH 32/34] Test hidden-check validation and negation edge case --- tests/test_run_evals.py | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/tests/test_run_evals.py b/tests/test_run_evals.py index 6e0e4ac..f92ca91 100644 --- a/tests/test_run_evals.py +++ b/tests/test_run_evals.py @@ -81,6 +81,19 @@ def test_check_environment_uses_isolated_home(self) -> None: self.assertEqual(str(home), env["HOME"]) self.assertEqual(str(home), env["USERPROFILE"]) + def test_validate_cases_rejects_invalid_inline_python_check(self) -> None: + case = self.sample_case() + case["checks"] = [ + { + "type": "command", + "argv": ["{python}", "-c", "if True print('bad')"], + } + ] + + errors = run_evals.validate_cases([case]) + + self.assertTrue(any("invalid inline Python" in error for error in errors)) + def test_validate_cases_rejects_unsafe_fixture_path(self) -> None: case = self.sample_case() case["fixture"]["files"] = {"../escape.py": "bad\n"} @@ -209,6 +222,24 @@ def test_final_not_claim_any_ignores_negated_claim(self) -> None: self.assertEqual("passed", result["status"]) self.assertEqual([], result["matched"]) + def test_final_not_claim_any_does_not_treat_not_only_as_negation(self) -> None: + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + result = run_evals.evaluate_check( + { + "type": "final_not_claim_any", + "terms": ["fully verified"], + }, + workspace=workspace, + before={}, + after={}, + final_output="The repository is not only fully verified but release-ready.", + allow_workspace_execution=False, + command_timeout=30, + ) + + self.assertEqual("failed", result["status"]) + def test_final_not_claim_any_rejects_positive_claim(self) -> None: with tempfile.TemporaryDirectory() as directory: workspace = Path(directory) From 2dc81d1b524947b2e9e56c8f72fb060e4f523f16 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:12:32 +0800 Subject: [PATCH 33/34] Regression-test flaky fixture module isolation --- tests/test_run_evals.py | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tests/test_run_evals.py b/tests/test_run_evals.py index f92ca91..9b1ee72 100644 --- a/tests/test_run_evals.py +++ b/tests/test_run_evals.py @@ -2,6 +2,7 @@ import os import shlex +import subprocess import sys import tempfile import unittest @@ -377,6 +378,32 @@ def test_staged_skill_excludes_eval_rubric(self) -> None: self.assertFalse((Path(directory) / "evals").exists()) self.assertFalse((Path(directory) / "tests").exists()) + def test_flaky_fixture_does_not_shadow_stdlib_token_module(self) -> None: + cases = run_evals.load_cases(ROOT / "evals" / "cases.json") + case = next(case for case in cases if case["id"] == "flaky-test") + + with tempfile.TemporaryDirectory() as directory: + workspace = Path(directory) + run_evals.materialize_fixture(case, workspace) + self.assertFalse((workspace / "token.py").exists()) + completed = subprocess.run( + [ + sys.executable, + "-c", + ( + "from token_value import token; " + "assert token('job', 7) == 'job-0007'" + ), + ], + cwd=workspace, + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + + self.assertEqual(0, completed.returncode, completed.stderr) + def test_repository_cases_validate(self) -> None: cases = run_evals.load_cases(ROOT / "evals" / "cases.json") self.assertEqual([], run_evals.validate_cases(cases)) From ab882e3728683b045356923cf45e2864d08b63a3 Mon Sep 17 00:00:00 2001 From: GeoGeek Date: Wed, 23 Sep 2026 15:17:40 +0800 Subject: [PATCH 34/34] Sharpen README for GitHub engineering style --- README.md | 479 +++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 330 insertions(+), 149 deletions(-) diff --git a/README.md b/README.md index d102219..c049f59 100644 --- a/README.md +++ b/README.md @@ -2,18 +2,16 @@ # engineering-quality -**Evidence-first engineering quality for coding agents.** +**Make coding agents prove their patches.** -A portable Agent Skill for **Codex, Claude Code, ChatGPT, and other coding agents** that makes implementation, code review, testing, debugging, refactoring, compatibility, and security work more disciplined and verifiable. - -`read the repo → model the contract → patch narrowly → try to break it → inspect the diff → show the evidence` +`RECON → CONTRACT → CHANGE → VERIFY → REVIEW → EVIDENCE` [![CI](https://github.com/GeoGeekLab/engineering-quality/actions/workflows/ci.yml/badge.svg)](https://github.com/GeoGeekLab/engineering-quality/actions/workflows/ci.yml) [![Release](https://img.shields.io/github/v/release/GeoGeekLab/engineering-quality?style=flat-square)](https://github.com/GeoGeekLab/engineering-quality/releases) [![License: MIT](https://img.shields.io/badge/license-MIT-2ea44f?style=flat-square)](LICENSE) [![Python](https://img.shields.io/badge/Python-3.10%20%7C%203.12%20%7C%203.14-3776AB?style=flat-square&logo=python&logoColor=white)](.github/workflows/ci.yml) -**Less taste. More invariants.** +**Less vibes. More invariants.**
@@ -21,33 +19,44 @@ A portable Agent Skill for **Codex, Claude Code, ChatGPT, and other coding agent -## Why this exists - -Coding agents are very good at producing plausible patches. Plausible is not the same as correct. +## What it is -A patch can compile and still violate a public contract. Tests can pass while missing the failure mode. A refactor can be elegant while quietly widening scope. A verification command can itself execute untrusted repository code. +`engineering-quality` is a portable Agent Skill for implementation, review, debugging, refactoring, compatibility, security, and verification work. -engineering-quality gives coding agents a compact engineering protocol: +It gives coding agents a compact operating protocol: ```text -RECON → CONTRACT → CHANGE → VERIFY → REVIEW → EVIDENCE +read the repo + ↓ +model the contract + ↓ +patch the smallest coherent surface + ↓ +try to break it + ↓ +inspect the diff + ↓ +show the evidence ``` -The goal is not more process. The goal is a patch you can defend. +The point is simple: -## Quick start +```text +plausible patch ≠ correct patch +green test ≠ complete evidence +clean diff ≠ safe rollout +works ≠ verified +``` -Host instructions were last checked against official vendor documentation on **2026-09-20**. See [host compatibility](docs/compatibility.md) for exact evidence levels. +## Install ### Codex -For local setup or experimentation: - ```text $skill-installer Install engineering-quality from https://github.com/GeoGeekLab/engineering-quality ``` -Manual user install: +Manual install: ```bash git clone https://github.com/GeoGeekLab/engineering-quality.git \ @@ -63,7 +72,7 @@ git clone https://github.com/GeoGeekLab/engineering-quality.git \ ~/.claude/skills/engineering-quality ``` -Or load the repository as a native single-skill plugin during development: +Plugin development: ```bash claude --plugin-dir /path/to/engineering-quality @@ -71,198 +80,339 @@ claude --plugin-dir /path/to/engineering-quality ### ChatGPT -On eligible ChatGPT Business, Enterprise, Healthcare, and Edu workspaces: - -1. open the [latest GitHub Release](https://github.com/GeoGeekLab/engineering-quality/releases/latest), -2. download `engineering-quality-.zip`, -3. in ChatGPT open **Plugins → Skills → Create → Upload from your computer**, -4. upload the ZIP. - -### Optional cross-agent installer +1. Open the [latest release](https://github.com/GeoGeekLab/engineering-quality/releases/latest). +2. Download `engineering-quality-.zip`. +3. Open **Plugins → Skills → Create → Upload from your computer**. +4. Upload the ZIP. -The third-party Vercel Labs CLI is available as a convenience: +### Cross-agent installer ```bash npx -y skills@1.7.0 add GeoGeekLab/engineering-quality ``` -It is not treated as the vendor-native distribution authority for OpenAI or Anthropic. +Host setup notes: [docs/compatibility.md](docs/compatibility.md) -## What changes when an agent uses it +## Protocol -Without an explicit engineering contract, a coding-agent answer can look like: +### 1. RECON + +Read before editing. + +- repository instructions +- architecture and ownership +- nearby tests +- public behavior +- data flow +- existing patterns +- verification commands +- compatibility and security surfaces + +### 2. CONTRACT + +Write down the invariant. + +```text +must change +must stay stable +must reject +must remain compatible +must be verified +``` -> Fixed the parser, cleaned up nearby helpers, and the tests look good. +### 3. CHANGE -engineering-quality pushes toward something more auditable: +Keep the patch tight. -> Contract: comma-delimited input remains valid; semicolon-delimited input must fail. -> Scope: changed the parser and its regression test only; unrelated utility code was left untouched. -> Verified: the focused regression test and repository checks passed. -> Not verified: no production traffic replay was available. +- no drive-by cleanup +- no abstraction for cosmetic deduplication +- no dependency without a reason +- no behavior change hidden inside a refactor +- no hand-editing generated output when a source generator exists -That difference is the project. +### 4. VERIFY -It does not prescribe one architecture, testing framework, or coding style. It asks the agent to identify the repository's actual contract, make the smallest coherent change, attack that change with relevant checks, inspect the resulting diff, and state exactly what the evidence proves. +Start narrow, then widen. -## Core guarantees +```text +regression +→ focused tests +→ static checks +→ type / compile +→ broader suite +→ build / package +→ risk-specific probes +``` -The Skill is built around a few priorities: +### 5. REVIEW -- **Correctness before aesthetics.** A beautiful implementation of the wrong behavior is still wrong. -- **Repository-local contract before generic taste.** Existing architecture, compatibility, error semantics, and tooling are evidence. -- **Small coherent changes.** Required tests, migrations, and docs belong in the patch; drive-by cleanup does not. -- **Security is part of correctness.** Trust boundaries, credentials, command execution, and unsafe defaults are first-class review concerns. -- **Compatibility is observable behavior.** Public APIs, schemas, CLI behavior, persistence, and deployment overlap are contracts until proven otherwise. -- **Evidence before confidence.** `Verified`, `Reasoned`, and `Not verified` mean different things. +Read the diff like an attacker and a maintainer. -## Evidence, not badges +Check: -This repository tries to make its own quality claims inspectable. +- correctness +- edge cases +- failure behavior +- compatibility +- authorization +- path and trust boundaries +- concurrency +- cleanup +- migrations +- generated code +- scope creep -### Repository and package evidence +### 6. EVIDENCE -`make check` validates: +Report what actually happened. ```text -repository integrity - ├── Skill + host metadata - ├── local links + governance contract - ├── immutable GitHub Action pins - ├── VERSION + CHANGELOG consistency - ├── executable behavioral-eval fixtures - ├── Python compilation + unit tests - ├── deterministic package verification - └── release-readiness invariants +Verified = executed and observed +Reasoned = supported by inspection +Not verified = still open ``` -CI runs the contract on Python **3.10**, **3.12**, and **3.14**. +## Where it bites -The package job performs a real artifact round trip: +The Skill is intentionally strict around failure modes that frequently survive a normal patch review: ```text -build → SHA-256 → upload → delete local copy → download → SHA-256 +path traversal → symlink escape → TOCTOU +object lookup → authorization → cross-tenant access +schema rename → mixed versions → rollback +parallelism → ordering → cancellation / cleanup +public API change → error type / code / message compatibility +generated code → source of truth → generator drift +test failure → retry luck → real flake cause +verification → command exists → command actually ran +focused task → unrelated red test → scope expansion ``` -### Release integrity +## Behavioral evals -Tagged releases publish: +The repository ships **17 executable miniature-repository scenarios**. -```text -engineering-quality-.zip -engineering-quality-.zip.sha256 +Current coverage includes: + +- minimal bug fixes +- public contract evolution +- wrong abstractions +- verification evidence +- path traversal and symlink escape +- TOCTOU file replacement +- tenant authorization +- speculative performance work +- bounded concurrency and cancellation +- behavior-preserving refactors +- dependency pressure +- flaky tests +- mixed-version and rollback-safe migrations +- generated-code drift +- swallowed errors +- public error compatibility +- scope expansion under unrelated failures + +Validate the suite: + +```bash +python scripts/run_evals.py --validate-only ``` -The ZIP also contains an internal `MANIFEST.sha256`. +Run one case: -Verify a downloaded archive: +```bash +python scripts/run_evals.py \ + --case security-toctou \ + --agent-command 'my-agent --prompt {task}' \ + --adapter-label my-agent \ + --allow-workspace-execution +``` + +Run the full suite: ```bash -sha256sum -c engineering-quality-.zip.sha256 +python scripts/run_evals.py \ + --agent-command 'my-agent --prompt {task}' \ + --adapter-label my-agent \ + --allow-workspace-execution \ + --output eval-results/my-agent.json +``` -gh attestation verify engineering-quality-.zip \ - --repo GeoGeekLab/engineering-quality +### Skill vs no-Skill + +Native Codex and Claude Code adapters support a clean A/B switch: + +```text +--skill-mode disabled +--skill-mode enabled ``` -Release artifacts are attested in GitHub Actions before publication. See [release engineering](docs/release.md). +Repeat runs with fresh workspaces: -### Behavioral evaluations +```bash +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode disabled' \ + --adapter-label codex-MODEL-no-skill \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/no-skill.json -The repository contains **14 executable miniature-repository scenarios** covering defect fixes, compatibility changes, security boundaries, concurrency, dependency pressure, flaky tests, migrations, generated code, swallowed errors, and scope expansion. +python scripts/run_evals.py \ + --agent-command '{python} {repo}/scripts/host_eval_adapter.py codex --model MODEL --skill-mode enabled' \ + --adapter-label codex-MODEL-skill \ + --repeat 5 \ + --allow-workspace-execution \ + --output eval-results/skill.json +``` -Validate the harness: +Compare: ```bash -python scripts/run_evals.py --validate-only +python scripts/compare_eval_results.py \ + eval-results/no-skill.json \ + eval-results/skill.json \ + --output eval-results/comparison.md ``` -Run a real host through the adapter layer when credentials and the native CLI are available. See [behavioral evaluations](evals/README.md). +The comparison reports: -The evidence boundary is deliberate: CI proves that the harness and deterministic checks work. It does **not** claim that a real Codex, Claude Code, ChatGPT, or other model passed the suite unless an actual host run produced that evidence. +```text +case pass rate +check-type pass rate +changed-file scope +agent wall-clock time +``` -## How it works +Qualitative rubric items stay visible for review instead of being flattened into one score. -| Stage | Agent behavior | -| --- | --- | -| **RECON** | Read architecture, conventions, ownership, checks, and sharp edges before editing. | -| **CONTRACT** | State what must change, what must remain stable, and what would prove success. | -| **CHANGE** | Modify the smallest coherent surface that satisfies the contract. | -| **VERIFY** | Try to falsify the patch with focused tests, static checks, builds, and risk-specific probes. | -| **REVIEW** | Inspect the complete diff for correctness, compatibility, security, concurrency, maintainability, and accidental scope. | -| **EVIDENCE** | Separate executed proof from inspection-based reasoning and unverified claims. | +More: [evals/README.md](evals/README.md) -`SKILL.md` is intentionally a bootloader rather than an encyclopedia. It loads the invariant set first, then routes into focused workflows and references only when needed. +## Hidden checks -## Tools included +The eval harness keeps second-order probes outside the fixture shown to the agent. -### Discover repository checks safely +Examples: -```bash -python scripts/project_checks.py /path/to/repository +```text +security-boundary + visible: ../ traversal + hidden: symlink escape + +security-toctou + visible: normal nested read + hidden: replace validated file before open + +concurrent-worker + visible: bounded concurrency + order + hidden: propagate failure + stop pending work + +database-migration-rollout + visible: additive schema migration + hidden: backfill + old writer + rollback-safe write + +generated-code-change + visible: generator + generated file + hidden: rerun generator and require zero diff ``` -Discovery does not execute the commands it finds. +Two harness checks exist specifically for these cases: -A target named `test`, `lint`, or `check` is not automatically safe: Make recipes, package scripts, test discovery, wrappers, compiler hooks, and plugins can execute arbitrary repository-controlled code. +- `final_not_claim_any` — rejects positive verification claims while allowing negated wording such as `not fully verified`. +- `command_no_changes` — executes a command and fails if it changes the workspace. -Execution therefore requires an explicit trust decision: +Inline hidden Python checks are compiled during `--validate-only`. + +## Repository checks + +Discover likely project checks: + +```bash +python scripts/project_checks.py /path/to/repository +``` + +Run discovered checks after reviewing them: ```bash python scripts/project_checks.py /path/to/repository \ --run --trust-repository ``` -### Run behavioral evals - -Generic adapter: +Run this repository's full quality contract: ```bash -python scripts/run_evals.py \ - --agent-command 'my-agent --prompt {task}' \ - --adapter-label my-agent \ - --allow-workspace-execution \ - --output eval-results/my-agent.json +make check +``` + +That covers: + +```text +repository integrity +├── Skill + host metadata +├── local links +├── governance contract +├── immutable GitHub Action pins +├── VERSION + CHANGELOG consistency +├── behavioral eval schema +├── behavioral eval fixtures +├── Python compilation +├── unit tests +├── package verification +└── release invariants ``` -Native Codex and Claude Code adapter examples are documented in [evals/README.md](evals/README.md). +CI runs on Python **3.10**, **3.12**, and **3.14**. -The runner uses fresh miniature Git repositories, hides the evaluation rubric from the agent, checks that the staged Skill was not mutated, and only forwards environment variables explicitly selected by the evaluator. +## Release integrity -## Where it helps most +Tagged releases publish: -engineering-quality is designed for tasks where a plausible edit is not enough: +```text +engineering-quality-.zip +engineering-quality-.zip.sha256 +``` + +The ZIP includes `MANIFEST.sha256`. -- feature implementation with an existing contract, -- bug fixes that need regression evidence, -- behavior-preserving refactors, -- code review focused on concrete failure modes, -- debugging under incomplete evidence, -- API/schema/migration compatibility, -- security-sensitive boundaries, -- concurrency and performance changes, -- generated-code workflows, -- coding-agent evaluation and verification. +Verify: -It is not a linter, formatter, static analyzer, or replacement for the repository's own tests. It composes with those tools. +```bash +sha256sum -c engineering-quality-.zip.sha256 + +gh attestation verify engineering-quality-.zip \ + --repo GeoGeekLab/engineering-quality +``` + +Package CI does a full round trip: + +```text +build +→ SHA-256 +→ upload +→ delete local copy +→ download +→ SHA-256 +``` + +More: [docs/release.md](docs/release.md) ## Review protocol -Findings are classified by impact rather than taste: +Review findings are ranked by impact. | Severity | Meaning | | --- | --- | | **Blocker** | Incorrect behavior, data loss, security exposure, broken contract, or reliably failing verification. | -| **Major** | Material failure under realistic conditions or significant maintenance/operational risk. | +| **Major** | Material failure under realistic conditions or significant operational / maintenance risk. | | **Minor** | Bounded clarity, resilience, test-quality, or consistency issue. | | **Note** | Optional improvement, question, or follow-up. | -A useful review finding should answer: +A useful finding answers four questions: ```text what can fail? under what condition? -why does this diff make that possible? -what evidence would close the finding? +why does this diff allow it? +what evidence closes the finding? ``` ## Quality model @@ -270,7 +420,11 @@ what evidence would close the finding? Not this: ```text -quality = more abstraction + more comments + more tests + more patterns +quality = + more abstraction + + more comments + + more tests + + more patterns ``` Closer to this: @@ -286,42 +440,69 @@ quality = - unnecessary churn ``` -This is an ordering of concerns, not a numeric score. +No magic number. No style points. -## Repository structure +## Repository layout ```text -SKILL.md portable behavioral contract +SKILL.md core protocol agents/openai.yaml OpenAI Skill metadata .claude-plugin/plugin.json Claude Code plugin metadata -workflows/ feature / bug-fix / refactor / review / debug / performance -references/ verification / testing / security / compatibility / concurrency -scripts/ validation / packaging / project checks / eval adapters -evals/ executable behavioral fixtures + result schema -docs/ compatibility / release / governance / distribution / mascot -assets/mascot/ Evi mascot + identity-system artwork -.github/ CI / release / issue forms / CODEOWNERS -``` -The release ZIP intentionally contains the runtime Skill payload rather than repository-maintenance files. - -## Project status and trust boundaries +workflows/ +├── feature.md +├── bug-fix.md +├── refactor.md +├── review.md +├── debug.md +└── performance.md + +references/ +├── principles.md +├── testing.md +├── verification.md +├── security.md +├── api-compatibility.md +├── performance-concurrency.md +└── change-discipline.md + +scripts/ +├── project_checks.py +├── run_evals.py +├── host_eval_adapter.py +├── compare_eval_results.py +├── package_skill.py +└── validate_skill.py + +evals/ +├── cases.json +├── schema.json +└── result-schema.json + +docs/ compatibility / release / distribution +assets/mascot/ Evi +.github/ CI / release / repo policy +``` -- Host installation guidance was last verified on **2026-09-20**. -- GitHub Actions dependencies are pinned to immutable commit SHAs and maintained by Dependabot. -- `main` and `v*.*.*` are protected by active GitHub rulesets. -- Releases use SHA-256 checksums and GitHub artifact attestations. -- Repository-defined checks execute only after explicit repository trust. -- Behavioral-eval reports distinguish deterministic evidence from qualitative rubric review. -- OpenAI Plugin Directory and Claude marketplace publication are **not** claimed until those external review/publication steps actually happen. +## Project status -See [compatibility](docs/compatibility.md), [release engineering](docs/release.md), [governance](GOVERNANCE.md), [security](SECURITY.md), and [distribution](docs/distribution.md). +- GitHub Actions are pinned to immutable commit SHAs. +- `main` and `v*.*.*` are protected by repository rulesets. +- Releases ship SHA-256 checksums and GitHub artifact attestations. +- Behavioral evals use fresh temporary Git repositories. +- The staged Skill payload is hashed before and after each run. +- Eval credentials are forwarded only when explicitly selected. +- Distribution ZIPs contain runtime Skill files, not repository-maintenance tooling. ## Contributing -Contributions should address a concrete failure mode or maintenance benefit, not merely introduce a preferred style. +Bring a failure mode, an invariant, or a measurable maintenance win. + +Start with [CONTRIBUTING.md](CONTRIBUTING.md). -Start with [CONTRIBUTING.md](CONTRIBUTING.md). Participation is covered by [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md), and vulnerabilities follow [SECURITY.md](SECURITY.md). +Security issues: [SECURITY.md](SECURITY.md) +Governance: [GOVERNANCE.md](GOVERNANCE.md) +Code of conduct: [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md) ## License