From 76cb5a38b14a4b150c226b85fc84106b3b4acfee Mon Sep 17 00:00:00 2001 From: Shane Wall Date: Mon, 21 Sep 2026 14:21:07 +1000 Subject: [PATCH] Refuse stale lab quality verdicts (#499) --- CHANGELOG.md | 7 ++ tools/argus_mcp/README.md | 11 ++ tools/argus_mcp/build.rs | 15 +++ tools/argus_mcp/src/build_identity.rs | 124 ++++++++++++++++++++++ tools/argus_mcp/src/lab.rs | 2 + tools/argus_mcp/src/lib.rs | 2 + tools/argus_mcp/src/main.rs | 2 + tools/argus_mcp/src/server.rs | 49 ++++++++- tools/argus_mcp/src/source_fingerprint.rs | 53 +++++++++ 9 files changed, 261 insertions(+), 4 deletions(-) create mode 100644 tools/argus_mcp/build.rs create mode 100644 tools/argus_mcp/src/build_identity.rs create mode 100644 tools/argus_mcp/src/source_fingerprint.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index b5734c4..dde864f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,13 @@ is in `tools/argus_mcp/README.md`. ## Unreleased +**STALE LABS CANNOT ISSUE QUALITY-GATE VERDICTS** (#499). Structured MCP +responses now identify the running package and build-time source fingerprint +beside the checkout identity. Read-only inspection remains usable under skew, +but compare and experiment surfaces refuse until the documented install command +is run and the MCP client restarted. This catches a long-lived old process even +when no newer staged binary exists. + **CONTROL AND CANDIDATE BOTS CAN SHARE A COUNTERBALANCED MATCH** (#374). `argus-mcp within-ab` runs two tapes with four bots, two per arm, then swaps the slot mask so personality cannot masquerade as treatment effect. Reports diff --git a/tools/argus_mcp/README.md b/tools/argus_mcp/README.md index c8d2de9..13a4f8b 100644 --- a/tools/argus_mcp/README.md +++ b/tools/argus_mcp/README.md @@ -19,6 +19,16 @@ Point the client at this installed binary so startup is not a cargo build and `cargo clean` cannot delete the configured server. Rerun the install command to update it, then restart the client. +Every structured MCP response carries `lab_identity`: the running package version and a +build-time fingerprint of `Cargo.toml`, `Cargo.lock`, `build.rs`, and the Rust +source, beside the same identity recomputed from the checkout. Inspection stays +available when those identities differ, but the lab refuses quality-gate +verdicts from `compare_runs`, `suggest_next` compare mode, `experiment`, +`campaign_experiment`, and `matrix_experiment`. Repair an identity mismatch with +the install command above, then restart the MCP client. This checkout comparison +catches an old installed process even when no newer `target-stage` build exists; +the older `lab_stale` stage banner remains a separate early-warning mechanism. + For a config saved before the stable install was documented, run this after the install: @@ -1327,6 +1337,7 @@ shell. | 0.22 | The lab joins the game: a real NetQuake client (`argus-mcp client observe/walk/walkrel`), the empirical link-verification harness (`argus-mcp probelinks`), engine-verdict files consumed by navgen, orphan-engine kill on failed matches, serialized engine tests. | | 0.21 | The operational gaps. STALENESS SELF-AWARENESS: at startup the server detects a newer staged build, auto-swaps it into place for the next restart (Windows allows renaming a running exe), and stamps `lab_stale` on every JSON response for the rest of the session - a stale server can never again hand out an unmarked opinion. HARVEST GUARD: every match starter (MCP tools, soak, cycle) refuses to launch over an un-harvested play session (the harvester now MOVES its inputs, so leftovers are the signal). PAIRED DEMO JOIN: brief_run folds the same-stem .dem into the brief (aim stats, highlight reel, tracks) - the whole "played, review" ritual is one call. `ship` (compile + install everywhere + MD5s) and `baseline_set` (rewrite runs/baselines.json safely) close the loop's last manual steps. `soak --parallel 2` runs two engines on separate ports, halving ladder wall clock. Last-seen session memory persists to runs/.lab_session.json across restarts; `see what=project` stops listing fifty pak-only maps; compare flags any human tape as review-only material. | | 0.23 | The issue-tracker sweep (GitHub #6-#9). Brief totals gain `grabs` and `acquisitions` (weapon switches + battle-grabs) so contested maps stop reading as consumption defects when `gl` (current-goal touches only, v3.17 boundary) looks starved; the no-pickups next_step keys on acquisitions now. The auto-swap resolves its staged twin via `ARGUS_ROOT` when the running image is a client copy outside `target/release` (the ~/.grok/bin binary can now swap itself; both client configs already set the env). Cartograph implication strings refreshed from the current graphs - no baked era counts (dm2's "31 lava-side waypoints" had outlived the lava slice by ten versions); a regression test keeps them honest. `ARGUS watch spawn` counts as pseudo-event `watch` (the v3.91 post-kill spawn watch). The dm4 `see what=map` timeout (#9) did not reproduce on 0.22+: warm and cold (mtime-invalidated) atlas rebuilds both return in under a second - the observed hang is attributed to the stale pre-swap client binary that the `ARGUS_ROOT` fix retires. Later under the same stamp: pseudo-event `sprintjump` (the v3.93 launch marker), the `client impulse ` CLI verb (roster control and dev teleport from the puppet's seat - the headless 4-player match that closed #2 and found the RosterName misalignment), and probelinks' `teleport_failures` count. | +| 0.30 | A stale lab can no longer issue a quality-gate verdict (GitHub #499). The build embeds a deterministic fingerprint of its manifest, lockfile, build script, and Rust source; every response reports that identity beside the live checkout identity. Read-only inspection remains available under skew, while compare and experiment surfaces refuse with the exact install-and-restart remedy. This closes the gap left by 0.21's stage-only banner: a long-lived old process is detected even when no staged binary exists. | | 0.29 | Response shaping, measured (GitHub #371). `brief_run` and `compare_runs` take `format=csv`: measured on a real dm4 tape a brief is 1765 bytes as CSV against 3464 as compact JSON, and more again against the pretty JSON the server actually sends. JSON stays the default. **The live tail paginates on SIZE, not line count**: telemetry lines vary by five times, so the old flat eighty-line cut returned between 1.5 and 8 KB depending on what the match happened to be doing, and a busy match returned the most. 6 KB budget, 120 line ceiling, `since_line` still the cursor, and always at least one line so a poll loop cannot stall. **THE TOOL SURFACE NOW HAS A PRICE ON IT**: 39 tools, 18,816 bytes, about 4,700 tokens on every request, with a bound in the suite so adding a tool is a decision rather than a drift. That measurement REFUSED the third part of the issue: retiring the three parked extras would save 1,206 bytes, 6 per cent, and break a documented spec capture, while 581 of those bytes were available by trimming the schema of `corpus` itself - which was the most expensive tool in the server, written one version earlier to save tokens. | | 0.28 | The corpus becomes something you can ask a question of (GitHub #372, #378, #386). `corpus` is one tool with a `what` selector, like `see`: `tapes` filters and aggregates an index of every committed tape, `cells` does the same for hotspot cells, `changes` runs change point detection over the dated series, `bisect` localises one step with a noisy oracle. It returns CSV, which is about half the tokens of the same table as JSON. The index is `runs/tape_index.tsv`, a cache rather than a source of truth: a tape it does not name is parsed on demand, and nothing re-parses 752 tapes on an ordinary call. **SQLITE WAS CONSIDERED AND REFUSED**: the corpus is a few hundred rows of a fixed schema, and a named filter-and-aggregate surface costs an agent less than discovering a schema and writing SQL against it. The detector reproduces the record: dm4 lava stepping 12.3 to 4.4 at `ab_dm4_B3` is the 2026-08-14 hazard fix, e1m1 engages 1.1 to 17.8 at `ab_e1m1_mover1` is #281, and it dates the September dm2 stall rise that `CLAUDE.md` calls "not bisected" to 2026-08-29 at `ab_dm2_v403_control` - while flagging on the row that the tick class differs either side, so it must not be read as code without a single-class re-run. **TWO PARSER DEFECTS THE INDEX FOUND**: the lab puppet emits ARGLOG rows and never spawns, so the human split read it as a person and every tape it connected to briefed as a human session, two committed baselines included; and human ARGLOG tracks only exist from v3.66, so 46 of 73 harvested sessions were sitting in the bot series. Both fixed, and no tape's numbers move - verified by diffing the whole index across the change. `docs/specs/2026-09-16-corpus-change-points.md`. 182 tests. | | 0.27 | The verdict gets an interval, a stopping rule and a pre-registered primary (GitHub #375, #376, #377). `argus-mcp measure` writes `docs/specs/2026-09-16-lab-measurement-limits.md`: the pooled within-arm sigma per map and metric over every committed same-build arm, and the smallest effect detectable at 3, 5 and 10 tapes a side. It says in print what the handoff said in prose - dm2 stalls at three tapes a side cannot see anything smaller than a 119 per cent change, and coverage at 9 per cent CV is the only metric worth reading off a small ladder. It also measures what four gates cost: each convicts a byte-identical pair 0 to 12 per cent of the time and the verdict as a whole convicts 19 per cent of the time, so `experiment` and `compare` take `primary`, the metric the change was predicted to move, and only that gate convicts. Every compare now carries a `stats` block: interquartile mean rather than median, a stratified bootstrap interval on the improvement (absent below two tapes a side, where resampling one observation is not an interval), a probability of improvement, and an SPRT call of accept, reject, continue or abandon with its bound derived from the detection limit. **Continue is the answer the lab could never give.** Validated on the null corpus: zero accepts across 58 decisions from ten same-build arms split in half, and the sixteen null pairs still read thirteen parities. 163 tests. | diff --git a/tools/argus_mcp/build.rs b/tools/argus_mcp/build.rs new file mode 100644 index 0000000..4e26e14 --- /dev/null +++ b/tools/argus_mcp/build.rs @@ -0,0 +1,15 @@ +#[path = "src/source_fingerprint.rs"] +mod source_fingerprint; + +fn main() { + let manifest = std::path::PathBuf::from( + std::env::var_os("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR"), + ); + let fingerprint = source_fingerprint::fingerprint(&manifest) + .expect("fingerprint Argus MCP sources at build time"); + println!("cargo:rustc-env=ARGUS_BUILD_SOURCE_FINGERPRINT={fingerprint}"); + println!("cargo:rerun-if-changed=build.rs"); + println!("cargo:rerun-if-changed=Cargo.toml"); + println!("cargo:rerun-if-changed=Cargo.lock"); + println!("cargo:rerun-if-changed=src"); +} diff --git a/tools/argus_mcp/src/build_identity.rs b/tools/argus_mcp/src/build_identity.rs new file mode 100644 index 0000000..9e45556 --- /dev/null +++ b/tools/argus_mcp/src/build_identity.rs @@ -0,0 +1,124 @@ +//! Identify the running lab build and refuse verdicts under source skew. + +use std::path::Path; + +use serde::Serialize; + +use crate::config::Config; + +pub const RUNNING_VERSION: &str = env!("CARGO_PKG_VERSION"); +pub const BUILD_SOURCE: &str = env!("ARGUS_BUILD_SOURCE_FINGERPRINT"); + +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +pub struct LabIdentity { + pub running_version: String, + pub build_source: String, + pub checkout_version: Option, + pub checkout_source: Option, + pub authoritative: bool, + #[serde(skip_serializing_if = "Option::is_none")] + pub warning: Option, +} + +fn checkout_version(manifest_dir: &Path) -> Result { + let path = manifest_dir.join("Cargo.toml"); + let text = + std::fs::read_to_string(&path).map_err(|error| format!("{}: {error}", path.display()))?; + let value: toml::Value = + toml::from_str(&text).map_err(|error| format!("{}: {error}", path.display()))?; + value + .get("package") + .and_then(|package| package.get("version")) + .and_then(toml::Value::as_str) + .map(str::to_string) + .ok_or_else(|| format!("{} has no package.version", path.display())) +} + +fn inspect_dir(manifest_dir: &Path, running_version: &str, build_source: &str) -> LabIdentity { + let version = checkout_version(manifest_dir).ok(); + let source = crate::source_fingerprint::fingerprint(manifest_dir).ok(); + let authoritative = + version.as_deref() == Some(running_version) && source.as_deref() == Some(build_source); + let warning = (!authoritative).then(|| { + format!( + "STALE LAB: running argus-mcp {running_version} ({build_source}) does not match checkout {} ({}). Quality-gate verdicts are refused. Run `cargo install --locked --path tools/argus_mcp --root tools/argus_mcp/install`, then restart the MCP client.", + version.as_deref().unwrap_or("unreadable"), + source.as_deref().unwrap_or("unreadable") + ) + }); + LabIdentity { + running_version: running_version.into(), + build_source: build_source.into(), + checkout_version: version, + checkout_source: source, + authoritative, + warning, + } +} + +pub fn inspect(cfg: &Config) -> LabIdentity { + inspect_dir( + &cfg.root.join("tools").join("argus_mcp"), + RUNNING_VERSION, + BUILD_SOURCE, + ) +} + +fn require_identity(identity: LabIdentity) -> Result { + if identity.authoritative { + Ok(identity) + } else { + Err(identity + .warning + .clone() + .unwrap_or_else(|| "stale lab".into())) + } +} + +pub fn require_authoritative(cfg: &Config) -> Result { + require_identity(inspect(cfg)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn current_checkout_matches_the_compiled_identity() { + let identity = inspect_dir( + Path::new(env!("CARGO_MANIFEST_DIR")), + RUNNING_VERSION, + BUILD_SOURCE, + ); + assert!(identity.authoritative, "{identity:?}"); + assert!(identity.warning.is_none()); + } + + #[test] + fn older_running_version_is_non_authoritative() { + let identity = inspect_dir( + Path::new(env!("CARGO_MANIFEST_DIR")), + "0.29.0", + BUILD_SOURCE, + ); + assert!(!identity.authoritative); + let refusal = require_identity(identity).unwrap_err(); + assert!(refusal.contains("0.29.0")); + assert!(refusal.contains("Quality-gate verdicts are refused")); + assert!(refusal.contains("cargo install --locked")); + } + + #[test] + fn different_source_fingerprint_is_non_authoritative() { + let identity = inspect_dir( + Path::new(env!("CARGO_MANIFEST_DIR")), + RUNNING_VERSION, + "fnv1a64:0000000000000000", + ); + assert!(!identity.authoritative); + assert!(identity + .warning + .unwrap() + .contains("Quality-gate verdicts are refused")); + } +} diff --git a/tools/argus_mcp/src/lab.rs b/tools/argus_mcp/src/lab.rs index 96c1315..b0adc41 100644 --- a/tools/argus_mcp/src/lab.rs +++ b/tools/argus_mcp/src/lab.rs @@ -7,6 +7,7 @@ use serde::Serialize; #[derive(Debug, Clone, Serialize)] pub struct LabStatus { + pub lab_identity: crate::build_identity::LabIdentity, pub ready: bool, pub config: ConfigReport, pub maps: Vec, @@ -61,6 +62,7 @@ pub fn lab_status(cfg: &Config, live: Option) -> LabStatus { .collect(); let recommend = recommend(ready, &maps, &recent_runs, live.as_ref()); LabStatus { + lab_identity: crate::build_identity::inspect(cfg), ready, config, maps, diff --git a/tools/argus_mcp/src/lib.rs b/tools/argus_mcp/src/lib.rs index d7ac48f..ac7c182 100644 --- a/tools/argus_mcp/src/lib.rs +++ b/tools/argus_mcp/src/lib.rs @@ -2,6 +2,7 @@ pub mod analyze; pub mod backup; pub mod benchmark; pub mod bsp; +pub mod build_identity; pub mod campaign; pub mod cartograph; pub(crate) mod child_process; @@ -34,6 +35,7 @@ pub mod see_alias; pub mod server; pub mod session; pub mod soak; +mod source_fingerprint; pub mod stale; pub mod stats; pub mod tape_view; diff --git a/tools/argus_mcp/src/main.rs b/tools/argus_mcp/src/main.rs index 3b1f8ec..afd92b6 100644 --- a/tools/argus_mcp/src/main.rs +++ b/tools/argus_mcp/src/main.rs @@ -547,6 +547,8 @@ Judge candidate tapes against control tapes as bands. With no controls, the map' } let ctrls = positional.get(1).map(|s| split(s)).unwrap_or_default(); let cfg = argus_mcp::config::Config::load().map_err(|e| anyhow::anyhow!("{e:?}"))?; + argus_mcp::build_identity::require_authoritative(&cfg) + .map_err(|e| anyhow::anyhow!(e))?; let report = if ctrls.is_empty() { argus_mcp::intel::compare_runs_band(&cfg, &cands, None, primary.as_deref()) .map_err(|e| anyhow::anyhow!(e))? diff --git a/tools/argus_mcp/src/server.rs b/tools/argus_mcp/src/server.rs index f907b7a..1f3ddf1 100644 --- a/tools/argus_mcp/src/server.rs +++ b/tools/argus_mcp/src/server.rs @@ -542,6 +542,13 @@ fn json_value(value: &T) -> Result { if let (Some(note), serde_json::Value::Object(map)) = (crate::stale::banner(), &mut v) { map.insert("lab_stale".into(), serde_json::Value::String(note)); } + if let (Ok(cfg), serde_json::Value::Object(map)) = (Config::load(), &mut v) { + map.insert( + "lab_identity".into(), + serde_json::to_value(crate::build_identity::inspect(&cfg)) + .map_err(|e| McpError::internal_error(e.to_string(), None))?, + ); + } Ok(v) } @@ -591,10 +598,22 @@ fn report_json_ok( } fn csv_text(body: String) -> String { - match crate::stale::banner() { - Some(note) => format!("# lab_stale: {note}\n{body}"), - None => body, + let mut headers = String::new(); + if let Ok(cfg) = Config::load() { + let identity = crate::build_identity::inspect(&cfg); + headers.push_str(&format!( + "# lab_identity: running={} build={} checkout={} source={} authoritative={}\n", + identity.running_version, + identity.build_source, + identity.checkout_version.as_deref().unwrap_or("unreadable"), + identity.checkout_source.as_deref().unwrap_or("unreadable"), + identity.authoritative, + )); + } + if let Some(note) = crate::stale::banner() { + headers.push_str(&format!("# lab_stale: {note}\n")); } + format!("{headers}{body}") } fn report_csv_ok(body: String) -> Result { @@ -630,7 +649,14 @@ fn png_block(path: &str) -> Option { fn tool_err(msg: impl Into) -> Result { let error = msg.into(); let hint = hint_for(&error); - let body = serde_json::json!({ "error": error, "hint": hint }); + let mut body = serde_json::json!({ "error": error, "hint": hint }); + if let (Ok(cfg), serde_json::Value::Object(map)) = (Config::load(), &mut body) { + map.insert( + "lab_identity".into(), + serde_json::to_value(crate::build_identity::inspect(&cfg)) + .map_err(|e| McpError::internal_error(e.to_string(), None))?, + ); + } Ok(CallToolResult::error(vec![ContentBlock::text( serde_json::to_string_pretty(&body).unwrap_or(error), )])) @@ -1628,6 +1654,9 @@ impl Argus { Ok(c) => c, Err(r) => return Ok(r), }; + if let Err(error) = crate::build_identity::require_authoritative(&cfg) { + return tool_err(error); + } let mut map = args.map.clone(); if args.log_a.is_none() && map.is_none() @@ -1853,6 +1882,9 @@ impl Argus { Err(r) => return Ok(r), }; if let Some(b) = args.log_b.as_deref() { + if let Err(error) = crate::build_identity::require_authoritative(&cfg) { + return tool_err(error); + } let a = args.log_a.as_deref().unwrap_or("baseline"); match intel_compare(&cfg, a, b, args.map.as_deref()) { Ok(r) => json_ok(&r.next_steps), @@ -2360,6 +2392,9 @@ impl Argus { Ok(c) => c, Err(r) => return Ok(r), }; + if let Err(error) = crate::build_identity::require_authoritative(&cfg) { + return tool_err(error); + } let progress = ProgressReporter::new(meta, peer); let repeats = args.repeats.unwrap_or(3).clamp(1, 5); let compile_units = u32::from(args.compile.unwrap_or(true)); @@ -2553,6 +2588,9 @@ impl Argus { Ok(c) => c, Err(r) => return Ok(r), }; + if let Err(error) = crate::build_identity::require_authoritative(&cfg) { + return tool_err(error); + } let progress = ProgressReporter::new(meta, peer); let compile_units = u32::from(args.compile.unwrap_or(true)); let total = f64::from(compile_units + 2); @@ -2713,6 +2751,9 @@ impl Argus { Ok(c) => c, Err(r) => return Ok(r), }; + if let Err(error) = crate::build_identity::require_authoritative(&cfg) { + return tool_err(error); + } let progress = ProgressReporter::new(meta, peer); let compile_units = usize::from(args.compile.unwrap_or(true)); let total = (compile_units + maps.len() + 1) as f64; diff --git a/tools/argus_mcp/src/source_fingerprint.rs b/tools/argus_mcp/src/source_fingerprint.rs new file mode 100644 index 0000000..4926174 --- /dev/null +++ b/tools/argus_mcp/src/source_fingerprint.rs @@ -0,0 +1,53 @@ +//! Small, dependency-free source fingerprint shared by build.rs and runtime. + +use std::path::{Path, PathBuf}; + +const FNV_OFFSET: u64 = 0xcbf29ce484222325; +const FNV_PRIME: u64 = 0x100000001b3; + +fn collect_rs(dir: &Path, files: &mut Vec) -> Result<(), String> { + let entries = std::fs::read_dir(dir).map_err(|error| format!("{}: {error}", dir.display()))?; + for entry in entries { + let path = entry.map_err(|error| error.to_string())?.path(); + if path.is_dir() { + collect_rs(&path, files)?; + } else if path.extension().is_some_and(|extension| extension == "rs") { + files.push(path); + } + } + Ok(()) +} + +fn feed(hash: &mut u64, bytes: &[u8]) { + for byte in bytes { + *hash ^= u64::from(*byte); + *hash = hash.wrapping_mul(FNV_PRIME); + } +} + +/// Fingerprint every Rust source plus the files that select dependencies and +/// build-time behavior. Paths participate so a rename is a different build. +pub fn fingerprint(manifest_dir: &Path) -> Result { + let mut files = vec![ + manifest_dir.join("Cargo.toml"), + manifest_dir.join("Cargo.lock"), + manifest_dir.join("build.rs"), + ]; + collect_rs(&manifest_dir.join("src"), &mut files)?; + files.sort(); + + let mut hash = FNV_OFFSET; + for path in files { + let relative = path + .strip_prefix(manifest_dir) + .map_err(|error| error.to_string())? + .to_string_lossy() + .replace('\\', "/"); + let bytes = std::fs::read(&path).map_err(|error| format!("{}: {error}", path.display()))?; + feed(&mut hash, relative.as_bytes()); + feed(&mut hash, &[0]); + feed(&mut hash, &bytes); + feed(&mut hash, &[0xff]); + } + Ok(format!("fnv1a64:{hash:016x}")) +}