From 57bfcfce778a98d8cb672f25407dca18dde42d45 Mon Sep 17 00:00:00 2001 From: Shane Wall Date: Mon, 21 Sep 2026 11:58:56 +1000 Subject: [PATCH] Measure map-matched movement humanness (#369) --- CHANGELOG.md | 7 + tools/argus_humanness.py | 384 ++++++++++++++++++++++++++++++++++++++ tools/argus_mcp/README.md | 16 ++ tools/test_tools_cli.py | 100 ++++++++++ 4 files changed, 507 insertions(+) create mode 100644 tools/argus_humanness.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 010636f..b400d95 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,13 @@ is in `tools/argus_mcp/README.md`. ## Unreleased +**HUMANNESS HAS A MAP-MATCHED MOVEMENT DISTANCE** (#369). The new +`tools/argus_humanness.py` compares bot track distributions with same-map +human `ARGLOG` references across speed, pauses, direction changes, roaming +and local dwell. It intentionally excludes demo aim streams whose differing +precision and PVS visibility would let a classifier learn the recorder +instead of believable behaviour. + **MATCH TASKS NOW HAVE A CHILD-CONTROLLING LIFECYCLE HARNESS** (#491). A deterministic fake engine drives negotiated completion, active and late cancellation, finalization, tool errors, successors, and shutdown without diff --git a/tools/argus_humanness.py b/tools/argus_humanness.py new file mode 100644 index 0000000..d8236e6 --- /dev/null +++ b/tools/argus_humanness.py @@ -0,0 +1,384 @@ +#!/usr/bin/env python3 +"""Map-matched movement distance between Argus bots and human tracks. + +The input is ordinary ARGLOG telemetry, not demo entity angles. Human and bot +positions therefore have the same cadence and precision, and bot tracks are not +PVS-culled. Lower scores mean the bot-track distribution is closer to the +human reference distribution; the score is evidence, never a quality gate. + +Examples: + python tools/argus_humanness.py --human + python tools/argus_humanness.py --reference "runs/shane_*.log" runs/ab_dm2_*.log + python tools/argus_humanness.py --format json runs/candidate.log +""" + +import argparse +import glob +import json +import math +import os +import re +import statistics +import sys +from collections import defaultdict +from pathlib import Path + + +SAMPLE = re.compile( + r"(?:BOTLOG|ARGLOG) (.+?) t\s+([\d.]+) pos '\s*(-?[\d.]+)\s+" + r"(-?[\d.]+)\s+(-?[\d.]+)' spd\s+(-?[\d.]+) yaw\s+(-?[\d.]+)" +) +SPAWN = re.compile(r"ARGEVT (.+?) (?:spawned|respawn)\b") +MAP = re.compile(r"ARGUS init on ([A-Za-z0-9_]+)\b") +MAP_FALLBACK = re.compile(r"SpawnServer:\s*([A-Za-z0-9_]+)\b") +INSTRUMENTS = {"labprobe", "unconnected"} +FEATURES = ( + "speed_median", + "active_speed_p90", + "pause_fraction", + "pause_mean_sec", + "direction_changes_pm", + "cells_pm", + "gyration", + "local_dwell_fraction", +) +# A reference can be genuinely tight (dm2 human median speed differs by less +# than 1 u/s across several sessions). Dividing by that tiny IQR would let one +# feature swamp the entire score. These floors are practical effect sizes in +# the feature's native unit, not fitted to a candidate. +SCALE_FLOORS = { + "speed_median": 32.0, + "active_speed_p90": 32.0, + "pause_fraction": 0.05, + "pause_mean_sec": 0.5, + "direction_changes_pm": 5.0, + "cells_pm": 10.0, + "gyration": 128.0, + "local_dwell_fraction": 0.10, +} + + +def percentile(values, q): + values = sorted(values) + if not values: + raise ValueError("percentile of an empty sequence") + if len(values) == 1: + return values[0] + position = (len(values) - 1) * q + lo = int(math.floor(position)) + hi = int(math.ceil(position)) + if lo == hi: + return values[lo] + weight = position - lo + return values[lo] * (1 - weight) + values[hi] * weight + + +def read_tape(path): + """Return map and role-separated tracks from one telemetry tape.""" + tracks = defaultdict(list) + spawned = set() + map_names = set() + fallback_map = None + with open(path, encoding="utf-8", errors="replace") as tape: + for line in tape: + match = MAP.search(line) + if match: + map_names.add(match.group(1).lower()) + elif fallback_map is None: + match = MAP_FALLBACK.search(line) + if match: + fallback_map = match.group(1).lower() + match = SPAWN.search(line) + if match: + spawned.add(match.group(1)) + match = SAMPLE.search(line) + if match: + name, t, x, y, z, speed, yaw = match.groups() + tracks[name].append( + { + "t": float(t), + "x": float(x), + "y": float(y), + "z": float(z), + "speed": max(0.0, float(speed)), + "yaw": float(yaw), + } + ) + for samples in tracks.values(): + samples.sort(key=lambda sample: sample["t"]) + # Match the Rust parser: without any spawn evidence, a legacy or sliced + # tape cannot distinguish a person from a bot and must not invent one. + humans = { + name: samples + for name, samples in tracks.items() + if spawned and name not in spawned and name not in INSTRUMENTS + } + bots = { + name: samples + for name, samples in tracks.items() + if name in spawned and name not in INSTRUMENTS + } + return { + "file": os.path.basename(path), + "path": str(path), + # A multi-map console log cannot support one map-matched distance. + "map": ( + next(iter(map_names)) + if len(map_names) == 1 + else (fallback_map if not map_names else None) + ), + "humans": humans, + "bots": bots, + } + + +def _pause_lengths(samples): + pauses = [] + start = None + previous = None + gaps = [ + b["t"] - a["t"] + for a, b in zip(samples, samples[1:]) + if 0 < b["t"] - a["t"] <= 2.0 + ] + cadence = statistics.median(gaps) if gaps else 0.0 + for sample in samples: + contiguous = previous is None or 0 < sample["t"] - previous["t"] <= 2.0 + if sample["speed"] < 20 and contiguous: + if start is None: + start = sample["t"] + else: + if start is not None and previous is not None: + pauses.append(max(cadence, previous["t"] - start + cadence)) + start = sample["t"] if sample["speed"] < 20 else None + previous = sample + if start is not None and previous is not None: + pauses.append(max(cadence, previous["t"] - start + cadence)) + return pauses + + +def _direction_changes(samples): + headings = [] + for first, second in zip(samples, samples[1:]): + dt = second["t"] - first["t"] + dx = second["x"] - first["x"] + dy = second["y"] - first["y"] + distance = math.hypot(dx, dy) + if not 0 < dt <= 2.0 or distance < 20 or distance > 512: + headings.append(None) + else: + headings.append(math.degrees(math.atan2(dy, dx))) + changes = 0 + previous = None + for heading in headings: + if heading is None: + previous = None + continue + if previous is not None: + delta = abs((heading - previous + 180) % 360 - 180) + if delta >= 45: + changes += 1 + previous = heading + return changes + + +def _local_dwell_fraction(samples, window_sec=30.0, radius=1024.0): + """Fraction of full trailing windows confined to one 1024u neighbourhood.""" + local = 0 + windows = 0 + left = 0 + for right, sample in enumerate(samples): + while left < right and sample["t"] - samples[left]["t"] > window_sec: + left += 1 + window = samples[left : right + 1] + if len(window) < 2 or window[-1]["t"] - window[0]["t"] < window_sec * 0.9: + continue + cx = sum(point["x"] for point in window) / len(window) + cy = sum(point["y"] for point in window) / len(window) + extent = max(math.hypot(point["x"] - cx, point["y"] - cy) for point in window) + windows += 1 + local += extent <= radius + return local / windows if windows else 0.0 + + +def track_metrics(samples): + """Summarize one sufficiently long, same-cadence ARGLOG track.""" + if len(samples) < 20: + return None + duration = samples[-1]["t"] - samples[0]["t"] + if duration < 10: + return None + speeds = [sample["speed"] for sample in samples] + active = [speed for speed in speeds if speed >= 20] + pauses = _pause_lengths(samples) + cx = sum(sample["x"] for sample in samples) / len(samples) + cy = sum(sample["y"] for sample in samples) / len(samples) + gyration = math.sqrt( + sum((sample["x"] - cx) ** 2 + (sample["y"] - cy) ** 2 for sample in samples) + / len(samples) + ) + cells = { + ( + math.floor(sample["x"] / 64), + math.floor(sample["y"] / 64), + math.floor(sample["z"] / 64), + ) + for sample in samples + } + return { + "speed_median": statistics.median(speeds), + "active_speed_p90": percentile(active or speeds, 0.90), + "pause_fraction": sum(speed < 20 for speed in speeds) / len(speeds), + "pause_mean_sec": statistics.mean(pauses) if pauses else 0.0, + "direction_changes_pm": _direction_changes(samples) * 60.0 / duration, + "cells_pm": len(cells) * 60.0 / duration, + "gyration": gyration, + "local_dwell_fraction": _local_dwell_fraction(samples), + } + + +def build_reference(tapes): + """Collect human track summaries independently for every map.""" + reference = defaultdict(lambda: defaultdict(list)) + counts = defaultdict(int) + for tape in tapes: + if not tape["map"]: + continue + for samples in tape["humans"].values(): + metrics = track_metrics(samples) + if metrics is None: + continue + counts[tape["map"]] += 1 + for feature in FEATURES: + reference[tape["map"]][feature].append(metrics[feature]) + return reference, counts + + +def normalized_wasserstein(candidate, human, scale_floor=1.0): + """Empirical W1 distance, scaled by the human reference spread.""" + if not candidate or not human: + return None + samples = max(len(candidate), len(human), 21) + distances = [] + for index in range(samples): + q = (index + 0.5) / samples + distances.append(abs(percentile(candidate, q) - percentile(human, q))) + raw = statistics.mean(distances) + spread = percentile(human, 0.75) - percentile(human, 0.25) + spread = max(spread, scale_floor) + return raw / spread + + +def score_tape(tape, reference, reference_counts): + bot_metrics = [ + metrics + for samples in tape["bots"].values() + if (metrics := track_metrics(samples)) is not None + ] + if not tape["map"] or tape["map"] not in reference or not bot_metrics: + return None + distances = {} + for feature in FEATURES: + distances[feature] = normalized_wasserstein( + [metrics[feature] for metrics in bot_metrics], + reference[tape["map"]][feature], + SCALE_FLOORS[feature], + ) + available = [value for value in distances.values() if value is not None] + return { + "file": tape["file"], + "map": tape["map"], + "bot_tracks": len(bot_metrics), + "human_reference_tracks": reference_counts[tape["map"]], + "distance": statistics.mean(available), + **distances, + } + + +def expand_paths(patterns): + paths = [] + seen = set() + for pattern in patterns: + matches = glob.glob(pattern) + for value in matches or [pattern]: + path = str(Path(value)) + if path not in seen and Path(path).is_file(): + paths.append(path) + seen.add(path) + return paths + + +def format_tsv(rows): + columns = ( + "file", + "map", + "bot_tracks", + "human_reference_tracks", + "distance", + *FEATURES, + ) + lines = ["\t".join(columns)] + for row in rows: + lines.append( + "\t".join( + f"{row[column]:.3f}" if isinstance(row[column], float) else str(row[column]) + for column in columns + ) + ) + return "\n".join(lines) + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("logs", nargs="*", help="candidate logs or glob patterns") + parser.add_argument( + "--reference", + action="append", + default=[], + metavar="GLOB", + help="human reference logs (repeatable; default: runs/shane_*.log)", + ) + parser.add_argument( + "--human", + action="store_true", + help="score bots in the human reference sessions themselves", + ) + parser.add_argument("--format", choices=("tsv", "json"), default="tsv") + args = parser.parse_args(argv) + + root = Path(__file__).resolve().parent.parent + reference_patterns = args.reference or [str(root / "runs" / "shane_*.log")] + reference_paths = expand_paths(reference_patterns) + if not reference_paths: + parser.error("no human reference logs matched") + candidate_paths = reference_paths if args.human or not args.logs else expand_paths(args.logs) + if not candidate_paths: + parser.error("no candidate logs matched") + + reference_tapes = [read_tape(path) for path in reference_paths] + reference, counts = build_reference(reference_tapes) + if not reference: + parser.error("reference logs contain no usable human ARGLOG tracks") + + rows = [] + skipped = [] + for path in candidate_paths: + tape = read_tape(path) + row = score_tape(tape, reference, counts) + if row is None: + skipped.append(tape["file"]) + else: + rows.append(row) + if not rows: + parser.error("no candidate had both usable bot tracks and a same-map human reference") + if skipped: + print("skipped without usable bots or same-map reference: " + ", ".join(skipped), file=sys.stderr) + if args.format == "json": + print(json.dumps(rows, indent=2, sort_keys=True)) + else: + print(format_tsv(rows)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/argus_mcp/README.md b/tools/argus_mcp/README.md index 91ddd62..5e4c676 100644 --- a/tools/argus_mcp/README.md +++ b/tools/argus_mcp/README.md @@ -1052,6 +1052,22 @@ target is zero. `tools/argus_tick.py` reports the rate of any tape or of the whole archive (`--all`). +`tools/argus_humanness.py` turns the trustworthy movement half of that +archive into a descriptive distance. It builds a separate human reference +for each map from ordinary `ARGLOG` tracks, then compares bot-track +distributions for speed, pauses, direction changes, cells per minute, +gyration and 30-second local dwell inside a 1024-unit neighbourhood. The +distance uses human interquartile spread with practical-effect floors, so a +nearly constant reference feature cannot dominate the combined score. The +shared telemetry stream avoids the +demo recorder's entity-angle precision and PVS-culling confounds. Lower is +closer to the human reference, but the score is evidence, not a quality gate: + +``` +python tools/argus_humanness.py --human +python tools/argus_humanness.py --reference "runs/shane_*.log" runs/candidate.log +``` + ## Quality bars Encoded from the project charter, not invented per call: diff --git a/tools/test_tools_cli.py b/tools/test_tools_cli.py index 9c460ac..21c2487 100644 --- a/tools/test_tools_cli.py +++ b/tools/test_tools_cli.py @@ -2,6 +2,7 @@ """CLI regression tests for developer scripts in tools/.""" import hashlib import json +import math import os import shutil import subprocess @@ -831,6 +832,105 @@ def test_longi_reproduces_the_v405_row(self): self.assertEqual(row["freezes"], "0") self.assertEqual(row["unstick"], "0") + def test_humanness_reader_uses_the_rust_role_rule(self): + sys.path.insert(0, str(ROOT / "tools")) + import argus_humanness + + with tempfile.TemporaryDirectory() as td: + tape = Path(td) / "roles.log" + lines = ["ARGUS init on dm4\n", "ARGEVT Reap spawned\n"] + for i in range(20): + lines.extend( + ( + f"ARGLOG Reap t {i}.0 pos '{i * 64} 0 24' spd 64 yaw 0 mode 0 st 0 gl 0 hp 100 frg 0\n", + f"ARGLOG Shane t {i}.0 pos '0 {i * 64} 24' spd 64 yaw 90 mode 0 st 0 gl 0 hp 100 frg 0\n", + f"ARGLOG labprobe t {i}.0 pos '0 0 24' spd 0 yaw 0 mode 0 st 0 gl 0 hp 100 frg 0\n", + ) + ) + tape.write_text("".join(lines), encoding="utf-8") + parsed = argus_humanness.read_tape(tape) + self.assertEqual(parsed["map"], "dm4") + self.assertEqual(set(parsed["bots"]), {"Reap"}) + self.assertEqual(set(parsed["humans"]), {"Shane"}) + + # A tail slice without spawn evidence cannot label every track human. + tape.write_text("".join(line for line in lines if "ARGEVT" not in line), encoding="utf-8") + parsed = argus_humanness.read_tape(tape) + self.assertEqual(parsed["humans"], {}) + self.assertEqual(parsed["bots"], {}) + + def test_humanness_distance_is_zero_for_equal_distributions(self): + sys.path.insert(0, str(ROOT / "tools")) + import argus_humanness + + self.assertAlmostEqual( + argus_humanness.normalized_wasserstein([1, 2, 4], [1, 2, 4]), + 0.0, + ) + self.assertGreater( + argus_humanness.normalized_wasserstein([10, 20, 40], [1, 2, 4]), + 1.0, + ) + + def test_humanness_cli_reports_map_matched_movement_distance(self): + def tape(path, human_scale, bot_scale): + lines = ["ARGUS init on dm4\n", "ARGEVT Reap spawned\n"] + for i in range(40): + hx = i * human_scale + bx = i * bot_scale + lines.extend( + ( + f"ARGLOG Shane t {i}.0 pos '{hx} {i % 4 * 64} 24' spd {human_scale} yaw 0 mode 0 st 0 gl 0 hp 100 frg 0\n", + f"ARGLOG Reap t {i}.0 pos '{bx} {i % 4 * 64} 24' spd {bot_scale} yaw 0 mode 0 st 0 gl 0 hp 100 frg 0\n", + ) + ) + path.write_text("".join(lines), encoding="utf-8") + + with tempfile.TemporaryDirectory() as td: + root = Path(td) + references = [] + for index, scale in enumerate((58, 64, 70)): + path = root / f"human_{index}.log" + tape(path, scale, scale) + references.append(path) + candidate = root / "candidate.log" + tape(candidate, 64, 64) + res = self.run_tool( + "argus_humanness.py", + "--reference", + str(root / "human_*.log"), + str(candidate), + ) + self.assertEqual(res.returncode, 0, res.stderr) + lines = res.stdout.splitlines() + row = dict(zip(lines[0].split("\t"), lines[1].split("\t"))) + self.assertEqual(row["map"], "dm4") + self.assertEqual(row["bot_tracks"], "1") + self.assertEqual(row["human_reference_tracks"], "3") + self.assertLess(float(row["distance"]), 1.0) + + def test_humanness_scores_a_real_tape_against_the_archive(self): + tape = ROOT / "runs" / "ab_dm4_B3.log" + if not tape.is_file(): + self.skipTest("dm4 candidate tape not present") + res = self.run_tool("argus_humanness.py", str(tape)) + self.assertEqual(res.returncode, 0, res.stderr) + lines = res.stdout.splitlines() + row = dict(zip(lines[0].split("\t"), lines[1].split("\t"))) + self.assertEqual(row["map"], "dm4") + self.assertEqual(row["bot_tracks"], "3") + self.assertGreaterEqual(int(row["human_reference_tracks"]), 3) + self.assertGreater(float(row["distance"]), 0.0) + for feature in ( + "speed_median", + "pause_fraction", + "direction_changes_pm", + "cells_pm", + "gyration", + "local_dwell_fraction", + ): + self.assertTrue(math.isfinite(float(row[feature])), feature) + def test_tick_estimator_separates_listen_from_dedicated(self): """A tape says which game it recorded, or it cannot be compared.""" pairs = [