From 4bc8a38f70abe32a9c5ad1e06fbf52c11f44e90b Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:35:51 +0800 Subject: [PATCH] feat(bench): add condition matrix, tracing, oracle grading, and report to the agent benchmark Refs #414. Implements the harness parts of SPEC-414: wright/knowledge/network cells with canaries and a scrubbed environment, a matrix runner with retries, the wright shim with envelope and serve tracing, entry snapshots, expectation detectors and friction metrics, per-turn usage and context reporting, upstream-oracle grading with disagreement output, negative-overlay calibration, and a report with intervals, paired comparison, and eval diagnostics. Adds a Claude Code reference adapter. Scenarios for the new OPY requirements and the guide-tuning loop are not included. --- .gitignore | 3 + benchmarks/agent/adapters/claude_code.py | 86 ++++ benchmarks/agent/agent_bench.py | 383 +++++++++++------- benchmarks/agent/bench_grade.py | 208 ++++++++++ benchmarks/agent/bench_report.py | 209 ++++++++++ benchmarks/agent/bench_trace.py | 278 +++++++++++++ benchmarks/agent/oracle/compile.js | 18 + benchmarks/agent/oracle/package-lock.json | 20 + benchmarks/agent/oracle/package.json | 5 + .../repair-runaway-loop/scenario.json | 1 + benchmarks/agent/test_agent_bench.py | 205 +++++++++- docs/agent-benchmark.md | 157 +++++-- .../SPEC-414-agent-benchmark-comparison.md | 18 +- 13 files changed, 1386 insertions(+), 205 deletions(-) create mode 100755 benchmarks/agent/adapters/claude_code.py create mode 100644 benchmarks/agent/bench_grade.py create mode 100644 benchmarks/agent/bench_report.py create mode 100644 benchmarks/agent/bench_trace.py create mode 100644 benchmarks/agent/oracle/compile.js create mode 100644 benchmarks/agent/oracle/package-lock.json create mode 100644 benchmarks/agent/oracle/package.json diff --git a/.gitignore b/.gitignore index b394305f..18ed3a2a 100644 --- a/.gitignore +++ b/.gitignore @@ -30,3 +30,6 @@ target/ .agents/ .opencode/ + +# Agent benchmark upstream oracle +benchmarks/agent/oracle/node_modules/ diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py new file mode 100755 index 00000000..c20dc27c --- /dev/null +++ b/benchmarks/agent/adapters/claude_code.py @@ -0,0 +1,86 @@ +#!/usr/bin/env python3 +"""Reference adapter: run Claude Code for one benchmark trial and report per-turn usage. + +Reads the prompt on stdin. Honors the BENCH_* contract (docs/agent-benchmark.md): tools follow BENCH_KNOWLEDGE, +BENCH_SKILL_DIR is installed as a plugin, and BENCH_USAGE / BENCH_TRANSCRIPT / BENCH_CONTEXT are written. +The agent binary is resolved on BENCH_HOST_PATH because the agent's own PATH hides Wright when the level is `none`. +The model comes from BENCH_MODEL (default `sonnet`). It removes web tools unless knowledge is `web`, but it does +not sandbox the network: pair it with the harness --canary-cmd to detect a reachable network under `off`. +Exit 75 signals a provider or infrastructure failure so the harness retries the trial. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +INFRA_EXIT = 75 +TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] +WEB_TOOLS = ["WebFetch", "WebSearch"] + + +def main() -> int: + env = os.environ + prompt = sys.stdin.read() + web = env["BENCH_KNOWLEDGE"] == "web" + cmd = [ + shutil.which("claude", path=env.get("BENCH_HOST_PATH")) or "claude", "-p", "--model", env.get("BENCH_MODEL", "sonnet"), + "--output-format", "stream-json", "--verbose", "--setting-sources", "project", "--strict-mcp-config", + "--no-session-persistence", "--permission-mode", "acceptEdits", + "--allowedTools", *(TOOLS + WEB_TOOLS if web else TOOLS), + ] + if not web: + cmd += ["--disallowedTools", *WEB_TOOLS] + loaded: list[str] = [] + if env.get("BENCH_SKILL_DIR"): + skill = Path(env["BENCH_SKILL_DIR"]) + plugin = Path(tempfile.mkdtemp(dir=env["BENCH_RUN_DIR"])) + (plugin / ".claude-plugin").mkdir() + (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skill"})) + shutil.copytree(skill, plugin / "skills" / skill.name) + cmd += ["--plugin-dir", str(plugin)] + loaded.append(skill.name) + proc = subprocess.Popen( + cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, + ) + proc.stdin.write(prompt) + proc.stdin.close() + final, errored = "", False + with open(env["BENCH_USAGE"], "w") as usage, open(env["BENCH_TRANSCRIPT"], "w") as transcript: + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + transcript.write(json.dumps({"t": time.time(), **event}) + "\n") + if event.get("type") == "assistant": + u = event["message"].get("usage") or {} + cached = u.get("cache_read_input_tokens") or 0 + written = u.get("cache_creation_input_tokens") or 0 + usage.write(json.dumps({ + "t": time.time(), "input": u.get("input_tokens"), "output": u.get("output_tokens"), + "cache_read": cached, "cache_write": written, "reasoning": None, + "context": (u.get("input_tokens") or 0) + cached + written, "context_limit": None, + }) + "\n") + if event.get("type") == "result": + final, errored = event.get("result", ""), bool(event.get("is_error")) + stderr = proc.stderr.read() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + sys.stdout.write(final) + sys.stderr.write(stderr) + if errored or code != 0: + transient = any(s in stderr.lower() + final.lower() for s in ("overloaded", "rate limit", "529", "timed out")) + return INFRA_EXIT if transient else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index cc86940f..7e36cf60 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -1,25 +1,35 @@ #!/usr/bin/env python3 -"""Wright agent benchmark harness (#414). Contract: docs/agent-benchmark.md.""" +"""Wright agent benchmark harness (#414). Contract: docs/agent-benchmark.md; design: docs/specs/SPEC-414-agent-benchmark-comparison.md.""" from __future__ import annotations import argparse +import hashlib import json import os import platform +import random import shutil import signal import subprocess import sys import time +from concurrent.futures import ThreadPoolExecutor from datetime import datetime, timezone from pathlib import Path +import bench_grade +import bench_report +import bench_trace + HERE = Path(__file__).resolve().parent ROOT = HERE.parent.parent SCENARIOS = HERE / "scenarios" -RESULT_CONTRACT = "wright-agent-bench/v1" -SHIM_ENV = ("WRIGHT_BENCH_REAL", "WRIGHT_BENCH_TRACE") +RESULT_CONTRACT = "wright-agent-bench/v2" +WRIGHT_LEVELS = ("none", "bin", "bin+skill") +KNOWLEDGE_LEVELS = ("none", "wiki", "web") +INFRA_EXIT = 75 # EX_TEMPFAIL: the adapter reports a provider or infrastructure failure, not an agent failure +ENV_KEEP = ("LANG", "LC_ALL", "TERM", "TMPDIR", "USER", "LOGNAME") def load_scenario(scenario_id: str) -> dict: @@ -33,86 +43,6 @@ def all_scenario_ids() -> list[str]: return sorted(p.name for p in SCENARIOS.iterdir() if (p / "scenario.json").is_file()) -def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: - proc = subprocess.run([wright, *args, "--format", "json"], capture_output=True, text=True) - try: - return proc.returncode, json.loads(proc.stdout) - except json.JSONDecodeError: - return proc.returncode, {"diagnostics": [{"code": "harness-output", "severity": "error", "message": proc.stderr.strip() or proc.stdout.strip()}]} - - -def serve_request(wright: str, entry: Path, request: dict) -> dict: - proc = subprocess.run([wright, "serve", str(entry)], input=json.dumps(request) + "\n", capture_output=True, text=True) - try: - return json.loads(proc.stdout.splitlines()[0]) - except (IndexError, json.JSONDecodeError): - return {"error": {"code": "harness-output", "message": proc.stderr.strip()}} - - -def check_result(check: dict, passed: bool, detail: str) -> dict: - return {"id": check["id"], "kind": check["kind"], "layer": check.get("layer", "agent"), "passed": passed, "detail": detail} - - -def run_check(check: dict, workspace: Path, entry: Path, wright: str, state: dict) -> dict: - kind = check["kind"] - if kind == "check": - code, envelope = wright_json(wright, ["check", str(entry)]) - state["diagnostics"] = envelope.get("diagnostics", []) - errors = [d for d in state["diagnostics"] if d.get("severity") == "error"] - return check_result(check, code == 0 and not errors, f"exit {code}, {len(errors)} error diagnostic(s)") - if kind == "lint": - _, envelope = wright_json(wright, ["lint", str(entry)]) - findings = [f for f in envelope.get("result", {}).get("findings", []) if f["code"] == check["code"]] - return check_result(check, len(findings) <= check.get("max", 0), f"{len(findings)} '{check['code']}' finding(s)") - if kind == "symbols": - response = serve_request(wright, entry, {"op": "symbols", "kind": check["symbolKind"]}) - found = len(response.get("result", [])) - return check_result(check, found >= check.get("min", 1), f"{found} '{check['symbolKind']}' symbol(s)") - if kind in ("contains", "absent"): - path = workspace / check["file"] - text = path.read_text() if path.is_file() else "" - texts = check["text"] if isinstance(check["text"], list) else [check["text"]] - count = sum(text.count(t) for t in texts) - if kind == "contains": - passed = count >= check.get("min", 1) and count <= check.get("max", count) - else: - passed = count == 0 - return check_result(check, passed, f"{count} occurrence(s) in {check['file']}") - if kind == "answer": - path = workspace / "answer.json" - try: - actual = json.loads(path.read_text()).get(check["key"]) - except (OSError, json.JSONDecodeError, AttributeError): - actual = None - return check_result(check, actual == check["expected"], f"answer[{check['key']}] = {json.dumps(actual)}") - raise SystemExit(f"unknown check kind '{kind}'") - - -def tree(root: Path) -> dict[str, bytes]: - return {str(p.relative_to(root)): p.read_bytes() for p in sorted(root.rglob("*")) if p.is_file()} - - -def unsafe_edits(scenario: dict, workspace: Path) -> list[str]: - seed = tree(scenario["dir"] / "seed") - now = tree(workspace) - writable = set(scenario["writable"]) - return sorted(name for name in seed.keys() | now.keys() if seed.get(name) != now.get(name) and name not in writable) - - -def grade(scenario: dict, workspace: Path, wright: str) -> dict: - entry = workspace / scenario["entry"] - state: dict = {} - checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] - return { - "checks": checks, - "passed": all(c["passed"] for c in checks), - "failedLayers": sorted({c["layer"] for c in checks if not c["passed"]}), - "diagnostics": state.get("diagnostics", []), - "unsafeEdits": unsafe_edits(scenario, workspace), - "unverifiedRuntimeClaims": scenario.get("runtimeOnly", []), - } - - def materialize(scenario: dict, workspace: Path, overlay: str | None = None) -> None: shutil.copytree(scenario["dir"] / "seed", workspace) if overlay: @@ -120,21 +50,31 @@ def materialize(scenario: dict, workspace: Path, overlay: str | None = None) -> def validate(wright: str, out: Path) -> bool: - """Each scenario must be solvable by the reference and unsolved by its seed.""" + """Graders are calibrated: the reference passes, the seed fails, each negative fails exactly its named checks.""" ok = True for scenario_id in all_scenario_ids(): scenario = load_scenario(scenario_id) + runs = [("seed", None), ("reference", "reference")] + [(f"negative-{n}", f"negative/{n}") for n in scenario.get("negatives", {})] results = {} - for name, overlay in (("seed", None), ("reference", "reference")): + for name, overlay in runs: workspace = out / scenario_id / name shutil.rmtree(workspace, ignore_errors=True) materialize(scenario, workspace, overlay) - results[name] = grade(scenario, workspace, wright) - good = results["reference"]["passed"] and not results["seed"]["passed"] and not results["reference"]["unsafeEdits"] - if not good: + results[name] = bench_grade.grade(scenario, workspace, wright) + problems = [] + if not results["reference"]["passed"]: + problems.append(f"reference failures={[c['id'] for c in results['reference']['checks'] if not c['passed']]}") + if results["seed"]["passed"]: + problems.append("seed passed") + if results["reference"]["unsafeEdits"]: + problems.append(f"unsafe={results['reference']['unsafeEdits']}") + for name, expected in scenario.get("negatives", {}).items(): + failed = sorted(c["id"] for c in results[f"negative-{name}"]["checks"] if not c["passed"]) + if failed != sorted(expected["fails"]): + problems.append(f"negative '{name}' failed {failed}, expected {sorted(expected['fails'])}") + if problems: ok = False - failed = [c for c in results["reference"]["checks"] if not c["passed"]] - print(f"INVALID {scenario_id}: seed passed={results['seed']['passed']}, reference failures={failed}, unsafe={results['reference']['unsafeEdits']}") + print(f"INVALID {scenario_id}: {'; '.join(problems)}") else: print(f"ok {scenario_id}") return ok @@ -146,110 +86,253 @@ def baseline_path(path: str) -> str: return os.pathsep.join(kept) -def shim_main(argv: list[str]) -> int: - start = time.monotonic() - code = subprocess.call([os.environ["WRIGHT_BENCH_REAL"], *argv]) - with open(os.environ["WRIGHT_BENCH_TRACE"], "a") as trace: - trace.write(json.dumps({"argv": argv, "exit": code, "seconds": round(time.monotonic() - start, 3)}) + "\n") - return code +def cell_label(cell: dict) -> str: + return f"{cell['wright']}/{cell['knowledge']}/{cell['network']}" -def summarize_trace(trace: Path) -> dict: - calls = [json.loads(line) for line in trace.read_text().splitlines()] if trace.is_file() else [] - commands = [next((a for a in c["argv"] if not a.startswith("-")), "") for c in calls] - by_command: dict[str, int] = {} - for command in commands: - by_command[command] = by_command.get(command, 0) + 1 - return { - "invocations": len(calls), - "byCommand": by_command, - "failedInvocations": sum(1 for c in calls if c["exit"] != 0), - "ownerOrEnvironmentGaps": [c for c in calls if c["exit"] >= 3], - } +def check_cell(cell: dict, args: argparse.Namespace) -> None: + if cell["wright"] not in WRIGHT_LEVELS or cell["knowledge"] not in KNOWLEDGE_LEVELS or cell["network"] not in ("off", "on"): + raise SystemExit(f"invalid condition {cell_label(cell)}") + if cell["knowledge"] == "web" and cell["network"] != "on": + raise SystemExit("knowledge 'web' requires network 'on'") + if cell["wright"] == "bin+skill" and not args.skill_dir: + raise SystemExit("wright level 'bin+skill' requires --skill-dir") + if cell["knowledge"] == "wiki" and not args.wiki_dir: + raise SystemExit("knowledge 'wiki' requires --wiki-dir") -def run_trial(scenario: dict, condition: str, args: argparse.Namespace, out: Path) -> dict: - shutil.rmtree(out, ignore_errors=True) - workspace = out / "workspace" - materialize(scenario, workspace) - prompt = (scenario["dir"] / "prompt.md").read_text() - trace = out / "wright-trace.jsonl" - env = {k: v for k, v in os.environ.items() if k not in SHIM_ENV} - if condition == "wright": +def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) -> dict: + """Scrubbed environment: allowlisted host variables, a fresh HOME, and the BENCH_* contract for the adapter.""" + home = out / "home" + home.mkdir(parents=True) + env = {k: os.environ[k] for k in (*ENV_KEEP, *args.env_pass) if k in os.environ} + env.setdefault("HOME", str(home)) + path = baseline_path(os.environ["PATH"]) + if cell["wright"] != "none": shim_dir = out / "bin" - shim_dir.mkdir(parents=True) + shim_dir.mkdir() shim = shim_dir / "wright" shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim "$@"\n') shim.chmod(0o755) - env.update(WRIGHT_BENCH_REAL=args.wright, WRIGHT_BENCH_TRACE=str(trace), PATH=f"{shim_dir}{os.pathsep}{baseline_path(env['PATH'])}") - else: - env["PATH"] = baseline_path(env["PATH"]) - start = time.monotonic() + env.update(WRIGHT_BENCH_REAL=args.wright, WRIGHT_BENCH_TRACE=str(out / "wright-trace.jsonl"), WRIGHT_BENCH_SIDECAR=str(out / "wright-calls")) + path = f"{shim_dir}{os.pathsep}{path}" + env.update( + PATH=path, + BENCH_HOST_PATH=os.environ["PATH"], BENCH_WORKSPACE=str(workspace), BENCH_RUN_DIR=str(out), BENCH_AGENT_ID=args.agent_id, + BENCH_USAGE=str(out / "usage.jsonl"), BENCH_TRANSCRIPT=str(out / "transcript.jsonl"), BENCH_CONTEXT=str(out / "context.json"), + BENCH_KNOWLEDGE=cell["knowledge"], BENCH_NETWORK=cell["network"], BENCH_WRIGHT=cell["wright"], + ) + if cell["wright"] == "bin+skill": + env["BENCH_SKILL_DIR"] = str(args.skill_dir) + return env + + +def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) -> str | None: + """A failed canary invalidates the run. Returns the reason, or None.""" + if cell["wright"] == "none" and shutil.which("wright", path=env["PATH"]): + return "wright reachable under wright level 'none'" + if cell["network"] == "off" and args.canary_cmd: + if subprocess.run(args.canary_cmd, shell=True, cwd=workspace, env=env, capture_output=True).returncode == 0: + return "network reachable under network 'off'" + return None + + +def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) -> tuple[int | None, str, str]: proc = subprocess.Popen( args.agent_cmd, shell=True, cwd=workspace, env=env, text=True, - stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, - start_new_session=True, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True, ) try: stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) - agent_exit = proc.returncode + return proc.returncode, stdout, stderr except subprocess.TimeoutExpired: - agent_exit = None try: os.killpg(proc.pid, signal.SIGKILL) except ProcessLookupError: pass stdout, stderr = proc.communicate() - stderr = f"{stderr}\ntimeout" if stderr else "timeout" - seconds = round(time.monotonic() - start, 1) + return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" + + +def context_report(out: Path, cell: dict) -> dict: + path = out / "context.json" + if not path.is_file(): + return {"reported": False} + loaded = json.loads(path.read_text()).get("loaded", []) + allowed = {"wright"} if cell["wright"] == "bin+skill" else set() + return {"reported": True, "loaded": loaded, "unexpected": sorted(set(loaded) - allowed)} + + +def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) -> dict: + check_cell(cell, args) + shutil.rmtree(out, ignore_errors=True) + out.mkdir(parents=True) + workspace = out / "workspace" + prompt = (scenario["dir"] / "prompt.md").read_text() + infra_retries = 0 + while True: + shutil.rmtree(workspace, ignore_errors=True) + for stale in ("home", "bin", "wright-trace.jsonl", "wright-calls", "usage.jsonl", "transcript.jsonl", "context.json", "snapshots"): + target = out / stale + shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) + materialize(scenario, workspace) + if cell["knowledge"] == "wiki": + (workspace / "wiki").symlink_to(args.wiki_dir.resolve()) + env = build_env(cell, args, out, workspace) + reason = canaries(cell, env, workspace, args) + if reason: + result = base_result(scenario, cell, args, out, 0.0, None) + result.update(invalid=reason) + (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + return result + snapshots = bench_trace.Snapshots(workspace, scenario.get("watch", [scenario["entry"]]), out / "snapshots") + snapshots.start() + start = time.monotonic() + agent_exit, stdout, stderr = run_agent(args, env, workspace, prompt) + seconds = round(time.monotonic() - start, 1) + snaps = snapshots.finish() + if agent_exit == INFRA_EXIT and infra_retries < args.infra_retries: + infra_retries += 1 + continue + break (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") - wright_version = subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip() - result = { + result = base_result(scenario, cell, args, out, seconds, agent_exit) + result["infraRetries"] = infra_retries + context = context_report(out, cell) + result["context"] = context + if context.get("unexpected"): + result["invalid"] = f"unexpected loaded context: {context['unexpected']}" + events = bench_trace.read_events(out / "wright-trace.jsonl") + result["wrightUse"] = bench_trace.summarize_trace(events) + result.update(bench_grade.grade(scenario, workspace, args.wright, out / "grading")) + entry = workspace / scenario["entry"] + final_sha = hashlib.sha256(entry.read_bytes()).hexdigest() if entry.is_file() else None + result["friction"] = bench_trace.friction(events) + result["expectations"] = bench_trace.detect_expectations(events, snaps, scenario, final_sha) + result["snapshots"] = snapshot_validity(scenario, snaps, args.wright, out) + first_valid = next((s["t"] for s in result["snapshots"]["series"] if s["valid"]), None) + result["usage"] = bench_trace.usage_summary(out / "usage.jsonl", first_valid) + (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + return result + + +def snapshot_validity(scenario: dict, snaps: list[dict], wright: str, out: Path) -> dict: + """Strict validity of every snapshot of the entry file; regressions are valid -> invalid transitions.""" + series = [] + for snap in [s for s in snaps if s["file"] == scenario["entry"]]: + valid = bench_grade.strict_valid(wright, Path(snap["path"]), out / "grading" / f"snapshot-{snap['i']:03d}") + series.append({"i": snap["i"], "t": snap["t"], "valid": valid}) + regressions = sum(1 for a, b in zip(series, series[1:]) if a["valid"] and not b["valid"]) + return {"count": len(series), "firstValidIndex": next((s["i"] for s in series if s["valid"]), None), "regressions": regressions, "series": series} + + +def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, seconds: float, agent_exit: int | None) -> dict: + return { "contract": RESULT_CONTRACT, "scenario": scenario["id"], "family": scenario["family"], "language": scenario["language"], - "condition": condition, + "split": scenario.get("split"), + "condition": dict(cell), "agent": {"id": args.agent_id, "command": args.agent_cmd, "exit": agent_exit, "seconds": seconds}, - "environment": {"os": platform.platform(), "python": platform.python_version(), "wright": wright_version, "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds")}, - "wrightUse": summarize_trace(trace), - **grade(scenario, workspace, args.wright), + "environment": { + "os": platform.platform(), "python": platform.python_version(), + "wright": subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip(), + "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), + }, } - (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") - return result + + +def trial_dir(base: Path, scenario_id: str, agent_id: str, cell: dict, trial: int) -> Path: + return base / scenario_id / agent_id / f"{cell_label(cell).replace('/', '_')}-{trial}" def cmd_run(args: argparse.Namespace) -> int: scenario = load_scenario(args.scenario) + cell = {"wright": args.wright_level, "knowledge": args.knowledge, "network": args.network} ok = True - for condition in args.conditions: - for trial in range(1, args.trials + 1): - result = run_trial(scenario, condition, args, args.out / args.scenario / f"{condition}-{trial}") - ok &= result["passed"] - print(f"{args.scenario} {condition} trial {trial}: {'PASS' if result['passed'] else 'FAIL'} layers={result['failedLayers']} wright-invocations={result['wrightUse']['invocations']}") + for trial in range(1, args.trials + 1): + result = run_trial(scenario, cell, args, trial_dir(args.out, args.scenario, args.agent_id, cell, trial)) + ok &= bool(result.get("passed")) and "invalid" not in result + print(f"{args.scenario} {cell_label(cell)} trial {trial}: {'INVALID ' + result['invalid'] if 'invalid' in result else 'PASS' if result['passed'] else 'FAIL'}" + f" layers={result.get('failedLayers')} wright-invocations={result.get('wrightUse', {}).get('invocations')}") return 0 if ok else 1 +def cmd_matrix(args: argparse.Namespace) -> int: + """Run every (scenario, agent, cell, trial) of a matrix file in randomized order; finished runs are skipped.""" + config = json.loads(args.config.read_text()) + jobs = [ + (s, agent, cell, t) + for s in config.get("scenarios") or all_scenario_ids() + for agent in config["agents"] + for cell in config["cells"] + for t in range(1, config.get("trials", 5) + 1) + ] + random.Random(config.get("seed", 0)).shuffle(jobs) + + def work(job: tuple) -> None: + scenario_id, agent, cell, trial = job + out = trial_dir(args.out, scenario_id, agent["id"], cell, trial) + if (out / "result.json").is_file(): + return + options = {k: Path(v) if k in ("skill_dir", "wiki_dir") and v else v for k, v in config.get("options", {}).items()} + trial_args = argparse.Namespace(**{**vars(args), "agent_id": agent["id"], "agent_cmd": agent["cmd"], **options}) + result = run_trial(load_scenario(scenario_id), cell, trial_args, out) + print(f"{scenario_id} {agent['id']} {cell_label(cell)} #{trial}: {'INVALID' if 'invalid' in result else 'PASS' if result['passed'] else 'FAIL'}", flush=True) + + with ThreadPoolExecutor(max_workers=config.get("parallel", 2)) as pool: + list(pool.map(work, jobs)) + return 0 + + +def cmd_setup_oracle(_: argparse.Namespace) -> int: + return subprocess.call(["npm", "ci", "--silent"], cwd=bench_grade.ORACLE) + + def main() -> int: if len(sys.argv) > 1 and sys.argv[1] == "shim": - return shim_main(sys.argv[2:]) + return bench_trace.shim_main(sys.argv[2:]) parser = argparse.ArgumentParser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) - for name in ("validate", "run"): + for name in ("validate", "run", "matrix"): p = sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") p.add_argument("--out", type=Path, default=ROOT / "target/agent-bench") - sub.choices["run"].add_argument("scenario", choices=all_scenario_ids()) - sub.choices["run"].add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace") - sub.choices["run"].add_argument("--agent-id", required=True, help="recorded agent/model/version label") - sub.choices["run"].add_argument("--conditions", nargs="+", choices=("baseline", "wright"), default=["baseline", "wright"]) - sub.choices["run"].add_argument("--trials", type=int, default=1) - sub.choices["run"].add_argument("--timeout", type=int, default=1800) + for name in ("run", "matrix"): + p = sub.choices[name] + p.add_argument("--skill-dir", type=Path, help="pinned guide directory, exposed to the adapter as BENCH_SKILL_DIR") + p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked read-only as ./wiki for knowledge 'wiki'") + p.add_argument("--env-pass", nargs="*", default=[], help="host variables passed through the environment scrub") + p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") + p.add_argument("--timeout", type=int, default=1800) + p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") + run = sub.choices["run"] + run.add_argument("scenario", choices=all_scenario_ids()) + run.add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace, BENCH_* describes the condition") + run.add_argument("--agent-id", required=True, help="recorded agent/model/version label") + run.add_argument("--wright-level", choices=WRIGHT_LEVELS, default="bin") + run.add_argument("--knowledge", choices=KNOWLEDGE_LEVELS, default="none") + run.add_argument("--network", choices=("off", "on"), default="off") + run.add_argument("--trials", type=int, default=1) + sub.choices["matrix"].add_argument("config", type=Path, help="JSON: agents[{id,cmd}], cells[{wright,knowledge,network}], scenarios, trials, parallel, seed, options") + sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") + report = sub.add_parser("report", help="summarize result.json files") + report.add_argument("dirs", nargs="+", type=Path) + report.add_argument("--regrade", action="store_true", help="re-grade stored workspaces twice and flag unstable graders") + report.add_argument("--wright", default=str(ROOT / "target/debug/wright")) args = parser.parse_args() - args.wright = str(Path(args.wright).resolve()) + if hasattr(args, "wright"): + args.wright = str(Path(args.wright).resolve()) + if hasattr(args, "out"): + args.out = args.out.resolve() if args.command == "validate": return 0 if validate(args.wright, args.out) else 1 - return cmd_run(args) + if args.command == "setup-oracle": + return cmd_setup_oracle(args) + if args.command == "report": + return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s)) + return cmd_run(args) if args.command == "run" else cmd_matrix(args) if __name__ == "__main__": diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py new file mode 100644 index 00000000..ec9a54a9 --- /dev/null +++ b/benchmarks/agent/bench_grade.py @@ -0,0 +1,208 @@ +"""Grading for the agent benchmark (#414, SPEC-414): check kinds, validity authorities, disagreement.""" + +from __future__ import annotations + +import hashlib +import json +import re +import shutil +import subprocess +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ORACLE = HERE / "oracle" +GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json") +UNSAFE_IGNORED = ("wiki",) + + +def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: + proc = subprocess.run([wright, *args, "--format", "json"], capture_output=True, text=True) + try: + return proc.returncode, json.loads(proc.stdout) + except json.JSONDecodeError: + return proc.returncode, {"diagnostics": [{"code": "harness-output", "severity": "error", "message": proc.stderr.strip() or proc.stdout.strip()}]} + + +def serve_request(wright: str, entry: Path, request: dict) -> dict: + proc = subprocess.run([wright, "serve", str(entry)], input=json.dumps(request) + "\n", capture_output=True, text=True) + try: + return json.loads(proc.stdout.splitlines()[0]) + except (IndexError, json.JSONDecodeError): + return {"error": {"code": "harness-output", "message": proc.stderr.strip()}} + + +def oracle_available() -> bool: + return bool(shutil.which("node")) and (ORACLE / "node_modules/overpy/package.json").is_file() + + +def oracle_version() -> str | None: + if not oracle_available(): + return None + return json.loads((ORACLE / "node_modules/overpy/package.json").read_text())["version"] + + +def oracle_compile(source: Path, out: Path) -> dict: + """Compile with the pinned upstream compiler; `unavailable` is explicit, never a pass.""" + if not oracle_available(): + return {"status": "unavailable"} + proc = subprocess.run(["node", str(ORACLE / "compile.js"), str(source), str(out)], capture_output=True, text=True) + try: + reply = json.loads(proc.stdout.strip().splitlines()[-1]) + except (IndexError, json.JSONDecodeError): + return {"status": "error", "error": (proc.stderr or proc.stdout).strip()[:300]} + return {"status": "ok"} if reply.get("ok") else {"status": "error", "error": reply.get("error")} + + +def wright_compile(wright: str, source: Path, out: Path) -> dict: + proc = subprocess.run([wright, "compile", str(source), "-o", str(out)], capture_output=True, text=True) + if proc.returncode == 0: + return {"status": "ok"} + return {"status": "error", "error": proc.stderr.strip()[:300]} + + +def authorities(wright: str, source: Path, scratch: Path) -> dict: + """Validity per authority, kept separate. Disagreement between upstream and Wright is an output.""" + scratch.mkdir(parents=True, exist_ok=True) + wright_out = scratch / "wright.txt" + result: dict = { + "wrightCheck": subprocess.run([wright, "check", str(source)], capture_output=True).returncode == 0, + "wrightCompile": wright_compile(wright, source, wright_out), + } + if result["wrightCompile"]["status"] == "ok": + result["workshopCheck"] = subprocess.run([wright, "check", str(wright_out)], capture_output=True).returncode == 0 + if source.suffix == ".opy": + result["oracle"] = oracle_compile(source, scratch / "oracle.txt") + known = {"ok", "error"} + theirs, ours = result["oracle"]["status"], result["wrightCompile"]["status"] + if theirs in known and ours in known and theirs != ours: + result["disagreement"] = { + "kind": "wright-accepts-oracle-rejects" if ours == "ok" else "oracle-accepts-wright-rejects", + "source": source.name, + "wright": result["wrightCompile"], + "oracle": result["oracle"], + } + return result + + +def compiled_text(state: dict, source: str) -> str | None: + """Compiled Workshop text from `wright` or the upstream `oracle`, when that authority succeeded.""" + auth = state["authorities"] + if source == "oracle": + return (state["scratch"] / "oracle.txt").read_text() if auth.get("oracle", {}).get("status") == "ok" else None + return (state["scratch"] / "wright.txt").read_text() if auth["wrightCompile"]["status"] == "ok" else None + + +def check_result(check: dict, passed: bool, detail: str, **extra) -> dict: + return {"id": check["id"], "kind": check["kind"], "layer": check.get("layer", "agent"), "passed": passed, "detail": detail, **extra} + + +def matches(text: str, check: dict) -> bool: + flags = re.I if check.get("ignoreCase") else 0 + return all(re.search(p, text, flags) for p in check.get("all", [])) and ( + not check.get("any") or any(re.search(p, text, flags) for p in check["any"]) + ) + + +def run_check(check: dict, workspace: Path, entry: Path, wright: str, state: dict) -> dict: + kind = check["kind"] + if kind == "check": + code, envelope = wright_json(wright, ["check", str(entry)]) + state["diagnostics"] = envelope.get("diagnostics", []) + errors = [d for d in state["diagnostics"] if d.get("severity") == "error"] + return check_result(check, code == 0 and not errors, f"exit {code}, {len(errors)} error diagnostic(s)") + if kind == "lint": + findings = [f for f in state["lint"] if f["code"] == check["code"]] + return check_result(check, len(findings) <= check.get("max", 0), f"{len(findings)} '{check['code']}' finding(s)") + if kind == "symbols": + response = serve_request(wright, entry, {"op": "symbols", "kind": check["symbolKind"]}) + found = len(response.get("result", [])) + return check_result(check, found >= check.get("min", 1), f"{found} '{check['symbolKind']}' symbol(s)") + if kind in ("contains", "absent"): + path = workspace / check["file"] + text = path.read_text() if path.is_file() else "" + texts = check["text"] if isinstance(check["text"], list) else [check["text"]] + count = sum(text.count(t) for t in texts) + if kind == "contains": + passed = count >= check.get("min", 1) and count <= check.get("max", count) + else: + passed = count == 0 + return check_result(check, passed, f"{count} occurrence(s) in {check['file']}") + if kind == "answer": + path = workspace / "answer.json" + try: + actual = json.loads(path.read_text()).get(check["key"]) + except (OSError, json.JSONDecodeError, AttributeError): + actual = None + return check_result(check, actual == check["expected"], f"answer[{check['key']}] = {json.dumps(actual)}") + if kind == "oracle": + status = state["authorities"].get("oracle", {"status": "unavailable"}) + detail = status.get("error") or status["status"] + if status["status"] == "unavailable": + detail = "upstream oracle not installed; run `agent_bench.py setup-oracle`" + return check_result(check, status["status"] == "ok", detail, unavailable=status["status"] == "unavailable") + if kind == "wright-compile": + status = state["authorities"]["wrightCompile"] + return check_result(check, status["status"] == "ok", status.get("error") or status["status"]) + if kind in ("compiled-contains", "compiled-absent"): + source = check.get("source", "oracle" if entry.suffix == ".opy" else "wright") + text = compiled_text(state, source) + if text is None: + return check_result(check, False, f"no compiled text from {source}", unavailable=source == "oracle" and not oracle_available()) + found = matches(text, check) + return check_result(check, found if kind == "compiled-contains" else not found, f"{'matched' if found else 'no match'} in {source} output") + raise SystemExit(f"unknown check kind '{kind}'") + + +def tree(root: Path) -> dict[str, bytes]: + return {str(p.relative_to(root)): p.read_bytes() for p in sorted(root.rglob("*")) if p.is_file()} + + +def unsafe_edits(scenario: dict, workspace: Path) -> list[str]: + seed = tree(scenario["dir"] / "seed") + now = tree(workspace) + writable = set(scenario["writable"]) + return sorted( + name for name in seed.keys() | now.keys() + if seed.get(name) != now.get(name) and name not in writable and name.split("/")[0] not in UNSAFE_IGNORED + ) + + +def grader_hash(scenario: dict) -> str: + digest = hashlib.sha256() + for path in [scenario["dir"] / "scenario.json", *(HERE / f for f in GRADER_FILES)]: + digest.update(path.name.encode() + (path.read_bytes() if path.is_file() else b"")) + return digest.hexdigest() + + +def grade(scenario: dict, workspace: Path, wright: str, scratch: Path | None = None) -> dict: + entry = workspace / scenario["entry"] + scratch = scratch or workspace.parent / f"{workspace.name}-grading" + shutil.rmtree(scratch, ignore_errors=True) + state: dict = {"scratch": scratch, "authorities": authorities(wright, entry, scratch) if entry.is_file() else {"wrightCompile": {"status": "error"}}} + _, lint = wright_json(wright, ["lint", str(entry)]) + state["lint"] = (lint.get("result") or {}).get("findings") or [] + checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] + lint_errors = sum(1 for f in state["lint"] if f.get("severity") == "error") + passed = all(c["passed"] for c in checks) + auth = state["authorities"] + return { + "checks": checks, + "passed": passed, + "usable": passed and lint_errors == 0, + "failedLayers": sorted({c["layer"] for c in checks if not c["passed"]}), + "diagnostics": state.get("diagnostics", []), + "authorities": {k: v for k, v in auth.items() if k != "disagreement"}, + "disagreement": auth.get("disagreement"), + "lintFindings": sorted({f["code"] for f in state["lint"]}), + "unsafeEdits": unsafe_edits(scenario, workspace), + "unverifiedRuntimeClaims": scenario.get("runtimeOnly", []), + "grader": {"hash": grader_hash(scenario), "oracle": oracle_version()}, + } + + +def strict_valid(wright: str, source: Path, scratch: Path) -> bool: + """Validity of one source snapshot: the upstream oracle when it applies, else Wright.""" + auth = authorities(wright, source, scratch) + if source.suffix == ".opy" and auth["oracle"]["status"] != "unavailable": + return auth["oracle"]["status"] == "ok" + return auth["wrightCompile"]["status"] == "ok" diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py new file mode 100644 index 00000000..b5f1e795 --- /dev/null +++ b/benchmarks/agent/bench_report.py @@ -0,0 +1,209 @@ +"""Aggregate agent benchmark results (#414, SPEC-414): intervals, paired comparison, efficiency, diagnostics.""" + +from __future__ import annotations + +import json +import math +import re +from collections import defaultdict +from pathlib import Path +from statistics import mean, pstdev + +BASELINE = "none/none/off" +HEADROOM = 0.95 + + +def wilson(k: int, n: int, z: float = 1.96) -> tuple[float, float]: + if n == 0: + return (0.0, 0.0) + p = k / n + denom = 1 + z * z / n + centre = (p + z * z / (2 * n)) / denom + half = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / denom + return (max(0.0, centre - half), min(1.0, centre + half)) + + +def label(result: dict) -> str: + c = result["condition"] + return f"{c['wright']}/{c['knowledge']}/{c['network']}" + + +def load(dirs: list[Path]) -> list[dict]: + results = [] + for base in dirs: + for path in sorted(base.rglob("result.json")): + result = json.loads(path.read_text()) + if str(result.get("contract", "")).startswith("wright-agent-bench/"): + result["_dir"] = path.parent + match = re.search(r"-(\d+)$", path.parent.name) + result["_trial"] = int(match.group(1)) if match else 0 + results.append(result) + return results + + +def rate(k: int, n: int) -> str: + lo, hi = wilson(k, n) + return f"{k}/{n} [{lo:.2f}-{hi:.2f}]" + + +def total_tokens(result: dict) -> int | None: + return (result.get("usage") or {}).get("totalTokens") + + +def group_rows(runs: list[dict]) -> dict: + usable = [r for r in runs if r.get("usable")] + tokens = [t for t in (total_tokens(r) for r in runs) if t is not None] + peaks = [p for p in ((r.get("usage") or {}).get("peakContext") for r in runs) if p] + return { + "n": len(runs), "usable": len(usable), "passed": sum(1 for r in runs if r.get("passed")), + "tokens": mean(tokens) if tokens else None, + "tokensPerUsable": (sum(tokens) / len(usable)) if tokens and usable and len(tokens) == len(runs) else None, + "peakContext": mean(peaks) if peaks else None, + "seconds": mean(r["agent"]["seconds"] for r in runs), + "usedWright": sum(1 for r in runs if r.get("wrightUse", {}).get("invocations")), + } + + +def fmt(value, digits: int = 0) -> str: + return "n/a" if value is None else f"{value:,.{digits}f}" + + +def paired(runs: list[dict]) -> list[str]: + by_key: dict[tuple, dict] = {(r["scenario"], r["agent"]["id"], r["_trial"], label(r)): r for r in runs} + lines = [] + cells = sorted({label(r) for r in runs} - {BASELINE}) + for agent in sorted({r["agent"]["id"] for r in runs}): + for cell in cells: + pairs = [(by_key[(s, a, t, BASELINE)], r) for (s, a, t, c), r in by_key.items() if a == agent and c == cell and (s, a, t, BASELINE) in by_key] + if not pairs: + continue + gain = sum(1 for b, r in pairs if r.get("usable") and not b.get("usable")) + loss = sum(1 for b, r in pairs if b.get("usable") and not r.get("usable")) + both = [(total_tokens(b), total_tokens(r)) for b, r in pairs if b.get("usable") and r.get("usable") and total_tokens(b) and total_tokens(r)] + saving = f"{mean(1 - r / b for b, r in both):+.0%} tokens (n={len(both)})" if both else "no both-usable pairs" + lines.append(f"| {agent} | {cell} vs {BASELINE} | {len(pairs)} | +{gain} / -{loss} | {saving} |") + return lines + + +def expectations(runs: list[dict]) -> list[str]: + lines = [] + for cell in sorted({label(r) for r in runs}): + cell_runs = [r for r in runs if label(r) == cell and r.get("expectations")] + if not cell_runs: + continue + cells = [] + for eid in sorted({e for r in cell_runs for e in r["expectations"]}): + statuses = [r["expectations"][eid]["status"] for r in cell_runs if eid in r["expectations"]] + ok, bad = statuses.count("pass"), statuses.count("fail") + cells.append(f"{eid} {ok}/{ok + bad}" if ok + bad else f"{eid} -") + lines.append(f"| {cell} | {' ยท '.join(cells)} |") + return lines + + +def diagnostics(runs: list[dict], invalid: list[dict]) -> list[str]: + notes = [] + base = [r for r in runs if label(r) == BASELINE] + if base and sum(1 for r in base if r.get("usable")) / len(base) >= HEADROOM: + notes.append(f"HEADROOM: baseline `{BASELINE}` usable rate is at least {HEADROOM:.0%}; the tasks cannot show a gain.") + infra = [r for r in runs if r["agent"]["exit"] != 0 or r.get("infraRetries")] + if infra: + notes.append(f"INFRASTRUCTURE: {len(infra)} run(s) exited non-zero, timed out, or needed an infrastructure retry.") + if invalid: + notes.append(f"INVALID: {len(invalid)} run(s) excluded ({', '.join(sorted({r['invalid'] for r in invalid}))}).") + groups: dict[tuple, list[dict]] = defaultdict(list) + for r in runs: + groups[(r["scenario"], r["agent"]["id"], label(r))].append(r) + noisy = [k for k, g in groups.items() if len(g) >= 3 and 0 < sum(1 for r in g if r.get("usable")) < len(g)] + if noisy: + notes.append(f"VARIANCE: {len(noisy)} scenario/agent/cell group(s) mix usable and unusable trials; compare with the smallest effect that matters before hillclimbing.") + return notes + + +def regrade_notes(runs: list[dict], wright: str, load_scenario) -> list[str]: + """Grader consistency: grading the same stored workspace twice must give one verdict.""" + import bench_grade + unstable = [] + for r in runs: + workspace = r["_dir"] / "workspace" + if not workspace.is_dir(): + continue + scenario = load_scenario(r["scenario"]) + first, second = (bench_grade.grade(scenario, workspace, wright, r["_dir"] / f"regrade-{i}") for i in (1, 2)) + if [(c["id"], c["passed"]) for c in first["checks"]] != [(c["id"], c["passed"]) for c in second["checks"]]: + unstable.append(str(r["_dir"])) + return [f"GRADER: unstable verdict on {len(unstable)} workspace(s): {unstable}"] if unstable else ["GRADER: consistent on every regraded workspace."] + + +def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, dict]: + invalid = [r for r in results if "invalid" in r] + runs = [r for r in results if "invalid" not in r] + out = ["# Agent benchmark report", "", f"{len(runs)} valid run(s), {len(invalid)} invalid.", ""] + summary: dict = {"cells": {}} + out += ["## Outcome by agent and condition", "", "| agent | condition | runs | usable | passed | used wright | tokens/run | tokens per usable | peak context | s/run |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] + for agent in sorted({r["agent"]["id"] for r in runs}): + for cell in sorted({label(r) for r in runs}): + group = [r for r in runs if r["agent"]["id"] == agent and label(r) == cell] + if not group: + continue + row = group_rows(group) + summary["cells"][f"{agent}|{cell}"] = row + out.append(f"| {agent} | {cell} | {row['n']} | {rate(row['usable'], row['n'])} | {row['passed']}/{row['n']} | {row['usedWright']}/{row['n']} | " + f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} |") + out += ["", "## By scenario", "", "| scenario | agent | condition | usable |", "| --- | --- | --- | --- |"] + groups: dict[tuple, list[dict]] = defaultdict(list) + for r in runs: + groups[(r["scenario"], r["agent"]["id"], label(r))].append(r) + for (scenario, agent, cell), g in sorted(groups.items()): + out.append(f"| {scenario} | {agent} | {cell} | {rate(sum(1 for r in g if r.get('usable')), len(g))} |") + splits = sorted({r["split"] for r in runs if r.get("split")}) + if splits: + out += ["", "## By split", "", "| split | condition | usable |", "| --- | --- | --- |"] + for split in splits: + for cell in sorted({label(r) for r in runs}): + g = [r for r in runs if r.get("split") == split and label(r) == cell] + if g: + out.append(f"| {split} | {cell} | {rate(sum(1 for r in g if r.get('usable')), len(g))} |") + pairs = paired(runs) + if pairs: + out += ["", f"## Paired against `{BASELINE}` (same scenario, agent, trial)", "", "| agent | comparison | pairs | usable gained/lost | tokens where both usable |", "| --- | --- | --- | --- | --- |", *pairs] + exp = expectations(runs) + if exp: + out += ["", "## Expectation rates (pass/(pass+fail); n/a and unavailable excluded)", "", "| condition | expectations |", "| --- | --- |", *exp] + disagreements = [(r["scenario"], r["disagreement"], str(r["_dir"])) for r in runs if r.get("disagreement")] + if disagreements: + out += ["", "## Authority disagreements (candidate owner issues)", ""] + out += [f"- `{s}` {d['kind']}: wright `{d['wright'].get('error') or d['wright']['status']}` / oracle `{d['oracle'].get('error') or d['oracle']['status']}` ({path})" for s, d, path in disagreements] + fr = defaultdict(lambda: defaultdict(int)) + for r in runs: + for k, v in (r.get("friction") or {}).items(): + if isinstance(v, int) and k != "callsToFirstSuccess": + fr[label(r)][k] += v + if fr: + keys = sorted({k for v in fr.values() for k in v}) + out += ["", "## Friction (sum over runs)", "", f"| condition | {' | '.join(keys)} |", f"| --- | {' | '.join('---' for _ in keys)} |"] + out += [f"| {cell} | {' | '.join(str(v.get(k, 0)) for k in keys)} |" for cell, v in sorted(fr.items())] + tok: dict[str, dict[str, list[int]]] = defaultdict(lambda: defaultdict(list)) + for r in runs: + for cmd, t in (r.get("wrightUse", {}).get("outputTokensEstimate") or {}).items(): + tok[cmd]["tokens"].append(t) + if tok: + out += ["", "## Wright output size per command (estimated tokens per run that used it)", "", "| command | runs | mean | max |", "| --- | --- | --- | --- |"] + out += [f"| {cmd} | {len(v['tokens'])} | {mean(v['tokens']):.0f} | {max(v['tokens'])} |" for cmd, v in sorted(tok.items())] + notes = diagnostics(runs, invalid) + (regrade or []) + out += ["", "## Diagnostics", ""] + ([f"- {n}" for n in notes] or ["- none"]) + summary["diagnostics"] = notes + summary["stdev"] = {k: pstdev([r["agent"]["seconds"] for r in runs if f"{r['agent']['id']}|{label(r)}" == k]) for k in summary["cells"]} if runs else {} + return "\n".join(out) + "\n", summary + + +def main(dirs: list[Path], wright: str, regrade: bool, load_scenario) -> int: + results = load(dirs) + if not results: + print("no results found") + return 1 + notes = regrade_notes([r for r in results if "invalid" not in r], wright, load_scenario) if regrade else None + text, summary = render(results, notes) + (dirs[0] / "report.md").write_text(text) + (dirs[0] / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") + print(text) + return 0 diff --git a/benchmarks/agent/bench_trace.py b/benchmarks/agent/bench_trace.py new file mode 100644 index 00000000..3ffe417b --- /dev/null +++ b/benchmarks/agent/bench_trace.py @@ -0,0 +1,278 @@ +"""Trace capture and analysis for the agent benchmark (#414, SPEC-414): shim, snapshots, expectations, usage.""" + +from __future__ import annotations + +import hashlib +import json +import os +import subprocess +import sys +import threading +import time +from pathlib import Path + +TOKEN_BYTES = 4 # estimation only: bytes per token for Wright output attribution +DECISION_COMMANDS = ("check", "lint", "analyze", "inspect") +VALIDATING = ("check", "lint", "analyze", "compile") +OUTPUT_FORMAT_FLAGS = ("--format", "-f") + + +def command_of(argv: list[str]) -> str: + return next((a for a in argv if not a.startswith("-")), "") + + +def wants_json(argv: list[str]) -> bool: + return any(a in OUTPUT_FORMAT_FLAGS and i + 1 < len(argv) and argv[i + 1] == "json" for i, a in enumerate(argv)) or "--format=json" in argv + + +def envelope_summary(envelope: dict) -> dict: + result = envelope.get("result") or {} + output = result.get("output") if isinstance(result.get("output"), dict) else {} + return { + "command": envelope.get("command"), + "ok": envelope.get("ok"), + "exit": envelope.get("exit"), + "codes": [d.get("code") for d in envelope.get("diagnostics", [])], + "inputIdentity": result.get("input_identity") or output.get("input_identity"), + "selection": result.get("selection") or envelope.get("selection"), + } + + +def append_event(event: dict) -> None: + with open(os.environ["WRIGHT_BENCH_TRACE"], "a") as trace: + trace.write(json.dumps(event) + "\n") + + +def sidecar_name() -> str: + return f"{time.time_ns()}-{os.getpid()}" + + +def shim_main(argv: list[str]) -> int: + """Run the real Wright and record the call. `serve` sessions are teed line by line.""" + real = os.environ["WRIGHT_BENCH_REAL"] + started = time.time() + if command_of(argv) == "serve": + return serve_tee(real, argv, started) + proc = subprocess.run([real, *argv], capture_output=True) + sys.stdout.buffer.write(proc.stdout) + sys.stdout.flush() + sys.stderr.buffer.write(proc.stderr) + sys.stderr.flush() + name = sidecar_name() + sidecar = Path(os.environ["WRIGHT_BENCH_SIDECAR"]) + sidecar.mkdir(parents=True, exist_ok=True) + (sidecar / f"{name}.out").write_bytes(proc.stdout) + (sidecar / f"{name}.err").write_bytes(proc.stderr) + envelope = None + if wants_json(argv): + try: + envelope = envelope_summary(json.loads(proc.stdout)) + except json.JSONDecodeError: + envelope = None + append_event({ + "type": "call", "t": started, "argv": argv, "cwd": os.getcwd(), "exit": proc.returncode, + "seconds": round(time.time() - started, 3), "stdoutBytes": len(proc.stdout), "stderrBytes": len(proc.stderr), + "stderrHead": proc.stderr.decode(errors="replace")[:300], "envelope": envelope, "sidecar": name, + }) + return proc.returncode + + +def serve_tee(real: str, argv: list[str], started: float) -> int: + proc = subprocess.Popen([real, *argv], stdin=subprocess.PIPE, stdout=subprocess.PIPE) + counts = {"req": 0, "res": 0} + + def log(direction: str, line: bytes) -> None: + counts[direction] += 1 + append_event({"type": "serve", "dir": direction, "t": time.time(), "line": line.decode(errors="replace")[:2000]}) + + def pump() -> None: + for line in proc.stdout: + sys.stdout.buffer.write(line) + sys.stdout.flush() + log("res", line) + + reader = threading.Thread(target=pump) + reader.start() + for line in sys.stdin.buffer: + proc.stdin.write(line) + proc.stdin.flush() + log("req", line) + proc.stdin.close() + code = proc.wait() + reader.join() + append_event({"type": "call", "t": started, "argv": argv, "cwd": os.getcwd(), "exit": code, "seconds": round(time.time() - started, 3), + "requests": counts["req"], "envelope": None}) + return code + + +def read_events(path: Path) -> list[dict]: + return [json.loads(line) for line in path.read_text().splitlines()] if path.is_file() else [] + + +class Snapshots(threading.Thread): + """Poll watched files during a run; keep a copy of every distinct content, with its time.""" + + def __init__(self, workspace: Path, files: list[str], out: Path, interval: float = 0.2): + super().__init__(daemon=True) + self.workspace, self.files, self.out, self.interval = workspace, files, out, interval + self.events: list[dict] = [] + self.stop_event = threading.Event() + self.last = {f: self.digest(f) for f in files} + out.mkdir(parents=True, exist_ok=True) + + def digest(self, name: str) -> str | None: + path = self.workspace / name + return hashlib.sha256(path.read_bytes()).hexdigest() if path.is_file() else None + + def scan(self) -> None: + for name in self.files: + digest = self.digest(name) + if digest is not None and digest != self.last[name]: + self.last[name] = digest + index = len(self.events) + 1 + target = self.out / f"{index:03d}-{Path(name).name}" + target.write_bytes((self.workspace / name).read_bytes()) + self.events.append({"i": index, "t": time.time(), "file": name, "sha256": digest, "path": str(target)}) + + def run(self) -> None: + while not self.stop_event.wait(self.interval): + self.scan() + + def finish(self) -> list[dict]: + self.stop_event.set() + self.join() + self.scan() + (self.out / "snapshots.json").write_text(json.dumps(self.events, indent=2) + "\n") + return self.events + + +def summarize_trace(events: list[dict]) -> dict: + calls = [e for e in events if e["type"] == "call"] + by_command: dict[str, int] = {} + for call in calls: + by_command[command_of(call["argv"])] = by_command.get(command_of(call["argv"]), 0) + 1 + return { + "invocations": len(calls), + "byCommand": by_command, + "failedInvocations": sum(1 for c in calls if c["exit"] != 0), + "ownerOrEnvironmentGaps": [c["argv"] for c in calls if c["exit"] >= 3], + "outputTokensEstimate": { + cmd: sum(c.get("stdoutBytes", 0) + c.get("stderrBytes", 0) for c in calls if command_of(c["argv"]) == cmd) // TOKEN_BYTES + for cmd in by_command if cmd + }, + } + + +def friction(events: list[dict]) -> dict: + calls = [e for e in events if e["type"] == "call"] + seen: list[tuple] = [] + repeats = 0 + for call in calls: + key = tuple(call["argv"]) + repeats += key in seen + seen.append(key) + serve_responses = [json.loads(e["line"]) for e in events if e["type"] == "serve" and e["dir"] == "res" and e["line"].startswith("{")] + return { + "usageErrors": sum(1 for c in calls if c["exit"] == 2), + "unknownSubcommands": sum(1 for c in calls if "unrecognized subcommand" in c.get("stderrHead", "")), + "helpLookups": sum(1 for c in calls if any(a in ("--help", "-h", "help") for a in c["argv"])), + "retriesAfterUnsupported": sum(1 for i, c in enumerate(calls) if c["exit"] >= 3 and tuple(c["argv"]) in [tuple(x["argv"]) for x in calls[i + 1:]]), + "malformedServeRequests": sum(1 for r in serve_responses if r.get("error", {}).get("code") == "malformed-request"), + "identicalRepeats": repeats, + "callsToFirstSuccess": next((i + 1 for i, c in enumerate(calls) if c["exit"] == 0 and not any(a in ("--help", "-h", "--version") for a in c["argv"])), None), + } + + +def serve_ops(events: list[dict]) -> list[str]: + ops = [] + for e in events: + if e["type"] == "serve" and e["dir"] == "req": + try: + ops.append(json.loads(e["line"]).get("op", "")) + except json.JSONDecodeError: + ops.append("") + return ops + + +def expectation(status: str, detail: str = "") -> dict: + return {"status": status, "detail": detail} + + +def detect_expectations(events: list[dict], snapshots: list[dict], scenario: dict, final_sha256: str | None) -> dict: + """SPEC-414 E01-E12 over the Wright trace. E05, E07, E09, E10 need the agent transcript: `unavailable`.""" + calls = [e for e in events if e["type"] == "call"] + ops = serve_ops(events) + used = bool(calls) + unavailable = expectation("unavailable", "needs the normalized agent transcript") + result = {k: unavailable for k in ("E05", "E07", "E09", "E10")} + if not used: + return {**result, **{k: expectation("na", "Wright not used") for k in ("E01", "E02", "E03", "E04", "E06", "E08", "E11", "E12")}} + first = calls[0] + discovery = any(a in ("--help", "-h", "help", "--version") for a in first["argv"]) or "capabilities" in ops[:1] + result["E01"] = expectation("pass" if discovery else "fail", f"first call: {' '.join(first['argv'])}") + decision = [c for c in calls if command_of(c["argv"]) in DECISION_COMMANDS] + structured = [c for c in decision if wants_json(c["argv"])] + if decision or ops: + rate = (len(structured) + len(ops)) / (len(decision) + len(ops)) + result["E02"] = expectation("pass" if rate >= 0.5 else "fail", f"structured share {rate:.2f}") + else: + result["E02"] = expectation("na", "no decision-driving calls") + if scenario.get("stabilityRisk"): + stability = any(command_of(c["argv"]) in ("lint", "analyze") for c in calls) or any(o in ("lint", "findings", "analyze") for o in ops) + result["E03"] = expectation("pass" if stability else "fail", "lint or analyze run" if stability else "only check-level validation") + else: + result["E03"] = expectation("na", "scenario has no stability risk") + last_edit = max((s["t"] for s in snapshots), default=None) + if last_edit is None: + result["E04"] = expectation("na", "no edits observed") + else: + after = [c for c in calls if command_of(c["argv"]) in VALIDATING and c["t"] + c["seconds"] >= last_edit] + matched = [c for c in after if c.get("envelope") and c["envelope"].get("inputIdentity") == final_sha256] + result["E04"] = expectation("pass" if after else "fail", f"{len(after)} validation(s) after last edit; {len(matched)} match the final content") + withheld = [c for c in calls if ((c.get("envelope") or {}).get("selection") or {}).get("withheld")] + flags = sum(1 for c in calls if any(a in ("--severity", "--rule-id", "--file", "--max") for a in c["argv"])) + result["E06"] = expectation("na" if not withheld and not flags else "info", f"{flags} selection-flag call(s), {len(withheld)} withheld result(s)") + unsupported = [c for c in calls if c["exit"] >= 3] + excess = [c for c in unsupported if sum(1 for x in calls if x["argv"] == c["argv"]) > 2] + result["E08"] = expectation("na" if not unsupported else ("fail" if excess else "pass"), f"{len(unsupported)} exit 3/4 call(s), {len(excess)} retried more than twice") + if ops: + malformed = friction(events)["malformedServeRequests"] + result["E11"] = expectation("pass" if ops[0] == "capabilities" and not malformed else "fail", f"first op '{ops[0]}', {malformed} malformed") + else: + result["E11"] = expectation("na", "no serve session") + edit_times = [s["t"] for s in snapshots] + wasted = 0 + for i, c in enumerate(calls): + for later in calls[i + 1:]: + if later["argv"] == c["argv"]: + wasted += not any(c["t"] <= t <= later["t"] for t in edit_times) + break + result["E12"] = expectation("pass" if wasted == 0 else "fail", f"{wasted} identical repeat(s) with no edit between") + return result + + +def usage_summary(path: Path, first_valid_t: float | None) -> dict | None: + """Per-turn usage rows written by the adapter: t, input, output, cache_read, cache_write, reasoning, context, context_limit.""" + if not path.is_file(): + return None + rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()] + if not rows: + return None + + def total(row: dict) -> int: + return sum(row.get(k) or 0 for k in ("input", "output", "cache_read", "cache_write", "reasoning")) + + peak = max(rows, key=lambda r: r.get("context") or 0) + limit = peak.get("context_limit") + to_first = None + if first_valid_t is not None: + upto = [r for r in rows if r.get("t") is not None and r["t"] <= first_valid_t] + to_first = {"turns": len(upto), "tokens": sum(total(r) for r in upto)} + return { + "turns": len(rows), + "tokens": {k: sum(r.get(k) or 0 for r in rows) for k in ("input", "output", "cache_read", "cache_write", "reasoning")}, + "totalTokens": sum(total(r) for r in rows), + "peakContext": peak.get("context"), + "peakContextShare": round(peak["context"] / limit, 4) if limit and peak.get("context") else None, + "toFirstValid": to_first, + } diff --git a/benchmarks/agent/oracle/compile.js b/benchmarks/agent/oracle/compile.js new file mode 100644 index 00000000..fea27ad3 --- /dev/null +++ b/benchmarks/agent/oracle/compile.js @@ -0,0 +1,18 @@ +// Compile one OverPy file with the pinned upstream compiler. +// usage: node compile.js ; prints one JSON line {ok, error?}. +const overpy = require("overpy"); +const fs = require("fs"); +const path = require("path"); + +(async () => { + await overpy.readyPromise; + const [source, out] = process.argv.slice(2); + try { + const result = await overpy.compile(fs.readFileSync(source, "utf8"), "en-US", path.dirname(path.resolve(source)), path.basename(source)); + fs.writeFileSync(out, result.result); + console.log(JSON.stringify({ ok: true })); + } catch (e) { + console.log(JSON.stringify({ ok: false, error: String((e && e.message) || e).split("\n")[0] })); + process.exit(1); + } +})(); diff --git a/benchmarks/agent/oracle/package-lock.json b/benchmarks/agent/oracle/package-lock.json new file mode 100644 index 00000000..38a7f1aa --- /dev/null +++ b/benchmarks/agent/oracle/package-lock.json @@ -0,0 +1,20 @@ +{ + "name": "oracle", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "dependencies": { + "overpy": "9.7.10" + } + }, + "node_modules/overpy": { + "version": "9.7.10", + "resolved": "https://registry.npmjs.org/overpy/-/overpy-9.7.10.tgz", + "integrity": "sha512-oX17nauJcPTaKIrRFY/rD0Rl8atqFUVv9Hg2TKH+A68/fC8+ZO344Mkd1A/Y0oOVp1hr5tktMBjzMEDDnMEYUw==", + "bin": { + "overpy": "cli.js" + } + } + } +} diff --git a/benchmarks/agent/oracle/package.json b/benchmarks/agent/oracle/package.json new file mode 100644 index 00000000..a28f426e --- /dev/null +++ b/benchmarks/agent/oracle/package.json @@ -0,0 +1,5 @@ +{ + "private": true, + "description": "Pinned upstream OverPy oracle for the agent benchmark (docs/agent-benchmark.md).", + "dependencies": { "overpy": "9.7.10" } +} diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json b/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json index 4d501935..f61ea11f 100644 --- a/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json +++ b/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json @@ -3,6 +3,7 @@ "family": "diagnosis", "language": "workshop", "entry": "mode.ws", + "stabilityRisk": true, "writable": ["mode.ws", "answer.json"], "runtimeOnly": ["The server no longer freezes in a live match."], "checks": [ diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 89bd91e9..c75bd3e4 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -1,44 +1,213 @@ +import argparse import json import os import shutil -import subprocess -import sys import tempfile import unittest from pathlib import Path import agent_bench +import bench_grade +import bench_report +import bench_trace WRIGHT = os.environ.get("WRIGHT_BIN", str(agent_bench.ROOT / "target/debug/wright")) +SCENARIO = "repair-runaway-loop" + + +def reference(scenario: str = SCENARIO) -> str: + return str(agent_bench.SCENARIOS / scenario / "reference") @unittest.skipUnless(Path(WRIGHT).is_file(), "build wright first or set WRIGHT_BIN") class AgentBenchTest(unittest.TestCase): def setUp(self): (agent_bench.ROOT / "target").mkdir(exist_ok=True) - self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")) + self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")).resolve() self.addCleanup(shutil.rmtree, self.out, True) + def trial(self, agent_cmd: str, wright: str = "bin", scenario: str = SCENARIO, **options) -> dict: + args = argparse.Namespace(**{ + "wright": str(Path(WRIGHT).resolve()), "agent_id": "fake", "agent_cmd": agent_cmd, "timeout": 60, "infra_retries": 2, + "env_pass": [], "canary_cmd": None, "skill_dir": None, "wiki_dir": None, **options, + }) + cell = {"wright": wright, "knowledge": "none", "network": "off"} + return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{wright}") + def test_scenarios_are_solvable_and_not_vacuous(self): - self.assertTrue(agent_bench.validate(WRIGHT, self.out)) - - def test_conditions_differ_only_in_wright_availability(self): - scenario = "repair-runaway-loop" - reference = agent_bench.SCENARIOS / scenario / "reference" - agent = f"cp {reference}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" - subprocess.run( - [sys.executable, agent_bench.__file__, "run", scenario, "--wright", WRIGHT, "--out", str(self.out), "--agent-id", "fake", "--agent-cmd", agent], - check=False, - capture_output=True, - ) - baseline = json.loads((self.out / scenario / "baseline-1/result.json").read_text()) - assisted = json.loads((self.out / scenario / "wright-1/result.json").read_text()) - self.assertEqual(baseline["wrightUse"]["invocations"], 0) - self.assertIn("missing-wright.txt", baseline["unsafeEdits"]) + self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) + + def test_levels_differ_only_in_wright_availability(self): + agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" + none = self.trial(agent, wright="none") + assisted = self.trial(agent, wright="bin") + self.assertEqual(none["wrightUse"]["invocations"], 0) + self.assertIn("missing-wright.txt", none["unsafeEdits"]) self.assertEqual(assisted["wrightUse"]["byCommand"], {"check": 1}) self.assertTrue(assisted["passed"]) self.assertEqual(assisted["unsafeEdits"], []) + def test_canary_rejects_reachable_wright_under_none(self): + cell = {"wright": "none", "knowledge": "none", "network": "off"} + args = argparse.Namespace(canary_cmd=None) + env = {"PATH": str(Path(WRIGHT).resolve().parent)} + self.assertIn("reachable", agent_bench.canaries(cell, env, self.out, args)) + self.assertIsNone(agent_bench.canaries({**cell, "wright": "bin"}, env, self.out, args)) + + def test_network_canary_invalidates_run(self): + result = self.trial("true", canary_cmd="true") + self.assertIn("network reachable", result["invalid"]) + self.assertIsNone(self.trial("true", canary_cmd="false").get("invalid")) + + def test_environment_is_scrubbed(self): + os.environ["BENCH_LEAK_PROBE"] = "leak" + self.addCleanup(os.environ.pop, "BENCH_LEAK_PROBE", None) + self.trial("env > env.txt") + env = (self.out / f"{SCENARIO}-bin/workspace/env.txt").read_text() + self.assertNotIn("BENCH_LEAK_PROBE", env) + self.assertIn("BENCH_KNOWLEDGE=none", env) + self.assertIn(f"HOME={self.out}/{SCENARIO}-bin/home", env) + + def test_trace_records_envelope_and_serve_sessions(self): + agent = (f"cp {reference()}/* . && wright lint mode.ws -f json >/dev/null; " + "printf '{\"op\":\"capabilities\"}\\n{\"op\":\"lint\"}\\n' | wright serve mode.ws >/dev/null") + result = self.trial(agent) + events = bench_trace.read_events(self.out / f"{SCENARIO}-bin/wright-trace.jsonl") + lint = next(e for e in events if e["type"] == "call" and bench_trace.command_of(e["argv"]) == "lint") + self.assertEqual(lint["envelope"]["command"], "lint") + self.assertRegex(lint["envelope"]["inputIdentity"], r"^[0-9a-f]{64}$") + self.assertEqual(bench_trace.serve_ops(events), ["capabilities", "lint"]) + self.assertEqual(result["expectations"]["E11"]["status"], "pass") + self.assertEqual(result["expectations"]["E03"]["status"], "pass") + + def test_final_state_validation_expectation(self): + edit = f"cp {reference()}/mode.ws mode.ws" + late = self.trial(f"wright check mode.ws >/dev/null; sleep 0.6; {edit}") + self.assertEqual(late["expectations"]["E04"]["status"], "fail") + tight = self.trial(f"{edit}; sleep 0.6; wright check mode.ws >/dev/null") + self.assertEqual(tight["expectations"]["E04"]["status"], "pass") + self.assertEqual(tight["snapshots"]["firstValidIndex"], 1) + + def test_usage_context_and_first_valid(self): + rows = [ + {"t": 1.0, "input": 10, "output": 5, "cache_read": 100, "cache_write": 0, "context": 110, "context_limit": 1000}, + {"t": 2.0, "input": 20, "output": 5, "cache_read": 200, "cache_write": 0, "context": 220, "context_limit": 1000}, + ] + path = self.out / "usage.jsonl" + path.write_text("\n".join(json.dumps(r) for r in rows)) + summary = bench_trace.usage_summary(path, first_valid_t=1.5) + self.assertEqual(summary["totalTokens"], 340) + self.assertEqual(summary["peakContext"], 220) + self.assertEqual(summary["peakContextShare"], 0.22) + self.assertEqual(summary["toFirstValid"], {"turns": 1, "tokens": 115}) + self.assertIsNone(bench_trace.usage_summary(self.out / "missing.jsonl", None)) + + def test_adapter_usage_reaches_result(self): + row = json.dumps({"t": 1.0, "input": 1, "output": 2, "cache_read": 3, "cache_write": 0, "context": 4, "context_limit": 8}) + result = self.trial(f"echo '{row}' > \"$BENCH_USAGE\"; echo '{{\"loaded\": [\"stray\"]}}' > \"$BENCH_CONTEXT\"") + self.assertEqual(result["usage"]["totalTokens"], 6) + self.assertIn("unexpected loaded context", result["invalid"]) + + def test_infrastructure_failures_are_retried(self): + agent = 'if [ -f "$BENCH_RUN_DIR/tried" ]; then exit 0; else touch "$BENCH_RUN_DIR/tried"; exit 75; fi' + self.assertEqual(self.trial(agent)["infraRetries"], 1) + + def tiny_scenarios(self, negative_fails: list[str]) -> Path: + scenarios = self.out / "scenarios" + directory = scenarios / "tiny" + for name, text in (("seed", ""), ("reference", "Y"), ("negative/nope", "N")): + (directory / name).mkdir(parents=True) + (directory / name / "mode.ws").write_text(text) + spec = { + "id": "tiny", "family": "modification", "language": "workshop", "entry": "mode.ws", "writable": ["mode.ws"], + "checks": [{"id": "has-y", "kind": "contains", "file": "mode.ws", "text": "Y"}], + "negatives": {"nope": {"fails": negative_fails}}, + } + (directory / "scenario.json").write_text(json.dumps(spec)) + original = agent_bench.SCENARIOS + agent_bench.SCENARIOS = scenarios + self.addCleanup(setattr, agent_bench, "SCENARIOS", original) + return directory + + def test_validate_checks_negatives_fail_exactly_as_declared(self): + self.tiny_scenarios(["has-y"]) + self.assertTrue(agent_bench.validate(WRIGHT, self.out / "v1")) + (agent_bench.SCENARIOS / "tiny/scenario.json").write_text(json.dumps({ + **json.loads((agent_bench.SCENARIOS / "tiny/scenario.json").read_text()), "negatives": {"nope": {"fails": []}}, + })) + self.assertFalse(agent_bench.validate(WRIGHT, self.out / "v2")) + + def test_oracle_check_never_passes_silently(self): + directory = self.tiny_scenarios([]) + spec = {**json.loads((directory / "scenario.json").read_text()), "checks": [{"id": "oracle", "kind": "oracle"}]} + graded = bench_grade.grade({**spec, "dir": directory}, directory / "reference", WRIGHT) + self.assertFalse(graded["checks"][0]["passed"]) # a raw Workshop entry has no upstream OverPy verdict + self.assertRegex(graded["grader"]["hash"], r"^[0-9a-f]{64}$") + + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") + def test_oracle_disagreement_is_reported(self): + source = self.out / "n.opy" + source.write_text('settings {"main": {"description": "t"}, "gamemodes": {"skirmish": {"enabledMaps": ["workshopIsland"]}}}\n' + 'globalvar A\nrule "a":\n @Event eachPlayer\n A = allTankHeroes()\n') + auth = bench_grade.authorities(WRIGHT, source, self.out / "auth") + self.assertEqual(auth["oracle"]["status"], "error") + if auth["wrightCompile"]["status"] == "ok": + self.assertEqual(auth["disagreement"]["kind"], "wright-accepts-oracle-rejects") + + +class DetectorTest(unittest.TestCase): + def call(self, argv, exit_code=0, t=0.0, **extra): + return {"type": "call", "t": t, "argv": argv, "exit": exit_code, "seconds": 0.1, "stdoutBytes": 40, "stderrBytes": 0, "stderrHead": "", "envelope": None, **extra} + + def test_discovery_and_structured_output(self): + good = bench_trace.detect_expectations([self.call(["--help"]), self.call(["check", "m.ws", "-f", "json"])], [], {}, None) + self.assertEqual((good["E01"]["status"], good["E02"]["status"]), ("pass", "pass")) + bad = bench_trace.detect_expectations([self.call(["check", "m.ws"])], [], {}, None) + self.assertEqual((bad["E01"]["status"], bad["E02"]["status"]), ("fail", "fail")) + + def test_retry_storm_and_friction(self): + events = [self.call(["update"], 4), self.call(["update"], 4), self.call(["update"], 4), self.call(["nope"], 2, stderrHead="unrecognized subcommand")] + self.assertEqual(bench_trace.detect_expectations(events, [], {}, None)["E08"]["status"], "fail") + friction = bench_trace.friction(events) + self.assertEqual((friction["usageErrors"], friction["unknownSubcommands"], friction["identicalRepeats"]), (1, 1, 2)) + + def test_unused_wright_is_not_applicable(self): + result = bench_trace.detect_expectations([], [], {"stabilityRisk": True}, None) + self.assertEqual(result["E03"]["status"], "na") + self.assertEqual(result["E05"]["status"], "unavailable") + + +class ReportTest(unittest.TestCase): + def result(self, cell, trial, usable, tokens, split=None): + wright, knowledge, network = cell.split("/") + return { + "contract": "wright-agent-bench/v2", "scenario": "s", "split": split, "_trial": trial, "_dir": Path("d"), + "condition": {"wright": wright, "knowledge": knowledge, "network": network}, "agent": {"id": "m", "exit": 0, "seconds": 1.0}, + "usable": usable, "passed": usable, "usage": {"totalTokens": tokens, "peakContext": tokens // 2}, "wrightUse": {"invocations": 1 if wright != "none" else 0}, + } + + def test_wilson_interval(self): + low, high = bench_report.wilson(5, 10) + self.assertAlmostEqual(low, 0.237, places=2) + self.assertAlmostEqual(high, 0.763, places=2) + self.assertEqual(bench_report.wilson(0, 0), (0.0, 0.0)) + + def test_paired_efficiency_counts_only_both_usable_and_failures_cost(self): + runs = [self.result("none/none/off", 1, True, 1000), self.result("bin/none/off", 1, True, 600), + self.result("none/none/off", 2, False, 900), self.result("bin/none/off", 2, True, 700)] + text, summary = bench_report.render(runs) + self.assertIn("+1 / -0", text) + self.assertIn("+40% tokens (n=1)", text) + self.assertEqual(summary["cells"]["m|bin/none/off"]["tokensPerUsable"], 650) + self.assertEqual(summary["cells"]["m|none/none/off"]["tokensPerUsable"], 1900) + + def test_headroom_and_invalid_runs_are_reported(self): + runs = [self.result("none/none/off", t, True, 100) for t in range(1, 5)] + runs.append({**self.result("bin/none/off", 1, True, 100), "invalid": "canary"}) + text, _ = bench_report.render(runs) + self.assertIn("HEADROOM", text) + self.assertIn("INVALID: 1 run(s) excluded", text) + if __name__ == "__main__": unittest.main() diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 62909f19..d890d77c 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -1,7 +1,8 @@ # Agent Benchmark -- Contract: `wright-agent-bench/v1` +- Contract: `wright-agent-bench/v2` - Harness: [`benchmarks/agent/agent_bench.py`](../benchmarks/agent/agent_bench.py) +- Design and requirements: [`SPEC-414`](specs/SPEC-414-agent-benchmark-comparison.md) The benchmark answers one product question: can a general coding agent, with no Workshop-specific prompt or skill injection, use a project and Wright to @@ -14,70 +15,156 @@ evidence of correctness (use `--trials`). - The scenario workspace: the seed project only. - The scenario prompt, delivered on the agent's stdin. It states the requirement in user terms and never names Wright commands or Workshop APIs. -- In the `wright` condition, the released `wright` CLI and `wright serve` +- In the `bin` condition, the released `wright` CLI and `wright serve` session on `PATH`. -Not allowed: a Workshop/OverPy/OSTW system prompt or skill pack, a generated -API reference, or task-specific hints. The agent, model, and version are -recorded (`--agent-id`); the contract does not depend on a vendor. +Not allowed in the primary condition: a Workshop/OverPy/OSTW system prompt or +skill pack, a generated API reference, or task-specific hints. The agent, +model, and version are recorded (`--agent-id`); the contract does not depend +on a vendor. Other conditions below are experiments and are labeled as such. ## Conditions -Both conditions use the same workspace and prompt. They differ only in `PATH`: -`baseline` removes every directory that provides a `wright` executable; -`wright` prepends a logging shim for the binary under test. A determined agent -can still find a Wright binary elsewhere on disk, so run baselines in a clean -environment when that matters. +A cell is `//`. The task and prompt are identical +in every cell. + +| Factor | Levels | +| --- | --- | +| `wright` | `none`: no directory providing `wright` is on `PATH`. `bin`: `wright` on `PATH` through a tracing shim. `bin+skill`: `bin`, plus the guide directory given by `--skill-dir`, which the adapter installs. | +| `knowledge` | `none`; `wiki`: `--wiki-dir` is linked read-only as `./wiki` (never counted as an edit); `web`: the adapter enables its web tools. | +| `network` | `off` or `on`; `web` requires `on`. | + +Each run is scrubbed: a fresh `HOME`, an allowlisted environment (`--env-pass` +names host variables to keep), and no host instruction files. Two canaries run +before the agent; a failed canary marks the run `invalid` and it is excluded +from results: `wright` must not be reachable under level `none`, and +`--canary-cmd` (a command that must fail when the network is `off`) must fail +in the agent environment. A determined agent can still find a Wright binary +elsewhere on disk, so run `none` in a clean environment when that matters. + +## Adapters + +`--agent-cmd` is a shell command run in the workspace with the prompt on stdin. +The harness describes the cell through environment variables, and the adapter +enforces it: `BENCH_WRIGHT`, `BENCH_KNOWLEDGE`, `BENCH_NETWORK`, +`BENCH_SKILL_DIR` (only for `bin+skill`), `BENCH_HOST_PATH` (the unscrubbed +`PATH`, for locating the agent binary itself; do not pass it to the agent), and +`BENCH_RUN_DIR`. The adapter reports, all optional: + +| Path (env var) | Content | +| --- | --- | +| `BENCH_USAGE` | JSONL, one row per model turn: `t` (epoch seconds), `input`, `output`, `cache_read`, `cache_write`, `reasoning`, `context`, `context_limit` | +| `BENCH_TRANSCRIPT` | normalized JSONL of the agent's events | +| `BENCH_CONTEXT` | `{"loaded": [...]}`; a loaded item other than the expected guide invalidates the run | + +Exit code 75 means a provider or infrastructure failure: the harness retries +the trial (`--infra-retries`) and records the retries. Any other non-zero exit +is reported as an agent or infrastructure failure in the report diagnostics. +[`adapters/claude_code.py`](../benchmarks/agent/adapters/claude_code.py) is the +reference adapter. ## Scenarios `benchmarks/agent/scenarios//` contains `scenario.json`, `prompt.md`, -`seed/` (the initial workspace), and `reference/` (files overlaid on the seed -to form a passing solution). Families: `greenfield`, `understanding`, -`modification`, `diagnosis`. `scenario.json` fields: +`seed/` (the initial workspace), `reference/` (files overlaid on the seed +to form a passing solution), and optional `negative//` overlays. +`scenario.json` fields: | Field | Meaning | | --- | --- | | `id`, `family`, `language` | Identity; `language` is `workshop`, `opy`, or `ostw`, and a scenario may use a language only once its owner declares the needed capability supported | -| `entry` | Source file that Wright checks | +| `entry` | Source file that Wright checks and that is snapshotted after each write (`watch` overrides the file list) | | `writable` | Files the agent may change; any other change is reported as an unsafe edit | | `runtimeOnly` | Claims that only the Overwatch runtime can verify; reported as unverified, never as passed | +| `stabilityRisk` | `true` when finishing safely needs `lint` or `analyze`, not only `check` (expectation E03) | +| `split` | Optional `train` or `test`, for reports and guide tuning | +| `negatives` | `{name: {"fails": [check ids]}}`; the overlay must fail exactly those checks | | `checks` | Deterministic checks, each with `id`, `kind`, and `layer` | Check kinds: `check` (`wright check` reports no errors), `lint` (at most `max` findings with lint `code`), `symbols` (at least `min` symbols of `symbolKind` via `wright serve`), `contains` / `absent` (source text, `text` may be a list -of alternatives, `min`/`max` occurrences), and `answer` (`answer.json` key equals -`expected`). `layer` names what a failure implicates: `agent` for a requirement -the produced work does not meet, or `workshop-rs` / `opy-rs` / `deltin-rs` / -`wright` for validity or analysis results owned by that layer. +of alternatives, `min`/`max` occurrences), `answer` (`answer.json` key equals +`expected`), `oracle` (the pinned upstream OverPy compiler accepts the entry), +`wright-compile` (`wright compile` succeeds), and `compiled-contains` / +`compiled-absent` (regexes `all` and `any` over the compiled Workshop text of +`source` `oracle` or `wright`). Prefer structural checks; a regex over compiled +text needs a `reference` that matches and a `negative` that does not. `layer` +names what a failure implicates: `agent` for a requirement the produced work +does not meet, or `workshop-rs` / `opy-rs` / `deltin-rs` / `wright` for +validity or analysis results owned by that layer. + +## Grading authorities + +Every result records validity per authority, never collapsed into one verdict: +`wrightCheck`, `wrightCompile`, `workshopCheck` (the emitted Workshop text +through `workshop-rs`), and, for `.opy` entries, `oracle`. The oracle is the +pinned upstream compiler under `benchmarks/agent/oracle/` (`node` plus +`agent_bench.py setup-oracle`); when it is not installed the status is +`unavailable` and an `oracle` check fails explicitly instead of passing. +When Wright and the oracle disagree, `disagreement` names the direction and both +diagnostics, which is the reproducer for an owner Issue. Results carry the +grader hash (`grader.hash`); a grader change means regrading every run. ## Scenario validity -`agent_bench.py validate` requires every scenario's reference to pass all checks -and its untouched seed to fail at least one. A reference that fails a check is a -product or engine gap named by that check's `layer`, not an agent failure. This -runs in the CI benchmark job; running agents does not. +`agent_bench.py validate` requires every scenario's reference to pass all checks, +its untouched seed to fail at least one, and every negative to fail exactly the +checks it names. A reference that fails a check is a product or engine gap named +by that check's `layer`, not an agent failure. This runs in the CI benchmark +job; running agents does not. + +## Running + +```sh +python3 benchmarks/agent/agent_bench.py run --agent-id LABEL --agent-cmd CMD \ + --wright-level bin --knowledge none --network off --trials 5 +python3 benchmarks/agent/agent_bench.py matrix matrix.json # agents x cells x scenarios x trials +python3 benchmarks/agent/agent_bench.py report target/agent-bench [--regrade] +``` + +`matrix.json` lists `agents` (`{id, cmd}`), `cells`, optional `scenarios`, +`trials`, `parallel`, `seed` (run order is shuffled by it), and `options` +(`skill_dir`, `wiki_dir`, `env_pass`, ...). Finished runs are skipped, so an +interrupted matrix resumes. ## Result -`agent_bench.py run --agent-cmd CMD --agent-id LABEL` runs each -condition and writes `//-/result.json`, with -the workspace, `agent.log`, and Wright trace beside it. +`agent_bench.py run` writes `///-/result.json`, +with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | Field | Meaning | | --- | --- | -| `checks`, `passed`, `failedLayers` | Per-check outcome and the layers implicated by failures | -| `diagnostics` | Remaining `wright check` diagnostics with their owner origin | -| `unsafeEdits` | Files changed outside `writable` | -| `wrightUse` | Wright invocations by subcommand, failed invocations (correction rounds), and exits of 3 or 4 (unsupported or internal failures: candidate owner or environment gaps) | +| `condition`, `agent`, `environment`, `grader` | Cell, agent label/command/exit/seconds, OS/Python/Wright version, grader hash and oracle version | +| `checks`, `passed`, `usable`, `failedLayers` | Per-check outcome; `usable` is `passed` with no error-severity lint finding | +| `authorities`, `disagreement` | Validity per authority and any Wright/oracle disagreement | +| `diagnostics`, `lintFindings`, `unsafeEdits` | Remaining `wright check` diagnostics, lint rule codes, files changed outside `writable` | | `unverifiedRuntimeClaims` | The scenario's `runtimeOnly` claims | -| `agent`, `environment` | Agent label, command, exit, duration; OS, Python, Wright version, timestamp | - -`wrightUse` is recorded per CLI invocation; requests inside one `wright serve` -session are not itemized. Comparing `baseline` and `wright` results for the same -scenario shows what Wright adds. Exit 3 or 4 entries and failed `layer` values -are the input for owner Issues. +| `wrightUse` | Invocations by subcommand, failures, exits of 3 or 4 (candidate owner or environment gaps), and estimated output tokens per command | +| `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, identical repeats; expectation E01-E12 verdicts | +| `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | +| `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | +| `invalid`, `infraRetries` | Present when the run was excluded or retried | + +`wrightUse` is recorded per CLI invocation; `wright serve` sessions are teed +line by line into the trace. Comparing cells for the same scenario shows what +Wright adds. Exit 3 or 4 entries, disagreements, and failed `layer` values are +the input for owner Issues. + +Expectations E05, E07, E09, and E10 need the normalized agent transcript and +report `unavailable` until an adapter provides it. Estimated tokens for Wright +output use four bytes per token; provider-reported usage is authoritative. + +## Report + +`agent_bench.py report` writes `report.md` and `summary.json`: usable and passed +counts with Wilson 95% intervals, tokens per run and per usable result, peak +context, paired comparison against `none/none/off` (same scenario, agent, and +trial; token comparison only where both are usable), per-scenario and per-split +tables, expectation rates, friction, output size per command, and diagnostics. +Diagnostics flag headroom (baseline usable rate of at least 95%), infrastructure +failures, invalid runs, trial variance, and, with `--regrade`, a grader that +gives different verdicts on the same stored workspace. ## Cadence diff --git a/docs/specs/SPEC-414-agent-benchmark-comparison.md b/docs/specs/SPEC-414-agent-benchmark-comparison.md index 474d6b8e..b306e5aa 100644 --- a/docs/specs/SPEC-414-agent-benchmark-comparison.md +++ b/docs/specs/SPEC-414-agent-benchmark-comparison.md @@ -136,6 +136,20 @@ Pilot observations that shaped this spec (Sonnet only, 2 scenarios, 18 runs; not decisions it drives are candidates for output-shape or selection-default Issues in `wright`. The guide's own token size is reported and counted in the `bin+skill` context. +### Eval quality and guide tuning + +- REQ-025: The report flags problems with the eval itself: baseline headroom (usable rate at or above 95%, + where no gain can show), run-to-run variance, infrastructure failures and timeouts, invalid runs, and a + grader that gives different verdicts on the same stored workspace. +- REQ-026: Scenarios may carry a `split` of `train` or `test`, and reports break results out by split. Hard + scenarios are chosen by human judgment of difficulty, not because a current model fails them. +- REQ-027: A guide-tuning loop over the optional guide changes one surface per round (description or body), + scores train and held-out test scenarios under `bin+skill`, keeps the change only when both improve, reverts + it otherwise, and analyzes the cause after two or three stalled rounds. Failure transcripts are never pasted + into the guide, and reference solutions and answer keys stay outside the agent workspace. Attributable + metrics apply: description changes are judged by trigger rate, body changes by the E01-E12 rates and + outcome. The noise floor is measured before tuning starts. + ### Models and statistics - REQ-016: Agents run through an adapter that takes model, workspace, prompt, scrubbed environment, tool list, @@ -186,8 +200,8 @@ Pilot observations that shaped this spec (Sonnet only, 2 scenarios, 18 runs; not - Q-001 [product]: which models, reasoning settings, and total budget the first full run uses; owner PM. - Q-002 [product]: is Tier 2 required for acceptance, or only Tier 1; owner PM. - Q-003 [verification]: rubric-based judgment (REQ-011) in scope for the first version, or deferred; owner QA. -- Q-004 [architecture]: whether the trace analyzer and adapters live in `benchmarks/agent` or a sibling - directory, given the harness is currently a single script; owner Architect. +- Q-004 [architecture]: the harness now lives in `benchmarks/agent` as small modules beside + `agent_bench.py` (grading, trace, report, adapters); confirm or redirect; owner Architect. - Q-005 [verification]: the source of the wiki snapshot and its license and pinning method; owner QA. - Q-006 [verification]: the common tokenizer used for cross-model attribution (REQ-021) and how its error is reported; owner QA.