From b467dcd5e69a19a2d3b69cce4a56c169f59bf8b5 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Thu, 1 Oct 2026 00:29:11 +0800 Subject: [PATCH] feat(bench): add pi and Devin adapters and an ancestor instruction-file canary Refs #414. Adds adapters for pi and the Devin CLI that report per-turn usage, transcript, and loaded context, with host configuration kept out of the run (no context files or extensions for pi; an isolated HOME, no cross-tool rules, and denied MCP tools for Devin). Runs now fail a canary when instruction files exist above the workspace, the default output directory moves outside the repository, and results record whether network off was checked. Updates the matrix example and the benchmark contract. --- benchmarks/agent/adapters/devin.py | 102 ++++++++++++++++++++++++++ benchmarks/agent/adapters/pi.py | 106 +++++++++++++++++++++++++++ benchmarks/agent/agent_bench.py | 14 +++- benchmarks/agent/bench_grade.py | 2 +- benchmarks/agent/matrix.example.json | 6 +- benchmarks/agent/test_adapters.py | 61 +++++++++++++++ benchmarks/agent/test_agent_bench.py | 15 +++- docs/agent-benchmark.md | 30 +++++++- 8 files changed, 327 insertions(+), 9 deletions(-) create mode 100755 benchmarks/agent/adapters/devin.py create mode 100755 benchmarks/agent/adapters/pi.py create mode 100644 benchmarks/agent/test_adapters.py diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py new file mode 100755 index 0000000..b49eabe --- /dev/null +++ b/benchmarks/agent/adapters/devin.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +"""Adapter for the Devin CLI (`devin -p`): run one benchmark trial and report per-turn usage. + +Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required +(for example `swe-2-max`, see `devin models list`). The run uses an isolated HOME that contains only the Devin +credentials copied from the HOME the harness passed (`--env-pass HOME`), and a config that reads no other tools' +rules or skills, so the user's global skills, plugins, and instruction files do not load. BENCH_SKILL_DIR is +installed as a project skill in the workspace. Web tools are denied unless knowledge is `web`; the shell can +still reach the network, so network `off` is not enforced: use the harness --canary-cmd to check it. +Set BENCH_DEVIN_SANDBOX=1 to add `--sandbox`. Exit 75 marks a provider or infrastructure failure for a retry. +""" + +from __future__ import annotations + +import copy +import json +import os +import re +import shutil +import subprocess +import sys +from datetime import datetime +from pathlib import Path + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily") +WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] +MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers +NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} + + +def isolated_config(user_config: dict, model: str, web: bool) -> dict: + config = copy.deepcopy(user_config) + config["read_config_from"] = NO_TOOL_CONFIG + config["permissions"] = {"deny": MCP_TOOLS + ([] if web else WEB_TOOLS)} + config.setdefault("agent", {})["model"] = model + return config + + +def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: + """Usage rows (one per agent step with metrics), loaded skill and rule names, and plugin skill names. + + Built-in skills are not reported. Plugin skills come from the account's managed plugins, which cannot be + switched off from a config; their MCP tools are denied, so they are listed apart from `loaded`.""" + rows = [] + for step in export.get("steps", []): + metrics = step.get("metrics") + if step.get("source") == "agent" and metrics: + cached = metrics.get("cached_tokens") or 0 + prompt = metrics.get("prompt_tokens") or 0 + rows.append({ + "t": datetime.fromisoformat(step["timestamp"]).timestamp(), "input": max(prompt - cached, 0), + "output": metrics.get("completion_tokens"), "cache_read": cached, "cache_write": None, + "reasoning": None, "context": prompt, "context_limit": None, + }) + loaded, plugins = [], [] + for step in export.get("steps", []): + message = step.get("message") or "" + if step.get("source") != "system": + continue + loaded += re.findall(r' int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=swe-2-max python3 adapters/devin.py") + run_dir, workspace = Path(env["BENCH_RUN_DIR"]), Path.cwd() + real_home = Path(env["HOME"]) + home = run_dir / "devin-home" + (home / ".local/share/devin").mkdir(parents=True) + shutil.copy(real_home / ".local/share/devin/credentials.toml", home / ".local/share/devin/credentials.toml") + config, export, prompt = run_dir / "devin-config.json", run_dir / "devin-export.json", run_dir / "devin-prompt.txt" + config.write_text(json.dumps(isolated_config(json.loads((real_home / ".config/devin/config.json").read_text()), model, env["BENCH_KNOWLEDGE"] == "web"))) + prompt.write_text(sys.stdin.read()) + if env.get("BENCH_SKILL_DIR"): + skill = Path(env["BENCH_SKILL_DIR"]) + shutil.copytree(skill, workspace / ".agents/skills" / skill.name) + devin = shutil.which("devin", path=env.get("BENCH_HOST_PATH")) or "devin" + cmd = [devin, "--config", str(config), "--model", model, "--respect-workspace-trust", "false", + "--permission-mode", "dangerous", "--export", str(export), "--prompt-file", str(prompt), "-p"] + if env.get("BENCH_DEVIN_SANDBOX") == "1": + cmd.insert(1, "--sandbox") + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)} + proc = subprocess.run(cmd, capture_output=True, text=True, env=child_env) + rows, loaded, plugins = parse_export(json.loads(export.read_text())) if export.is_file() else ([], [], []) + Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows)) + Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else []))) + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins})) + sys.stdout.write(proc.stdout) + sys.stderr.write(proc.stderr) + if proc.returncode != 0: + return INFRA_EXIT if any(s in (proc.stdout + proc.stderr).lower() for s in TRANSIENT) else proc.returncode + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py new file mode 100755 index 0000000..ea5ba58 --- /dev/null +++ b/benchmarks/agent/adapters/pi.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python3 +"""Adapter for pi (`pi -p --mode json`): run one benchmark trial and report per-turn usage. + +Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required +(`provider/id`, see `pi --list-models`); BENCH_THINKING optionally sets `--thinking`. Discovery of context files, +extensions, prompt templates, themes, and skills is disabled; BENCH_SKILL_DIR is loaded explicitly. A provider that +is registered by an extension (for example Gemini through pi-antigravity) needs BENCH_PI_EXTENSIONS, a comma-separated +list of extension paths. Web tools come from BENCH_PI_WEB_EXTENSIONS (for example pi-web-access), loaded only when +knowledge is `web`. Network `off` is not enforced, because the shell can still reach the network: use the harness +--canary-cmd to check it. Exit 75 marks a provider or infrastructure failure for a retry. +""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import sys +import time +from pathlib import Path + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily") +SCALE = {"K": 1_000, "M": 1_000_000} + + +def context_limit(pi: str, model: str) -> int | None: + """Context window of the model from `pi --list-models`, or None.""" + listing = subprocess.run([pi, "--list-models", model.split("/")[-1]], capture_output=True, text=True).stdout + for line in listing.splitlines(): + cols = line.split() + if len(cols) > 2 and cols[1] == model.split("/")[-1]: + match = re.fullmatch(r"([\d.]+)([KM])", cols[2]) + return int(float(match.group(1)) * SCALE[match.group(2)]) if match else None + return None + + +def skill_names(system_message: dict) -> list[str]: + return re.findall(r"([^<]+)", (system_message.get("sections") or {}).get("skills", "")) + + +def usage_row(message: dict, limit: int | None, now: float) -> dict: + u = message.get("usage") or {} + read, write = u.get("cacheRead") or 0, u.get("cacheWrite") or 0 + return { + "t": now, "input": u.get("input"), "output": u.get("output"), "cache_read": read, "cache_write": write, + "reasoning": u.get("reasoning"), "context": (u.get("input") or 0) + read + write, "context_limit": limit, + } + + +def message_text(message: dict) -> str: + return "".join(part.get("text", "") for part in message.get("content", []) if isinstance(part, dict) and part.get("type") == "text") + + +def main() -> int: + env = os.environ + prompt = sys.stdin.read() + pi = shutil.which("pi", path=env.get("BENCH_HOST_PATH")) or "pi" + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=provider/id python3 adapters/pi.py") + cmd = [pi, "-p", "--mode", "json", "--no-session", "--no-context-files", "--no-extensions", "--no-prompt-templates", + "--no-themes", "--no-skills", "--tools", "read,bash,edit,write", "--model", model] + extensions = env.get("BENCH_PI_EXTENSIONS", "").split(",") + (env.get("BENCH_PI_WEB_EXTENSIONS", "").split(",") if env["BENCH_KNOWLEDGE"] == "web" else []) + for extension in filter(None, extensions): + cmd += ["-e", extension] + if env.get("BENCH_SKILL_DIR"): + cmd += ["--skill", env["BENCH_SKILL_DIR"]] + if env.get("BENCH_THINKING"): + cmd += ["--thinking", env["BENCH_THINKING"]] + limit = context_limit(pi, model) + child_env = {k: v for k, v in env.items() if k != "BENCH_HOST_PATH"} + proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + proc.stdin.write(prompt) + proc.stdin.close() + loaded: list[str] = [] + final, error = "", "" + with open(env["BENCH_USAGE"], "w") as usage, open(env["BENCH_TRANSCRIPT"], "w") as transcript: + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + now = time.time() + message = event.get("message") or {} + if event["type"] != "message_update": + transcript.write(json.dumps({"t": now, **event}) + "\n") + if event["type"] == "message_start" and message.get("role") == "system": + loaded = skill_names(message) + if event["type"] == "message_end" and message.get("role") == "assistant": + usage.write(json.dumps(usage_row(message, limit, now)) + "\n") + final = message_text(message) or final + if message.get("stopReason") == "error": + error = str(message.get("errorMessage") or message_text(message)) + stderr = proc.stderr.read() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 7e36cf6..719a490 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -127,8 +127,18 @@ def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) return env +INSTRUCTION_FILES = ("AGENTS.md", "CLAUDE.md", "GEMINI.md", ".cursorrules", ".github/copilot-instructions.md", ".windsurf/rules", ".cursor/rules") + + +def ancestor_instructions(workspace: Path) -> list[str]: + """Instruction files an agent would discover by walking up from the workspace.""" + return [str(parent / name) for parent in workspace.resolve().parents for name in INSTRUCTION_FILES if (parent / name).exists()] + + def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) -> str | None: """A failed canary invalidates the run. Returns the reason, or None.""" + if args.check_ancestors and (found := ancestor_instructions(workspace)): + return f"instruction files in ancestor directories of the workspace: {found}; use --out outside the repository" if cell["wright"] == "none" and shutil.which("wright", path=env["PATH"]): return "wright reachable under wright level 'none'" if cell["network"] == "off" and args.canary_cmd: @@ -198,6 +208,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") result = base_result(scenario, cell, args, out, seconds, agent_exit) result["infraRetries"] = infra_retries + result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only" context = context_report(out, cell) result["context"] = context if context.get("unexpected"): @@ -298,12 +309,13 @@ def main() -> int: for name in ("validate", "run", "matrix"): p = sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") - p.add_argument("--out", type=Path, default=ROOT / "target/agent-bench") + p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") for name in ("run", "matrix"): p = sub.choices[name] p.add_argument("--skill-dir", type=Path, help="pinned guide directory, exposed to the adapter as BENCH_SKILL_DIR") p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked read-only as ./wiki for knowledge 'wiki'") p.add_argument("--env-pass", nargs="*", default=[], help="host variables passed through the environment scrub") + p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace") p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py index ec9a54a..b94fd1c 100644 --- a/benchmarks/agent/bench_grade.py +++ b/benchmarks/agent/bench_grade.py @@ -12,7 +12,7 @@ HERE = Path(__file__).resolve().parent ORACLE = HERE / "oracle" GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json") -UNSAFE_IGNORED = ("wiki",) +UNSAFE_IGNORED = ("wiki", ".agents", ".devin") # linked wiki and skills installed through the agent's own mechanism def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: diff --git a/benchmarks/agent/matrix.example.json b/benchmarks/agent/matrix.example.json index c99d62d..a28ef01 100644 --- a/benchmarks/agent/matrix.example.json +++ b/benchmarks/agent/matrix.example.json @@ -1,6 +1,8 @@ { "agents": [ - {"id": "example-model", "cmd": "BENCH_MODEL=example-model python3 benchmarks/agent/adapters/claude_code.py"} + {"id": "pi-gpt-6-luna", "cmd": "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/wright/benchmarks/agent/adapters/pi.py"}, + {"id": "pi-gemini-3.6-flash", "cmd": "BENCH_MODEL=antigravity/gemini-3.6-flash BENCH_PI_EXTENSIONS=/abs/path/pi-antigravity BENCH_PI_WEB_EXTENSIONS=/abs/path/pi-web-access python3 /abs/path/wright/benchmarks/agent/adapters/pi.py"}, + {"id": "devin-swe-2-max", "cmd": "BENCH_MODEL=swe-2-max python3 /abs/path/wright/benchmarks/agent/adapters/devin.py"} ], "cells": [ {"wright": "none", "knowledge": "none", "network": "off"}, @@ -12,5 +14,5 @@ "trials": 5, "parallel": 2, "seed": 1, - "options": {"skill_dir": "../skills/skills/wright", "wiki_dir": "path/to/pinned/wiki", "env_pass": ["HOME", "ANTHROPIC_API_KEY"]} + "options": {"skill_dir": "/abs/path/skills/skills/wright", "wiki_dir": "/abs/path/pinned-wiki", "env_pass": ["HOME"]} } diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py new file mode 100644 index 0000000..7dfe490 --- /dev/null +++ b/benchmarks/agent/test_adapters.py @@ -0,0 +1,61 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent / "adapters")) + +import devin +import pi + + +class PiAdapterTest(unittest.TestCase): + def test_usage_row_counts_cache_in_context(self): + message = {"usage": {"input": 586, "output": 5, "cacheRead": 400, "cacheWrite": 100, "reasoning": 3}} + row = pi.usage_row(message, 272_000, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["cache_write"], row["context"], row["context_limit"]), (586, 400, 100, 1086, 272_000)) + + def test_loaded_skills_and_final_text(self): + system = {"sections": {"skills": "wrightother"}} + self.assertEqual(pi.skill_names(system), ["wright", "other"]) + self.assertEqual(pi.skill_names({"sections": {}}), []) + self.assertEqual(pi.message_text({"content": [{"type": "text", "text": "a"}, {"type": "tool_use"}, {"type": "text", "text": "b"}]}), "ab") + + +class DevinAdapterTest(unittest.TestCase): + export = {"steps": [ + {"source": "system", "message": '\n'}, + {"source": "system", "message": ( + "\n" + "- **wright**: A guide. (source: /w/.agents/skills/wright/SKILL.md)\n" + "- **devin-cli**: Docs. (source: /h/share/devin/docs)\n" + "- **upload-secrets**: Secrets. (source: builtin:upload-secrets)\n" + "- **context7-mcp**: Docs. (source: /h/cli/plugins/cache/x/skills/context7-mcp/SKILL.md)\n")}, + {"source": "user", "message": "task"}, + {"source": "agent", "timestamp": "2026-09-30T16:18:40+00:00", "metrics": {"prompt_tokens": 1000, "completion_tokens": 20, "cached_tokens": 600}}, + {"source": "agent", "timestamp": "2026-09-30T16:18:41+00:00", "message": "no metrics"}, + ]} + + def test_usage_rows_split_cached_prompt_tokens(self): + rows, _, _ = devin.parse_export(self.export) + self.assertEqual(len(rows), 1) + self.assertEqual((rows[0]["input"], rows[0]["cache_read"], rows[0]["output"], rows[0]["context"]), (400, 600, 20, 1000)) + + def test_loaded_context_excludes_builtins_and_lists_plugins_apart(self): + _, loaded, plugins = devin.parse_export(self.export) + self.assertEqual(loaded, ["AGENTS", "wright"]) + self.assertEqual(plugins, ["context7-mcp"]) + + def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): + base = {"read_config_from": {"claude": True}, "permissions": {"allow": ["Exec(*)"]}, "agent": {"model": "old"}} + closed = devin.isolated_config(base, "swe-2-max", web=False) + opened = devin.isolated_config(base, "swe-2-max", web=True) + self.assertFalse(any(closed["read_config_from"].values())) + self.assertEqual(closed["agent"]["model"], "swe-2-max") + self.assertIn("web_search", closed["permissions"]["deny"]) + self.assertNotIn("web_search", opened["permissions"]["deny"]) + self.assertIn("mcp_call_tool", opened["permissions"]["deny"]) + self.assertNotIn("allow", closed["permissions"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index c75bd3e..c8726ba 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -29,7 +29,7 @@ def setUp(self): def trial(self, agent_cmd: str, wright: str = "bin", scenario: str = SCENARIO, **options) -> dict: args = argparse.Namespace(**{ "wright": str(Path(WRIGHT).resolve()), "agent_id": "fake", "agent_cmd": agent_cmd, "timeout": 60, "infra_retries": 2, - "env_pass": [], "canary_cmd": None, "skill_dir": None, "wiki_dir": None, **options, + "env_pass": [], "canary_cmd": None, "skill_dir": None, "wiki_dir": None, "check_ancestors": False, **options, }) cell = {"wright": wright, "knowledge": "none", "network": "off"} return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{wright}") @@ -49,11 +49,22 @@ def test_levels_differ_only_in_wright_availability(self): def test_canary_rejects_reachable_wright_under_none(self): cell = {"wright": "none", "knowledge": "none", "network": "off"} - args = argparse.Namespace(canary_cmd=None) + args = argparse.Namespace(canary_cmd=None, check_ancestors=False) env = {"PATH": str(Path(WRIGHT).resolve().parent)} self.assertIn("reachable", agent_bench.canaries(cell, env, self.out, args)) self.assertIsNone(agent_bench.canaries({**cell, "wright": "bin"}, env, self.out, args)) + def test_canary_rejects_instruction_files_above_the_workspace(self): + outside = Path(tempfile.mkdtemp()).resolve() # outside the repository, whose own AGENTS.md would match + self.addCleanup(shutil.rmtree, outside, True) + workspace = outside / "repo/runs/w" + workspace.mkdir(parents=True) + cell = {"wright": "bin", "knowledge": "none", "network": "on"} + args = argparse.Namespace(canary_cmd=None, check_ancestors=True) + self.assertIsNone(agent_bench.canaries(cell, {"PATH": ""}, workspace, args)) + (outside / "repo/AGENTS.md").write_text("instructions") + self.assertIn("AGENTS.md", agent_bench.canaries(cell, {"PATH": ""}, workspace, args)) + def test_network_canary_invalidates_run(self): result = self.trial("true", canary_cmd="true") self.assertIn("network reachable", result["invalid"]) diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 1d67296..b2232fd 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -60,8 +60,32 @@ enforces it: `BENCH_WRIGHT`, `BENCH_KNOWLEDGE`, `BENCH_NETWORK`, Exit code 75 means a provider or infrastructure failure: the harness retries the trial (`--infra-retries`) and records the retries. Any other non-zero exit is reported as an agent or infrastructure failure in the report diagnostics. -[`adapters/claude_code.py`](../benchmarks/agent/adapters/claude_code.py) is the -reference adapter. + +The scenario `prompt.md` is the only text the harness gives an agent: it adds no +system prompt, hint, or Workshop context. Each agent keeps its own default system +prompt, which is part of what is measured. The adapter, not the harness, must +keep host configuration out of the run, and it reports what loaded through +`BENCH_CONTEXT`. + +| Adapter | Agent | Notes | +| --- | --- | --- | +| [`claude_code.py`](../benchmarks/agent/adapters/claude_code.py) | Claude Code | `BENCH_MODEL` selects the model. Removes web tools unless knowledge is `web`. | +| [`pi.py`](../benchmarks/agent/adapters/pi.py) | pi | `BENCH_MODEL=provider/id`. Disables context files, extensions, and skill discovery, and loads the guide explicitly. Providers registered by an extension need `BENCH_PI_EXTENSIONS`; web tools come from `BENCH_PI_WEB_EXTENSIONS`. | +| [`devin.py`](../benchmarks/agent/adapters/devin.py) | Devin CLI | `BENCH_MODEL` (for example `swe-2-max`). Runs with an isolated `HOME` holding only the Devin credentials, a config that reads no other tool's rules or skills, and MCP tools denied. Managed plugin skills are listed apart from `loaded`. | + +Set the model in `--agent-cmd` (the environment is scrubbed), and give the +adapter and skill paths as absolute paths, because the agent runs in the +workspace: `--agent-cmd "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/adapters/pi.py"`. +Pass `--env-pass HOME` when the agent authenticates from the real home directory. + +Two checks protect the context. The workspace must not sit below a directory +that holds instruction files (`AGENTS.md`, `CLAUDE.md`, and similar), because +agents discover them by walking up; the default `--out` is +`~/.cache/wright-agent-bench` for that reason, and a violation marks the run +`invalid` (`--no-ancestor-check` disables it). Network `off` is enforced only +when `--canary-cmd` is given and fails inside the agent environment; without it +the result records `networkEnforcement: declared-only`, which is what the shell +tools of pi and Devin provide today. ## Scenarios @@ -145,7 +169,7 @@ with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, identical repeats; expectation E01-E12 verdicts | | `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | | `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | -| `invalid`, `infraRetries` | Present when the run was excluded or retried | +| `invalid`, `infraRetries`, `networkEnforcement` | Present when the run was excluded or retried; whether network `off` was checked by a canary or only declared | `wrightUse` is recorded per CLI invocation; `wright serve` sessions are teed line by line into the trace. Comparing cells for the same scenario shows what