Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
102 changes: 102 additions & 0 deletions benchmarks/agent/adapters/devin.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
#!/usr/bin/env python3
"""Adapter for the Devin CLI (`devin -p`): run one benchmark trial and report per-turn usage.

Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required
(for example `swe-2-max`, see `devin models list`). The run uses an isolated HOME that contains only the Devin
credentials copied from the HOME the harness passed (`--env-pass HOME`), and a config that reads no other tools'
rules or skills, so the user's global skills, plugins, and instruction files do not load. BENCH_SKILL_DIR is
installed as a project skill in the workspace. Web tools are denied unless knowledge is `web`; the shell can
still reach the network, so network `off` is not enforced: use the harness --canary-cmd to check it.
Set BENCH_DEVIN_SANDBOX=1 to add `--sandbox`. Exit 75 marks a provider or infrastructure failure for a retry.
"""

from __future__ import annotations

import copy
import json
import os
import re
import shutil
import subprocess
import sys
from datetime import datetime
from pathlib import Path

INFRA_EXIT = 75
TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily")
WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"]
MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers
NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False}


def isolated_config(user_config: dict, model: str, web: bool) -> dict:
config = copy.deepcopy(user_config)
config["read_config_from"] = NO_TOOL_CONFIG
config["permissions"] = {"deny": MCP_TOOLS + ([] if web else WEB_TOOLS)}
config.setdefault("agent", {})["model"] = model
return config


def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]:
"""Usage rows (one per agent step with metrics), loaded skill and rule names, and plugin skill names.

Built-in skills are not reported. Plugin skills come from the account's managed plugins, which cannot be
switched off from a config; their MCP tools are denied, so they are listed apart from `loaded`."""
rows = []
for step in export.get("steps", []):
metrics = step.get("metrics")
if step.get("source") == "agent" and metrics:
cached = metrics.get("cached_tokens") or 0
prompt = metrics.get("prompt_tokens") or 0
rows.append({
"t": datetime.fromisoformat(step["timestamp"]).timestamp(), "input": max(prompt - cached, 0),
"output": metrics.get("completion_tokens"), "cache_read": cached, "cache_write": None,
"reasoning": None, "context": prompt, "context_limit": None,
})
loaded, plugins = [], []
for step in export.get("steps", []):
message = step.get("message") or ""
if step.get("source") != "system":
continue
loaded += re.findall(r'<rule name="([^"]+)"', message)
for name, source in re.findall(r"^- \*\*([^*]+)\*\*:.*?\(source: ([^)]*)\)\s*$", message, re.M):
if source.startswith("builtin:") or "/share/devin/docs" in source:
continue
(plugins if "/plugins/cache/" in source else loaded).append(name)
return rows, loaded, plugins


def main() -> int:
env = os.environ
model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=swe-2-max python3 adapters/devin.py")
run_dir, workspace = Path(env["BENCH_RUN_DIR"]), Path.cwd()
real_home = Path(env["HOME"])
home = run_dir / "devin-home"
(home / ".local/share/devin").mkdir(parents=True)
shutil.copy(real_home / ".local/share/devin/credentials.toml", home / ".local/share/devin/credentials.toml")
config, export, prompt = run_dir / "devin-config.json", run_dir / "devin-export.json", run_dir / "devin-prompt.txt"
config.write_text(json.dumps(isolated_config(json.loads((real_home / ".config/devin/config.json").read_text()), model, env["BENCH_KNOWLEDGE"] == "web")))
prompt.write_text(sys.stdin.read())
if env.get("BENCH_SKILL_DIR"):
skill = Path(env["BENCH_SKILL_DIR"])
shutil.copytree(skill, workspace / ".agents/skills" / skill.name)
devin = shutil.which("devin", path=env.get("BENCH_HOST_PATH")) or "devin"
cmd = [devin, "--config", str(config), "--model", model, "--respect-workspace-trust", "false",
"--permission-mode", "dangerous", "--export", str(export), "--prompt-file", str(prompt), "-p"]
if env.get("BENCH_DEVIN_SANDBOX") == "1":
cmd.insert(1, "--sandbox")
child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)}
proc = subprocess.run(cmd, capture_output=True, text=True, env=child_env)
rows, loaded, plugins = parse_export(json.loads(export.read_text())) if export.is_file() else ([], [], [])
Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows))
Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else [])))
Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins}))
sys.stdout.write(proc.stdout)
sys.stderr.write(proc.stderr)
if proc.returncode != 0:
return INFRA_EXIT if any(s in (proc.stdout + proc.stderr).lower() for s in TRANSIENT) else proc.returncode
return 0


if __name__ == "__main__":
sys.exit(main())
106 changes: 106 additions & 0 deletions benchmarks/agent/adapters/pi.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
#!/usr/bin/env python3
"""Adapter for pi (`pi -p --mode json`): run one benchmark trial and report per-turn usage.

Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required
(`provider/id`, see `pi --list-models`); BENCH_THINKING optionally sets `--thinking`. Discovery of context files,
extensions, prompt templates, themes, and skills is disabled; BENCH_SKILL_DIR is loaded explicitly. A provider that
is registered by an extension (for example Gemini through pi-antigravity) needs BENCH_PI_EXTENSIONS, a comma-separated
list of extension paths. Web tools come from BENCH_PI_WEB_EXTENSIONS (for example pi-web-access), loaded only when
knowledge is `web`. Network `off` is not enforced, because the shell can still reach the network: use the harness
--canary-cmd to check it. Exit 75 marks a provider or infrastructure failure for a retry.
"""

from __future__ import annotations

import json
import os
import re
import shutil
import subprocess
import sys
import time
from pathlib import Path

INFRA_EXIT = 75
TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily")
SCALE = {"K": 1_000, "M": 1_000_000}


def context_limit(pi: str, model: str) -> int | None:
"""Context window of the model from `pi --list-models`, or None."""
listing = subprocess.run([pi, "--list-models", model.split("/")[-1]], capture_output=True, text=True).stdout
for line in listing.splitlines():
cols = line.split()
if len(cols) > 2 and cols[1] == model.split("/")[-1]:
match = re.fullmatch(r"([\d.]+)([KM])", cols[2])
return int(float(match.group(1)) * SCALE[match.group(2)]) if match else None
return None


def skill_names(system_message: dict) -> list[str]:
return re.findall(r"<name>([^<]+)</name>", (system_message.get("sections") or {}).get("skills", ""))


def usage_row(message: dict, limit: int | None, now: float) -> dict:
u = message.get("usage") or {}
read, write = u.get("cacheRead") or 0, u.get("cacheWrite") or 0
return {
"t": now, "input": u.get("input"), "output": u.get("output"), "cache_read": read, "cache_write": write,
"reasoning": u.get("reasoning"), "context": (u.get("input") or 0) + read + write, "context_limit": limit,
}


def message_text(message: dict) -> str:
return "".join(part.get("text", "") for part in message.get("content", []) if isinstance(part, dict) and part.get("type") == "text")


def main() -> int:
env = os.environ
prompt = sys.stdin.read()
pi = shutil.which("pi", path=env.get("BENCH_HOST_PATH")) or "pi"
model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=provider/id python3 adapters/pi.py")
cmd = [pi, "-p", "--mode", "json", "--no-session", "--no-context-files", "--no-extensions", "--no-prompt-templates",
"--no-themes", "--no-skills", "--tools", "read,bash,edit,write", "--model", model]
extensions = env.get("BENCH_PI_EXTENSIONS", "").split(",") + (env.get("BENCH_PI_WEB_EXTENSIONS", "").split(",") if env["BENCH_KNOWLEDGE"] == "web" else [])
for extension in filter(None, extensions):
cmd += ["-e", extension]
if env.get("BENCH_SKILL_DIR"):
cmd += ["--skill", env["BENCH_SKILL_DIR"]]
if env.get("BENCH_THINKING"):
cmd += ["--thinking", env["BENCH_THINKING"]]
limit = context_limit(pi, model)
child_env = {k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env)
proc.stdin.write(prompt)
proc.stdin.close()
loaded: list[str] = []
final, error = "", ""
with open(env["BENCH_USAGE"], "w") as usage, open(env["BENCH_TRANSCRIPT"], "w") as transcript:
for line in proc.stdout:
try:
event = json.loads(line)
except json.JSONDecodeError:
continue
now = time.time()
message = event.get("message") or {}
if event["type"] != "message_update":
transcript.write(json.dumps({"t": now, **event}) + "\n")
if event["type"] == "message_start" and message.get("role") == "system":
loaded = skill_names(message)
if event["type"] == "message_end" and message.get("role") == "assistant":
usage.write(json.dumps(usage_row(message, limit, now)) + "\n")
final = message_text(message) or final
if message.get("stopReason") == "error":
error = str(message.get("errorMessage") or message_text(message))
stderr = proc.stderr.read()
code = proc.wait()
Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded}))
sys.stdout.write(final)
sys.stderr.write(stderr)
if error or code != 0:
return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1)
return 0


if __name__ == "__main__":
sys.exit(main())
14 changes: 13 additions & 1 deletion benchmarks/agent/agent_bench.py
Original file line number Diff line number Diff line change
Expand Up @@ -127,8 +127,18 @@ def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path)
return env


INSTRUCTION_FILES = ("AGENTS.md", "CLAUDE.md", "GEMINI.md", ".cursorrules", ".github/copilot-instructions.md", ".windsurf/rules", ".cursor/rules")


def ancestor_instructions(workspace: Path) -> list[str]:
"""Instruction files an agent would discover by walking up from the workspace."""
return [str(parent / name) for parent in workspace.resolve().parents for name in INSTRUCTION_FILES if (parent / name).exists()]


def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) -> str | None:
"""A failed canary invalidates the run. Returns the reason, or None."""
if args.check_ancestors and (found := ancestor_instructions(workspace)):
return f"instruction files in ancestor directories of the workspace: {found}; use --out outside the repository"
if cell["wright"] == "none" and shutil.which("wright", path=env["PATH"]):
return "wright reachable under wright level 'none'"
if cell["network"] == "off" and args.canary_cmd:
Expand Down Expand Up @@ -198,6 +208,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) -
(out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n")
result = base_result(scenario, cell, args, out, seconds, agent_exit)
result["infraRetries"] = infra_retries
result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only"
context = context_report(out, cell)
result["context"] = context
if context.get("unexpected"):
Expand Down Expand Up @@ -298,12 +309,13 @@ def main() -> int:
for name in ("validate", "run", "matrix"):
p = sub.add_parser(name)
p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test")
p.add_argument("--out", type=Path, default=ROOT / "target/agent-bench")
p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files")
for name in ("run", "matrix"):
p = sub.choices[name]
p.add_argument("--skill-dir", type=Path, help="pinned guide directory, exposed to the adapter as BENCH_SKILL_DIR")
p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked read-only as ./wiki for knowledge 'wiki'")
p.add_argument("--env-pass", nargs="*", default=[], help="host variables passed through the environment scrub")
p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace")
p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'")
p.add_argument("--timeout", type=int, default=1800)
p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)")
Expand Down
2 changes: 1 addition & 1 deletion benchmarks/agent/bench_grade.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@
HERE = Path(__file__).resolve().parent
ORACLE = HERE / "oracle"
GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json")
UNSAFE_IGNORED = ("wiki",)
UNSAFE_IGNORED = ("wiki", ".agents", ".devin") # linked wiki and skills installed through the agent's own mechanism


def wright_json(wright: str, args: list[str]) -> tuple[int, dict]:
Expand Down
6 changes: 4 additions & 2 deletions benchmarks/agent/matrix.example.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
{
"agents": [
{"id": "example-model", "cmd": "BENCH_MODEL=example-model python3 benchmarks/agent/adapters/claude_code.py"}
{"id": "pi-gpt-6-luna", "cmd": "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/wright/benchmarks/agent/adapters/pi.py"},
{"id": "pi-gemini-3.6-flash", "cmd": "BENCH_MODEL=antigravity/gemini-3.6-flash BENCH_PI_EXTENSIONS=/abs/path/pi-antigravity BENCH_PI_WEB_EXTENSIONS=/abs/path/pi-web-access python3 /abs/path/wright/benchmarks/agent/adapters/pi.py"},
{"id": "devin-swe-2-max", "cmd": "BENCH_MODEL=swe-2-max python3 /abs/path/wright/benchmarks/agent/adapters/devin.py"}
],
"cells": [
{"wright": "none", "knowledge": "none", "network": "off"},
Expand All @@ -12,5 +14,5 @@
"trials": 5,
"parallel": 2,
"seed": 1,
"options": {"skill_dir": "../skills/skills/wright", "wiki_dir": "path/to/pinned/wiki", "env_pass": ["HOME", "ANTHROPIC_API_KEY"]}
"options": {"skill_dir": "/abs/path/skills/skills/wright", "wiki_dir": "/abs/path/pinned-wiki", "env_pass": ["HOME"]}
}
61 changes: 61 additions & 0 deletions benchmarks/agent/test_adapters.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
import sys
import unittest
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent / "adapters"))

import devin
import pi


class PiAdapterTest(unittest.TestCase):
def test_usage_row_counts_cache_in_context(self):
message = {"usage": {"input": 586, "output": 5, "cacheRead": 400, "cacheWrite": 100, "reasoning": 3}}
row = pi.usage_row(message, 272_000, 1.0)
self.assertEqual((row["input"], row["cache_read"], row["cache_write"], row["context"], row["context_limit"]), (586, 400, 100, 1086, 272_000))

def test_loaded_skills_and_final_text(self):
system = {"sections": {"skills": "<available_skills><skill><name>wright</name></skill><skill><name>other</name></skill></available_skills>"}}
self.assertEqual(pi.skill_names(system), ["wright", "other"])
self.assertEqual(pi.skill_names({"sections": {}}), [])
self.assertEqual(pi.message_text({"content": [{"type": "text", "text": "a"}, {"type": "tool_use"}, {"type": "text", "text": "b"}]}), "ab")


class DevinAdapterTest(unittest.TestCase):
export = {"steps": [
{"source": "system", "message": '<rules type="always-on">\n<rule name="AGENTS" path="/x/AGENTS.md">'},
{"source": "system", "message": (
"<available_skills>\n"
"- **wright**: A guide. (source: /w/.agents/skills/wright/SKILL.md)\n"
"- **devin-cli**: Docs. (source: /h/share/devin/docs)\n"
"- **upload-secrets**: Secrets. (source: builtin:upload-secrets)\n"
"- **context7-mcp**: Docs. (source: /h/cli/plugins/cache/x/skills/context7-mcp/SKILL.md)\n")},
{"source": "user", "message": "task"},
{"source": "agent", "timestamp": "2026-09-30T16:18:40+00:00", "metrics": {"prompt_tokens": 1000, "completion_tokens": 20, "cached_tokens": 600}},
{"source": "agent", "timestamp": "2026-09-30T16:18:41+00:00", "message": "no metrics"},
]}

def test_usage_rows_split_cached_prompt_tokens(self):
rows, _, _ = devin.parse_export(self.export)
self.assertEqual(len(rows), 1)
self.assertEqual((rows[0]["input"], rows[0]["cache_read"], rows[0]["output"], rows[0]["context"]), (400, 600, 20, 1000))

def test_loaded_context_excludes_builtins_and_lists_plugins_apart(self):
_, loaded, plugins = devin.parse_export(self.export)
self.assertEqual(loaded, ["AGENTS", "wright"])
self.assertEqual(plugins, ["context7-mcp"])

def test_config_reads_no_other_tools_and_denies_web_unless_asked(self):
base = {"read_config_from": {"claude": True}, "permissions": {"allow": ["Exec(*)"]}, "agent": {"model": "old"}}
closed = devin.isolated_config(base, "swe-2-max", web=False)
opened = devin.isolated_config(base, "swe-2-max", web=True)
self.assertFalse(any(closed["read_config_from"].values()))
self.assertEqual(closed["agent"]["model"], "swe-2-max")
self.assertIn("web_search", closed["permissions"]["deny"])
self.assertNotIn("web_search", opened["permissions"]["deny"])
self.assertIn("mcp_call_tool", opened["permissions"]["deny"])
self.assertNotIn("allow", closed["permissions"])


if __name__ == "__main__":
unittest.main()
Loading
Loading