diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py new file mode 100644 index 00000000..eab1d515 --- /dev/null +++ b/benchmarks/agent/adapters/agy.py @@ -0,0 +1,80 @@ +#!/usr/bin/env python3 +"""Antigravity print adapter: isolated HOME and per-step streaming usage.""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + + +def usage_row(usage: dict, timestamp: float) -> dict: + cached = usage.get("cache_read_tokens") or 0 + thinking = usage.get("thinking_tokens") or 0 + return {"t": timestamp, "input": usage.get("input_tokens"), + "output": max((usage.get("output_tokens") or 0) - thinking, 0), + "cache_read": cached, "cache_write": None, "reasoning": thinking, + "context": (usage.get("input_tokens") or 0) + cached, "context_limit": None} + + +def main() -> int: + env = os.environ + run = Path(env["BENCH_RUN_DIR"]) + home = run / "agy-home" + state = home / ".gemini/antigravity-cli" + state.mkdir(parents=True, exist_ok=True) + for name in ("antigravity-oauth-token", "installation_id"): + shutil.copy(Path(env["HOME"]) / ".gemini/antigravity-cli" / name, state / name) + deny = ["mcp(*)", "execute_url(*)"] + if env["BENCH_KNOWLEDGE"] != "web": + deny.append("read_url(*)") + (state / "settings.json").write_text(json.dumps({"allowNonWorkspaceAccess": True, "permissions": {"allow": ["command(*)", "read_file(*)", f"write_file({Path.cwd()})"], "deny": deny}})) + installed = [] + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, Path.cwd() / ".agents/skills" / skill.name) + installed.append(skill.name) + binary = shutil.which("agy", path=env.get("BENCH_HOST_PATH")) or "agy" + command = [binary, "--model", env["BENCH_MODEL"], "--effort", env["BENCH_THINKING"], + "--output-format", "stream-json", "--print-timeout", "0", "--print", sys.stdin.read()] + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)} + seen, result, init, unexpected = set(), {}, {}, set() + with (run / "agy-stderr.log").open("w") as stderr, open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: + process = subprocess.Popen(command, env=child_env, stdout=subprocess.PIPE, stderr=stderr, text=True) + for line in process.stdout: + event = json.loads(line) + now = time.time() + transcript.write(json.dumps({"t": now, **event}) + "\n") + if event["event"] == "init": + init = event["init"] + step = event.get("step_update") or {} + tool = step.get("tool_name", "") + if env["BENCH_NETWORK"] == "off" and tool in {"search_web", "read_url_content", "browser_subagent", "open_browser_url"}: + unexpected.add("network-tool:" + tool) + process.terminate() + + if step.get("state") == "DONE" and step.get("usage") and step["step_index"] not in seen: + seen.add(step["step_index"]) + usage.write(json.dumps(usage_row(step["usage"], now)) + "\n") + if event["event"] == "result": + result = event["result"] + code = process.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "model": env["BENCH_MODEL"], "effort": env["BENCH_THINKING"], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) + (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": env["BENCH_THINKING"], "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) + sys.stdout.write(result.get("response", "")) + stderr_text = (run / "agy-stderr.log").read_text() + sys.stderr.write(stderr_text) + error = str(result.get("error", "")) + stderr_text + if "no output produced" in stderr_text and "headless" in stderr_text: + return 75 + if code or result.get("status") != "SUCCESS": + return 75 if any(s in error.lower() for s in ("quota", "rate limit", "429", "temporarily", "503", "credits")) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index c20dc27c..add3a71a 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -2,7 +2,7 @@ """Reference adapter: run Claude Code for one benchmark trial and report per-turn usage. Reads the prompt on stdin. Honors the BENCH_* contract (docs/agent-benchmark.md): tools follow BENCH_KNOWLEDGE, -BENCH_SKILL_DIR is installed as a plugin, and BENCH_USAGE / BENCH_TRANSCRIPT / BENCH_CONTEXT are written. +BENCH_SKILL_DIRS (one directory per installed skill) are installed as plugin skills, and BENCH_USAGE / BENCH_TRANSCRIPT / BENCH_CONTEXT are written. The agent binary is resolved on BENCH_HOST_PATH because the agent's own PATH hides Wright when the level is `none`. The model comes from BENCH_MODEL (default `sonnet`). It removes web tools unless knowledge is `web`, but it does not sandbox the network: pair it with the harness --canary-cmd to detect a reachable network under `off`. @@ -38,14 +38,16 @@ def main() -> int: if not web: cmd += ["--disallowedTools", *WEB_TOOLS] loaded: list[str] = [] - if env.get("BENCH_SKILL_DIR"): - skill = Path(env["BENCH_SKILL_DIR"]) + skills = [Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p] + if skills: plugin = Path(tempfile.mkdtemp(dir=env["BENCH_RUN_DIR"])) (plugin / ".claude-plugin").mkdir() - (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skill"})) - shutil.copytree(skill, plugin / "skills" / skill.name) + (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skills"})) + for skill in skills: + shutil.copytree(skill, plugin / "skills" / skill.name) + loaded.append(skill.name) cmd += ["--plugin-dir", str(plugin)] - loaded.append(skill.name) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "model": env.get("BENCH_MODEL", "sonnet"), "tools": TOOLS + (WEB_TOOLS if web else [])}, indent=2)) proc = subprocess.Popen( cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, @@ -53,7 +55,7 @@ def main() -> int: proc.stdin.write(prompt) proc.stdin.close() final, errored = "", False - with open(env["BENCH_USAGE"], "w") as usage, open(env["BENCH_TRANSCRIPT"], "w") as transcript: + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage for line in proc.stdout: try: event = json.loads(line) diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py new file mode 100644 index 00000000..cecc6848 --- /dev/null +++ b/benchmarks/agent/adapters/codex.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""Codex exec adapter with an isolated HOME, session usage, and observed skill context.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import sys +import time +from datetime import datetime +from pathlib import Path + + +def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: + cached = usage.get("cached_input_tokens") or 0 + reasoning = usage.get("reasoning_output_tokens") or 0 + return {"t": timestamp, "input": max((usage.get("input_tokens") or 0) - cached, 0), + "output": max((usage.get("output_tokens") or 0) - reasoning, 0), + "cache_read": cached, "cache_write": usage.get("cache_write_input_tokens"), + "reasoning": reasoning, "context": usage.get("input_tokens"), "context_limit": limit} + + +def loaded_skills(text: str) -> tuple[list[str], list[str]]: + roots = dict(re.findall(r"^- `([^`]+)` = `([^`]+)`", text, re.M)) + custom, builtin = [], [] + for name, path in re.findall(r"^- ([^:\n]+): .*?\(file: ([^)]+)\)$", text, re.M): + alias, _, relative = path.partition("/") + resolved = str(Path(roots.get(alias, alias)) / relative) + (builtin if "/skills/.system/" in resolved else custom).append(name) + return custom, builtin + + +def session_usage(state: Path): + for path in state.glob("sessions/**/*.jsonl"): + for line in path.read_text().splitlines(): + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + payload = event.get("payload") or {} + info = payload.get("info") + if payload.get("type") == "token_count" and info: + yield json.dumps(info["total_token_usage"], sort_keys=True), usage_row(info["last_token_usage"], datetime.fromisoformat(event["timestamp"]).timestamp(), info.get("model_context_window")) + + +def main() -> int: + env = os.environ + model = env["BENCH_MODEL"] + effort = env["BENCH_THINKING"] + run = Path(env["BENCH_RUN_DIR"]) + home = run / "codex-home" + state = home / ".codex" + state.mkdir(parents=True, exist_ok=True) + shutil.copy(Path(env["HOME"]) / ".codex/auth.json", state / "auth.json") + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, Path.cwd() / ".agents/skills" / skill.name) + binary = shutil.which("codex", path=env.get("BENCH_HOST_PATH")) or "codex" + # Codex's own seatbelt cannot be applied inside the harness --file-sandbox (macOS refuses nested sandboxes), so it is off + # here: the harness sandbox is the file-write boundary, and network `off` is declared-only for this adapter. + command = [binary, "exec", "--json", "--ignore-user-config", "--ignore-rules", "--skip-git-repo-check", + "--disable", "apps", "--disable", "plugins", "--disable", "remote_plugin", + "--disable", "skill_mcp_dependency_install", + "--sandbox", "danger-full-access", "-c", 'approval_policy="never"', + "-c", 'web_search="disabled"', "-c", f'model_reasoning_effort="{effort}"', "-m", model, "-"] + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "CODEX_HOME": str(state)} + prompt = sys.stdin.read() + final, errors, summary, seen, servers, item_types = "", [], None, set(), set(), set() + with (run / "codex-stderr.log").open("w") as stderr, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript, open(env["BENCH_USAGE"], "w", buffering=1) as usage: + process = subprocess.Popen(command, env=child_env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr, text=True) + process.stdin.write(prompt) + process.stdin.close() + for line in process.stdout: + event = json.loads(line) + transcript.write(json.dumps({"t": time.time(), **event}) + "\n") + item = event.get("item") or {} + item_types.add(item.get("type")) + if item.get("type") == "mcp_tool_call": + servers.add(item.get("server", "unknown")) + if item.get("type") == "agent_message": + final = item.get("text", final) + if event["type"] in ("error", "turn.failed"): + errors.append(str(event.get("message") or event.get("error"))) + if event["type"] == "turn.completed": + summary = event.get("usage") + for key, row in session_usage(state): + if key not in seen: + seen.add(key) + usage.write(json.dumps(row) + "\n") + code = process.wait() + for key, row in session_usage(state): + if key not in seen: + seen.add(key) + usage.write(json.dumps(row) + "\n") + if not seen and summary: + row = usage_row(summary, time.time(), None) + row["context"] = None + usage.write(json.dumps(row) + "\n") + loaded, builtin, observed = [], [], {} + for path in state.glob("sessions/**/*.jsonl"): + for line in path.read_text().splitlines(): + event = json.loads(line) + payload = event.get("payload") or {} + if event["type"] == "turn_context": + observed = {k: payload.get(k) for k in ("model", "effort")} + if event["type"] == "response_item" and payload.get("role") == "developer": + for part in payload.get("content", []): + names, builtins = loaded_skills(part.get("text", "")) + loaded += names + builtin += builtins + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": sorted(set(loaded) | {f"unexpected-mcp:{server}" for server in servers}), "builtinSkills": sorted(set(builtin))})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "codex", "model": observed.get("model") or model, "effort": observed.get("effort") or effort, "sandbox": "danger-full-access inside the harness file sandbox", "toolsObserved": sorted(t for t in item_types if t)}, indent=2)) + (run / "adapter.json").write_text(json.dumps({"agent": "codex", "requestedModel": model, "requestedEffort": effort, "observed": observed, "usageSource": "session-token-count" if observed else "turn-summary"}, indent=2)) + sys.stdout.write(final) + stderr_text = (run / "codex-stderr.log").read_text() + sys.stderr.write(stderr_text) + failure = " ".join(errors) + stderr_text + if code or errors: + return 75 if any(s in failure.lower() for s in ("rate limit", "usage limit", "429", "quota", "temporarily", "503")) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py new file mode 100755 index 00000000..383a7ba1 --- /dev/null +++ b/benchmarks/agent/adapters/devin.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +"""Adapter for the Devin CLI (`devin -p`): run one benchmark trial and report per-turn usage. + +Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required +(for example `swe-2-max`, see `devin models list`). The run uses an isolated HOME that contains only the Devin +credentials copied from the HOME the harness passed (`--env-pass HOME`), and a config that reads no other tools' +rules or skills. Managed plugin skills are reported separately; --file-sandbox blocks outside-workspace +instruction files. BENCH_SKILL_DIRS (one directory per installed skill) are installed as project skills in the workspace. Web tools are denied unless knowledge is `web`; the shell can +still reach the network, so network `off` is not enforced: use the harness --canary-cmd to check it. +Set BENCH_DEVIN_SANDBOX=1 to add `--sandbox`. Exit 75 marks a provider or infrastructure failure for a retry. +""" + +from __future__ import annotations + +import copy +import json +import os +import re +import shutil +import subprocess +import sys +from datetime import datetime +from pathlib import Path + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] +MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers +NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} + + +def isolated_config(user_config: dict, model: str, web: bool) -> dict: + config = copy.deepcopy(user_config) + config["read_config_from"] = NO_TOOL_CONFIG + config["permissions"] = {"deny": MCP_TOOLS + ([] if web else WEB_TOOLS)} + config.setdefault("agent", {})["model"] = model + return config + + +def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: + """Usage rows (one per agent step with metrics), loaded skill and rule names, and plugin skill names. + + Built-in skills are not reported. Plugin skills come from the account's managed plugins, which cannot be + switched off from a config; their MCP tools are denied, so they are listed apart from `loaded`.""" + rows = [] + for step in export.get("steps", []): + metrics = step.get("metrics") + if step.get("source") == "agent" and metrics: + cached = metrics.get("cached_tokens") or 0 + prompt = metrics.get("prompt_tokens") or 0 + rows.append({ + "t": datetime.fromisoformat(step["timestamp"]).timestamp(), "input": max(prompt - cached, 0), + "output": metrics.get("completion_tokens"), "cache_read": cached, "cache_write": None, + "reasoning": None, "context": prompt, "context_limit": None, + }) + loaded, plugins = [], [] + for step in export.get("steps", []): + message = step.get("message") or "" + if step.get("source") != "system": + continue + loaded += re.findall(r' int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=swe-2-max python3 adapters/devin.py") + run_dir, workspace = Path(env["BENCH_RUN_DIR"]), Path.cwd() + real_home = Path(env["HOME"]) + home = run_dir / "devin-home" + (home / ".local/share/devin").mkdir(parents=True) + shutil.copy(real_home / ".local/share/devin/credentials.toml", home / ".local/share/devin/credentials.toml") + config, export, prompt = run_dir / "devin-config.json", run_dir / "devin-export.json", run_dir / "devin-prompt.txt" + config.write_text(json.dumps(isolated_config(json.loads((real_home / ".config/devin/config.json").read_text()), model, env["BENCH_KNOWLEDGE"] == "web"))) + prompt.write_text(sys.stdin.read()) + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, workspace / ".agents/skills" / skill.name) + devin = shutil.which("devin", path=env.get("BENCH_HOST_PATH")) or "devin" + cmd = [devin, "--config", str(config), "--model", model, "--respect-workspace-trust", "false", + "--permission-mode", "dangerous", "--export", str(export), "--prompt-file", str(prompt), "-p"] + if env.get("BENCH_DEVIN_SANDBOX") == "1": + cmd.insert(1, "--sandbox") + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)} + proc = subprocess.run(cmd, capture_output=True, text=True, env=child_env) + rows, loaded, plugins = parse_export(json.loads(export.read_text())) if export.is_file() else ([], [], []) + Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows)) + Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else []))) + exported = json.loads(export.read_text()) if export.is_file() else {} + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "model": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins})) + sys.stdout.write(proc.stdout) + sys.stderr.write(proc.stderr) + if proc.returncode != 0: + return INFRA_EXIT if any(s in (proc.stdout + proc.stderr).lower() for s in TRANSIENT) else proc.returncode + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py new file mode 100755 index 00000000..134874af --- /dev/null +++ b/benchmarks/agent/adapters/pi.py @@ -0,0 +1,119 @@ +#!/usr/bin/env python3 +"""Adapter for pi (`pi -p --mode json`): run one benchmark trial and report per-turn usage. + +Reads the prompt on stdin and honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is required +(`provider/id`, see `pi --list-models`); BENCH_THINKING optionally sets `--thinking`. Discovery of context files, +extensions, prompt templates, themes, and skills is disabled; BENCH_SKILL_DIRS (one directory per installed skill) are loaded explicitly. A provider that +is registered by an extension (for example Gemini through pi-antigravity) needs BENCH_PI_EXTENSIONS, a comma-separated +list of extension paths. Web tools come from BENCH_PI_WEB_EXTENSIONS (for example pi-web-access), loaded only when +knowledge is `web`. Network `off` is not enforced, because the shell can still reach the network: use the harness +--canary-cmd to check it. Exit 75 marks a provider or infrastructure failure for a retry. +""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import sys +import time +from pathlib import Path + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +SCALE = {"K": 1_000, "M": 1_000_000} + + +def context_limit(pi: str, model: str, env: dict, extensions: list[str]) -> int | None: + """Context window of the model from `pi --list-models`, or None.""" + command = [pi, "--no-extensions"] + for extension in filter(None, extensions): + command += ["-e", extension] + listing = subprocess.run([*command, "--list-models", model.split("/")[-1]], env=env, capture_output=True, text=True).stdout + for line in listing.splitlines(): + cols = line.split() + if len(cols) > 2 and cols[0] == model.split("/")[0] and cols[1] == model.split("/")[-1]: + match = re.fullmatch(r"([\d.]+)([KM])", cols[2]) + return int(float(match.group(1)) * SCALE[match.group(2)]) if match else None + return None + + +def skill_names(system_message: dict) -> list[str]: + return re.findall(r"([^<]+)", (system_message.get("sections") or {}).get("skills", "")) + + +def usage_row(message: dict, limit: int | None, now: float) -> dict: + u = message.get("usage") or {} + read, write = u.get("cacheRead") or 0, u.get("cacheWrite") or 0 + return { + "t": now, "input": u.get("input"), "output": max((u.get("output") or 0) - (u.get("reasoning") or 0), 0), "cache_read": read, "cache_write": write, + "reasoning": u.get("reasoning"), "context": (u.get("input") or 0) + read + write, "context_limit": limit, + } + + +def message_text(message: dict) -> str: + return "".join(part.get("text", "") for part in message.get("content", []) if isinstance(part, dict) and part.get("type") == "text") + + +def main() -> int: + env = os.environ + prompt = sys.stdin.read() + pi = shutil.which("pi", path=env.get("BENCH_HOST_PATH")) or "pi" + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=provider/id python3 adapters/pi.py") + home = Path(env["BENCH_RUN_DIR"]) / "pi-home" + state = home / ".pi/agent" + state.mkdir(parents=True, exist_ok=True) + for name in ("auth.json", "antigravity-accounts.json"): + source = Path(env["HOME"]) / ".pi/agent" / name + if source.is_file(): + shutil.copy(source, state / name) + cmd = [pi, "-p", "--mode", "json", "--no-session", "--no-context-files", "--no-extensions", "--no-prompt-templates", + "--no-themes", "--no-skills", "--tools", "read,bash,edit,write", "--model", model] + extensions = env.get("BENCH_PI_EXTENSIONS", "").split(",") + (env.get("BENCH_PI_WEB_EXTENSIONS", "").split(",") if env["BENCH_KNOWLEDGE"] == "web" else []) + for extension in filter(None, extensions): + cmd += ["-e", extension] + for skill in (p for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + cmd += ["--skill", skill] + if env.get("BENCH_THINKING"): + cmd += ["--thinking", env["BENCH_THINKING"]] + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "PI_CODING_AGENT_DIR": str(state)} + limit = context_limit(pi, model, child_env, extensions) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) + proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + proc.stdin.write(prompt) + proc.stdin.close() + loaded: list[str] = [] + final, error = "", "" + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + now = time.time() + message = event.get("message") or {} + if event["type"] != "message_update": + transcript.write(json.dumps({"t": now, **event}) + "\n") + if event["type"] == "message_start" and message.get("role") == "system": + loaded = skill_names(message) + if event["type"] == "message_end" and message.get("role") == "assistant": + usage.write(json.dumps(usage_row(message, limit, now)) + "\n") + final = message_text(message) or final + if message.get("stopReason") == "error": + error = str(message.get("errorMessage") or message_text(message)) + else: + error = "" + stderr = proc.stderr.read() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 7e36cf60..4d608ce1 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -4,15 +4,18 @@ from __future__ import annotations import argparse +import functools import hashlib import json import os import platform import random +import re import shutil import signal import subprocess import sys +import threading import time from concurrent.futures import ThreadPoolExecutor from datetime import datetime, timezone @@ -20,13 +23,18 @@ import bench_grade import bench_report +import bench_score import bench_trace +import bench_wiki +import wiki_skill HERE = Path(__file__).resolve().parent ROOT = HERE.parent.parent SCENARIOS = HERE / "scenarios" -RESULT_CONTRACT = "wright-agent-bench/v2" -WRIGHT_LEVELS = ("none", "bin", "bin+skill") +RESULT_CONTRACT = "wright-agent-bench/v3" +TOOLS = ("none", "wright", "overpy") +SKILLS = ("wright-skill", "workshop-skill", "opy-skill", "workshop-format-skill") +SKILL_LANGUAGE = {"opy-skill": "opy", "workshop-format-skill": "workshop"} # skills that teach one language and apply to its scenarios only KNOWLEDGE_LEVELS = ("none", "wiki", "web") INFRA_EXIT = 75 # EX_TEMPFAIL: the adapter reports a provider or infrastructure failure, not an agent failure ENV_KEEP = ("LANG", "LC_ALL", "TERM", "TMPDIR", "USER", "LOGNAME") @@ -81,24 +89,72 @@ def validate(wright: str, out: Path) -> bool: def baseline_path(path: str) -> str: - """PATH without any directory that provides a `wright` executable.""" - kept = [d for d in path.split(os.pathsep) if d and not shutil.which("wright", path=d)] + """PATH without any directory that provides a benchmark tool (`wright` or `overpy`).""" + kept = [d for d in path.split(os.pathsep) if d and not any(shutil.which(tool, path=d) for tool in ("wright", "overpy"))] return os.pathsep.join(kept) +def normalize_cell(raw: dict) -> dict: + return {"tool": raw["tool"], "skills": sorted(raw.get("skills") or []), "knowledge": raw["knowledge"], "network": raw["network"]} + + def cell_label(cell: dict) -> str: - return f"{cell['wright']}/{cell['knowledge']}/{cell['network']}" + """`tool[+skill...]/knowledge/network`, for example `wright+wright-skill/none/off`; the baseline is `none/none/off`.""" + return f"{'+'.join([cell['tool'], *cell['skills']])}/{cell['knowledge']}/{cell['network']}" + + +def applicable(scenario: dict, cell: dict) -> bool: + """The `overpy` tool and the language skills apply to their own language only; a raw Workshop scenario has no OverPy cell.""" + if cell["tool"] == "overpy" and scenario["language"] != "opy": + return False + return all(SKILL_LANGUAGE.get(s, scenario["language"]) == scenario["language"] for s in cell["skills"]) + + +def skill_name(directory: Path) -> str: + match = re.search(r"^name:\s*(\S+)", (directory / "SKILL.md").read_text(), re.M) + return match.group(1) if match else directory.name + + +def skill_identity(logical: str, directory: Path) -> dict: + """Name, content hash, and build record (when the skill has one) of an installed skill.""" + record = {"name": logical, "skillName": skill_name(directory), "sha256": wiki_skill.content_hash(directory)} + build = directory / "BUILD.json" + if build.is_file(): + record["build"] = json.loads(build.read_text()) + if record["build"].get("skillSha256") != record["sha256"]: + raise SystemExit(f"skill content mismatch for {logical}: {directory}") + return record + + +def overpy_launcher(out: Path, host_path: str) -> Path: + """A launcher for the pinned OverPy CLI, so the tool shim has a real executable to call.""" + node = shutil.which("node", path=host_path) + cli = bench_grade.ORACLE / "node_modules/overpy/cli.js" + if not node or not cli.is_file(): + raise SystemExit("tool 'overpy' requires node and the pinned oracle; run `agent_bench.py setup-oracle`") + launcher = out / "real" / "overpy" + launcher.parent.mkdir(exist_ok=True) + launcher.write_text(f'#!/bin/sh\nexec "{node}" "{cli}" "$@"\n') + launcher.chmod(0o755) + return launcher def check_cell(cell: dict, args: argparse.Namespace) -> None: - if cell["wright"] not in WRIGHT_LEVELS or cell["knowledge"] not in KNOWLEDGE_LEVELS or cell["network"] not in ("off", "on"): + if cell["tool"] not in TOOLS or cell["knowledge"] not in KNOWLEDGE_LEVELS or cell["network"] not in ("off", "on") or any(s not in SKILLS for s in cell["skills"]): raise SystemExit(f"invalid condition {cell_label(cell)}") if cell["knowledge"] == "web" and cell["network"] != "on": raise SystemExit("knowledge 'web' requires network 'on'") - if cell["wright"] == "bin+skill" and not args.skill_dir: - raise SystemExit("wright level 'bin+skill' requires --skill-dir") - if cell["knowledge"] == "wiki" and not args.wiki_dir: - raise SystemExit("knowledge 'wiki' requires --wiki-dir") + for name in cell["skills"]: + directory = (args.skill_dirs or {}).get(name) + if not directory or not (Path(directory) / "SKILL.md").is_file(): + raise SystemExit(f"skill '{name}' requires --skill {name}=DIR with a SKILL.md") + skill_identity(name, Path(directory)) + if cell["tool"] == "overpy" and not bench_grade.oracle_available(): + raise SystemExit("tool 'overpy' requires the pinned oracle; run `agent_bench.py setup-oracle`") + if cell["knowledge"] == "wiki": + if not (args.wiki_dir and (Path(args.wiki_dir) / "SNAPSHOT.json").is_file()): + raise SystemExit("knowledge 'wiki' requires --wiki-dir pointing at a snapshot (see `agent_bench.py wiki-snapshot`)") + bench_wiki.identity(Path(args.wiki_dir)) def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) -> dict: @@ -108,29 +164,42 @@ def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) env = {k: os.environ[k] for k in (*ENV_KEEP, *args.env_pass) if k in os.environ} env.setdefault("HOME", str(home)) path = baseline_path(os.environ["PATH"]) - if cell["wright"] != "none": + if cell["tool"] != "none": shim_dir = out / "bin" shim_dir.mkdir() - shim = shim_dir / "wright" - shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim "$@"\n') + shim = shim_dir / cell["tool"] + shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim {cell["tool"]} "$@"\n') shim.chmod(0o755) - env.update(WRIGHT_BENCH_REAL=args.wright, WRIGHT_BENCH_TRACE=str(out / "wright-trace.jsonl"), WRIGHT_BENCH_SIDECAR=str(out / "wright-calls")) + real = args.wright if cell["tool"] == "wright" else str(overpy_launcher(out, os.environ["PATH"])) + env.update({f"BENCH_TOOL_REAL_{cell['tool'].upper()}": real, "BENCH_TOOL_TRACE": str(out / "tool-trace.jsonl"), "BENCH_TOOL_SIDECAR": str(out / "tool-calls")}) path = f"{shim_dir}{os.pathsep}{path}" env.update( PATH=path, BENCH_HOST_PATH=os.environ["PATH"], BENCH_WORKSPACE=str(workspace), BENCH_RUN_DIR=str(out), BENCH_AGENT_ID=args.agent_id, BENCH_USAGE=str(out / "usage.jsonl"), BENCH_TRANSCRIPT=str(out / "transcript.jsonl"), BENCH_CONTEXT=str(out / "context.json"), - BENCH_KNOWLEDGE=cell["knowledge"], BENCH_NETWORK=cell["network"], BENCH_WRIGHT=cell["wright"], + BENCH_AGENT_INFO=str(out / "agent-info.json"), + BENCH_KNOWLEDGE=cell["knowledge"], BENCH_NETWORK=cell["network"], BENCH_TOOL=cell["tool"], BENCH_SKILLS=",".join(cell["skills"]), ) - if cell["wright"] == "bin+skill": - env["BENCH_SKILL_DIR"] = str(args.skill_dir) + if cell["skills"]: + env["BENCH_SKILL_DIRS"] = os.pathsep.join(str(Path(args.skill_dirs[name]).resolve()) for name in cell["skills"]) return env +INSTRUCTION_FILES = ("AGENTS.md", "CLAUDE.md", "GEMINI.md", ".cursorrules", ".github/copilot-instructions.md", ".windsurf/rules", ".cursor/rules") + + +def ancestor_instructions(workspace: Path) -> list[str]: + """Instruction files an agent would discover by walking up from the workspace.""" + return [str(parent / name) for parent in workspace.resolve().parents for name in INSTRUCTION_FILES if (parent / name).exists()] + + def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) -> str | None: """A failed canary invalidates the run. Returns the reason, or None.""" - if cell["wright"] == "none" and shutil.which("wright", path=env["PATH"]): - return "wright reachable under wright level 'none'" + if args.check_ancestors and (found := ancestor_instructions(workspace)): + return f"instruction files in ancestor directories of the workspace: {found}; use --out outside the repository" + for tool in ("wright", "overpy"): + if cell["tool"] != tool and shutil.which(tool, path=env["PATH"]): + return f"{tool} reachable although the tool is '{cell['tool']}'" if cell["network"] == "off" and args.canary_cmd: if subprocess.run(args.canary_cmd, shell=True, cwd=workspace, env=env, capture_output=True).returncode == 0: return "network reachable under network 'off'" @@ -138,8 +207,22 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) -> tuple[int | None, str, str]: + command = args.agent_cmd + if getattr(args, "file_sandbox", False): + if sys.platform != "darwin" or not shutil.which("sandbox-exec"): + raise SystemExit("--file-sandbox requires macOS sandbox-exec; refusing an unprotected run") + run_dir = Path(env["BENCH_RUN_DIR"]) + temporary = run_dir / "tmp" + temporary.mkdir(exist_ok=True) + env = {**env, "TMPDIR": str(temporary), "PYTHONDONTWRITEBYTECODE": "1"} + profile = run_dir / "agent.sb" + profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' + f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' + '(deny file-read-data (require-all (regex "/(AGENTS|CLAUDE|GEMINI)[.]md$") ' + f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n') + command = ["sandbox-exec", "-f", str(profile), "/bin/sh", "-c", args.agent_cmd] proc = subprocess.Popen( - args.agent_cmd, shell=True, cwd=workspace, env=env, text=True, + command, shell=isinstance(command, str), cwd=workspace, env=env, text=True, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True, ) try: @@ -154,35 +237,40 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" -def context_report(out: Path, cell: dict) -> dict: +def context_report(out: Path, skill_names: list[str]) -> dict: path = out / "context.json" if not path.is_file(): return {"reported": False} - loaded = json.loads(path.read_text()).get("loaded", []) - allowed = {"wright"} if cell["wright"] == "bin+skill" else set() - return {"reported": True, "loaded": loaded, "unexpected": sorted(set(loaded) - allowed)} + report = json.loads(path.read_text()) + if "loaded" not in report: + return {"reported": False, **report} + loaded = report["loaded"] + allowed = set(skill_names or []) + return {**report, "reported": True, "loaded": loaded, "unexpected": sorted(set(loaded) - allowed)} def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) -> dict: check_cell(cell, args) + if not applicable(scenario, cell): + raise SystemExit(f"condition {cell_label(cell)} does not apply to the {scenario['language']} scenario {scenario['id']}") shutil.rmtree(out, ignore_errors=True) out.mkdir(parents=True) workspace = out / "workspace" - prompt = (scenario["dir"] / "prompt.md").read_text() + prompt = (scenario["dir"] / "prompt.md").read_text() # the exact pinned prompt: the harness adds no text infra_retries = 0 while True: shutil.rmtree(workspace, ignore_errors=True) - for stale in ("home", "bin", "wright-trace.jsonl", "wright-calls", "usage.jsonl", "transcript.jsonl", "context.json", "snapshots"): + for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots"): target = out / stale shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) materialize(scenario, workspace) if cell["knowledge"] == "wiki": - (workspace / "wiki").symlink_to(args.wiki_dir.resolve()) + shutil.copytree(Path(args.wiki_dir), workspace / "wiki") # a real copy: tools that skip symlinks (rg) must see it env = build_env(cell, args, out, workspace) reason = canaries(cell, env, workspace, args) if reason: result = base_result(scenario, cell, args, out, 0.0, None) - result.update(invalid=reason) + result.update(invalid=reason, status="invalid") (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") return result snapshots = bench_trace.Snapshots(workspace, scenario.get("watch", [scenario["entry"]]), out / "snapshots") @@ -198,13 +286,19 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") result = base_result(scenario, cell, args, out, seconds, agent_exit) result["infraRetries"] = infra_retries - context = context_report(out, cell) + result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only" + result["fileWriteEnforcement"] = "trial-directory-only" if getattr(args, "file_sandbox", False) else "unrestricted" + context = context_report(out, [skill_name(Path(args.skill_dirs[name])) for name in cell["skills"]]) result["context"] = context if context.get("unexpected"): result["invalid"] = f"unexpected loaded context: {context['unexpected']}" - events = bench_trace.read_events(out / "wright-trace.jsonl") - result["wrightUse"] = bench_trace.summarize_trace(events) + if (out / "agent-info.json").is_file(): + result["agentInfo"] = json.loads((out / "agent-info.json").read_text()) + events = bench_trace.read_events(out / "tool-trace.jsonl") + result["toolUse"] = bench_trace.summarize_trace(events) result.update(bench_grade.grade(scenario, workspace, args.wright, out / "grading")) + if any(c.get("unavailable") for c in result["checks"]): + result["invalid"] = "a required grader was unavailable" entry = workspace / scenario["entry"] final_sha = hashlib.sha256(entry.read_bytes()).hexdigest() if entry.is_file() else None result["friction"] = bench_trace.friction(events) @@ -212,10 +306,23 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - result["snapshots"] = snapshot_validity(scenario, snaps, args.wright, out) first_valid = next((s["t"] for s in result["snapshots"]["series"] if s["valid"]), None) result["usage"] = bench_trace.usage_summary(out / "usage.jsonl", first_valid) + result["status"] = run_status(result) (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") return result +def run_status(result: dict) -> str: + """How the run ended, separate from whether its artifact was usable.""" + if "invalid" in result: + return "invalid" + code = result["agent"]["exit"] + if code == INFRA_EXIT: + return "provider-interrupted" + if code is None: + return "timeout" + return "agent-error" if code else "completed" + + def snapshot_validity(scenario: dict, snaps: list[dict], wright: str, out: Path) -> dict: """Strict validity of every snapshot of the entry file; regressions are valid -> invalid transitions.""" series = [] @@ -226,6 +333,23 @@ def snapshot_validity(scenario: dict, snaps: list[dict], wright: str, out: Path) return {"count": len(series), "firstValidIndex": next((s["i"] for s in series if s["valid"]), None), "regressions": regressions, "series": series} +@functools.lru_cache(maxsize=None) +def harness_commit() -> str: + proc = subprocess.run(["git", "-C", str(HERE), "rev-parse", "HEAD"], capture_output=True, text=True) + dirty = subprocess.run(["git", "-C", str(HERE), "status", "--porcelain", "--", "."], capture_output=True, text=True).stdout.strip() + return proc.stdout.strip() + ("+dirty" if dirty else "") + + +@functools.lru_cache(maxsize=None) +def file_sha256(path: str) -> str: + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +@functools.lru_cache(maxsize=None) +def cached_suite(scenarios: str) -> tuple: + return tuple(bench_grade.suite_identity(Path(scenarios)).items()) + + def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, seconds: float, agent_exit: int | None) -> dict: return { "contract": RESULT_CONTRACT, @@ -233,12 +357,18 @@ def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, "family": scenario["family"], "language": scenario["language"], "split": scenario.get("split"), - "condition": dict(cell), + "condition": {**cell, "label": cell_label(cell)}, "agent": {"id": args.agent_id, "command": args.agent_cmd, "exit": agent_exit, "seconds": seconds}, + "protocol": {"timeoutSeconds": args.timeout, "infraRetries": args.infra_retries}, "environment": { "os": platform.platform(), "python": platform.python_version(), "wright": subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip(), + "wrightSha256": file_sha256(args.wright), + "harness": harness_commit(), + "suite": dict(cached_suite(str(SCENARIOS))), "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "skills": {name: skill_identity(name, Path(args.skill_dirs[name])) for name in cell["skills"]}, + **({"wiki": bench_wiki.identity(Path(args.wiki_dir))} if cell["knowledge"] == "wiki" else {}), }, } @@ -247,42 +377,81 @@ def trial_dir(base: Path, scenario_id: str, agent_id: str, cell: dict, trial: in return base / scenario_id / agent_id / f"{cell_label(cell).replace('/', '_')}-{trial}" +def verdict_word(result: dict) -> str: + return result["status"].upper() if result["status"] != "completed" else "PASS" if result["passed"] else "FAIL" + + def cmd_run(args: argparse.Namespace) -> int: scenario = load_scenario(args.scenario) - cell = {"wright": args.wright_level, "knowledge": args.knowledge, "network": args.network} - ok = True + cell = normalize_cell({"tool": args.tool, "skills": args.skills, "knowledge": args.knowledge, "network": args.network}) + code = 0 for trial in range(1, args.trials + 1): result = run_trial(scenario, cell, args, trial_dir(args.out, args.scenario, args.agent_id, cell, trial)) - ok &= bool(result.get("passed")) and "invalid" not in result - print(f"{args.scenario} {cell_label(cell)} trial {trial}: {'INVALID ' + result['invalid'] if 'invalid' in result else 'PASS' if result['passed'] else 'FAIL'}" - f" layers={result.get('failedLayers')} wright-invocations={result.get('wrightUse', {}).get('invocations')}") - return 0 if ok else 1 + used = sum(u["invocations"] for u in result.get("toolUse", {}).values()) + print(f"{args.scenario} {cell_label(cell)} trial {trial}: {verdict_word(result)} layers={result.get('failedLayers')} tool-invocations={used}") + code = max(code, 3 if result["status"] == "provider-interrupted" else 1 if result["status"] != "completed" or not result["passed"] else 0) + return code + + +STOP_AFTER_INTERRUPTIONS = 2 def cmd_matrix(args: argparse.Namespace) -> int: - """Run every (scenario, agent, cell, trial) of a matrix file in randomized order; finished runs are skipped.""" + """Run every applicable (scenario, agent, cell, trial) of a matrix file in randomized order; finished runs are skipped. + + Exit 0 when every run completed (graded failures are results, not errors), 3 when provider interruptions occurred, 1 otherwise. + After repeated provider interruptions in a row the remaining jobs are left unattempted instead of burning through them.""" config = json.loads(args.config.read_text()) - jobs = [ - (s, agent, cell, t) - for s in config.get("scenarios") or all_scenario_ids() - for agent in config["agents"] - for cell in config["cells"] - for t in range(1, config.get("trials", 5) + 1) - ] + cells = [normalize_cell(c) for c in config["cells"]] + jobs, skipped = [], [] + for scenario_id in config.get("scenarios") or all_scenario_ids(): + scenario = load_scenario(scenario_id) + for agent in config["agents"]: + for cell in cells: + if not applicable(scenario, cell): + skipped.append((scenario_id, cell_label(cell))) + continue + jobs += [(scenario_id, agent, cell, t) for t in range(1, config.get("trials", 5) + 1)] random.Random(config.get("seed", 0)).shuffle(jobs) + for scenario_id, label in sorted(set(skipped)): + print(f"not applicable: {scenario_id} {label}", flush=True) + state = {"streak": 0, "interrupted": 0, "unattempted": 0, "failed": 0} + lock = threading.Lock() def work(job: tuple) -> None: scenario_id, agent, cell, trial = job out = trial_dir(args.out, scenario_id, agent["id"], cell, trial) if (out / "result.json").is_file(): return - options = {k: Path(v) if k in ("skill_dir", "wiki_dir") and v else v for k, v in config.get("options", {}).items()} + with lock: + if state["streak"] >= STOP_AFTER_INTERRUPTIONS: + state["unattempted"] += 1 + return + options = {k: ({n: Path(v) for n, v in val.items()} if k == "skill_dirs" else Path(val) if k == "wiki_dir" and val else val) for k, val in config.get("options", {}).items()} trial_args = argparse.Namespace(**{**vars(args), "agent_id": agent["id"], "agent_cmd": agent["cmd"], **options}) result = run_trial(load_scenario(scenario_id), cell, trial_args, out) - print(f"{scenario_id} {agent['id']} {cell_label(cell)} #{trial}: {'INVALID' if 'invalid' in result else 'PASS' if result['passed'] else 'FAIL'}", flush=True) + with lock: + state["streak"] = state["streak"] + 1 if result["status"] == "provider-interrupted" else 0 + state["interrupted"] += result["status"] == "provider-interrupted" + state["failed"] += result["status"] in ("invalid", "agent-error") + print(f"{scenario_id} {agent['id']} {cell_label(cell)} #{trial}: {verdict_word(result)}", flush=True) with ThreadPoolExecutor(max_workers=config.get("parallel", 2)) as pool: list(pool.map(work, jobs)) + if state["unattempted"]: + print(f"stopped after {STOP_AFTER_INTERRUPTIONS} provider interruptions in a row: {state['unattempted']} job(s) left unattempted; rerun to resume", flush=True) + return 3 if state["interrupted"] or state["unattempted"] else 1 if state["failed"] else 0 + + +def cmd_wiki_snapshot(args: argparse.Namespace) -> int: + record = bench_wiki.snapshot(args.base, args.dir, tuple(args.categories)) + print(f"{len(record['documents'])} document(s) from {record['source']} into {args.dir}\nsnapshotSha256 {record['snapshotSha256']}") + return 0 + + +def cmd_wiki_skill(args: argparse.Namespace) -> int: + record = wiki_skill.build(args.snapshot, args.out_dir, json.loads(args.catalog.read_text()), json.loads(args.opy_manifest.read_text()), wiki_skill.upstream_source_text()) + print(json.dumps(record, indent=2)) return 0 @@ -298,40 +467,71 @@ def main() -> int: for name in ("validate", "run", "matrix"): p = sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") - p.add_argument("--out", type=Path, default=ROOT / "target/agent-bench") + p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") for name in ("run", "matrix"): p = sub.choices[name] - p.add_argument("--skill-dir", type=Path, help="pinned guide directory, exposed to the adapter as BENCH_SKILL_DIR") - p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked read-only as ./wiki for knowledge 'wiki'") + p.add_argument("--skill-dir", action="append", default=[], metavar="NAME=DIR", help=f"pinned skill directory for one of {', '.join(SKILLS)}; repeatable") + p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked as ./wiki for knowledge 'wiki'; content hashes are verified") p.add_argument("--env-pass", nargs="*", default=[], help="host variables passed through the environment scrub") + p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace") p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) + p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") run = sub.choices["run"] run.add_argument("scenario", choices=all_scenario_ids()) run.add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace, BENCH_* describes the condition") run.add_argument("--agent-id", required=True, help="recorded agent/model/version label") - run.add_argument("--wright-level", choices=WRIGHT_LEVELS, default="bin") + run.add_argument("--tool", choices=TOOLS, default="wright") + run.add_argument("--skills", nargs="*", choices=SKILLS, default=[], help="skills installed in this condition") run.add_argument("--knowledge", choices=KNOWLEDGE_LEVELS, default="none") run.add_argument("--network", choices=("off", "on"), default="off") run.add_argument("--trials", type=int, default=1) - sub.choices["matrix"].add_argument("config", type=Path, help="JSON: agents[{id,cmd}], cells[{wright,knowledge,network}], scenarios, trials, parallel, seed, options") + sub.choices["matrix"].add_argument("config", type=Path, help="JSON: agents[{id,cmd}], cells[{tool,skills,knowledge,network}], scenarios, trials, parallel, seed, options") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") + skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") + skill.add_argument("--snapshot", type=Path, required=True) + skill.add_argument("--out-dir", type=Path, required=True, help="new skill directory (not overwritten)") + skill.add_argument("--catalog", type=Path, required=True, help="workshop-rs catalog.json, for Workshop names and ids") + skill.add_argument("--opy-manifest", type=Path, required=True, help="opy-rs manifest.json, for upstream OverPy spellings") + wiki = sub.add_parser("wiki-snapshot", help="fetch the Workshop wiki Markdown mirror into a pinned local snapshot") + wiki.add_argument("--dir", type=Path, default=Path.home() / ".cache/wright-agent-bench-wiki") + wiki.add_argument("--base", default=bench_wiki.BASE) + wiki.add_argument("--categories", nargs="+", default=list(bench_wiki.CATEGORIES), help="wiki categories to crawl (add tutorials for the second tier)") report = sub.add_parser("report", help="summarize result.json files") report.add_argument("dirs", nargs="+", type=Path) report.add_argument("--regrade", action="store_true", help="re-grade stored workspaces twice and flag unstable graders") report.add_argument("--wright", default=str(ROOT / "target/debug/wright")) + report.add_argument("--reference", default=bench_report.BASELINE, help="condition label the paired comparison is made against") + score = sub.add_parser("score", help="compute the Wright Agent Score card of each language track from canonical test runs") + score.add_argument("dirs", nargs="+", type=Path) + score.add_argument("--language", choices=("workshop", "opy"), action="append", help="track to score; both when omitted") args = parser.parse_args() if hasattr(args, "wright"): args.wright = str(Path(args.wright).resolve()) if hasattr(args, "out"): args.out = args.out.resolve() + if hasattr(args, "skill_dir"): + args.skill_dirs = {} + for item in args.skill_dir: + name, _, directory = item.partition("=") + if name not in SKILLS or not directory: + raise SystemExit(f"--skill-dir expects NAME=DIR with NAME one of {', '.join(SKILLS)}: {item}") + args.skill_dirs[name] = Path(directory) if args.command == "validate": return 0 if validate(args.wright, args.out) else 1 if args.command == "setup-oracle": return cmd_setup_oracle(args) + if args.command == "wiki-snapshot": + return cmd_wiki_snapshot(args) + if args.command == "wiki-skill": + return cmd_wiki_skill(args) if args.command == "report": - return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s)) + return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), args.reference) + if args.command == "score": + languages = args.language or ["workshop", "opy"] + expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} + return bench_score.main(args.dirs, languages, expected, None) return cmd_run(args) if args.command == "run" else cmd_matrix(args) diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py index ec9a54a9..6e07a4fc 100644 --- a/benchmarks/agent/bench_grade.py +++ b/benchmarks/agent/bench_grade.py @@ -12,7 +12,8 @@ HERE = Path(__file__).resolve().parent ORACLE = HERE / "oracle" GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json") -UNSAFE_IGNORED = ("wiki",) +SUITE_VERSION = "v1" +UNSAFE_IGNORED = ("wiki", ".agents", ".devin") # linked wiki and skills installed through the agent's own mechanism def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: @@ -174,6 +175,33 @@ def grader_hash(scenario: dict) -> str: return digest.hexdigest() +def suite_identity(scenarios: Path) -> dict: + """Version and content hash of the whole suite: every scenario file, the grader sources, and the oracle lock.""" + digest = hashlib.sha256() + files = sorted(p for p in scenarios.rglob("*") if p.is_file()) + for path in files: + digest.update(str(path.relative_to(scenarios)).encode() + b"\0" + path.read_bytes()) + for name in GRADER_FILES: + path = HERE / name + digest.update(name.encode() + b"\0" + (path.read_bytes() if path.is_file() else b"")) + return {"version": SUITE_VERSION, "hash": digest.hexdigest(), "scenarios": sum(1 for p in scenarios.iterdir() if (p / "scenario.json").is_file())} + + +def usable_verdict(checks: list[dict], lint_errors: int, unsafe: list[str]) -> list[str]: + """Blocking conditions of `usable`. Empty means usable: all checks pass, no error-severity lint finding, no edit outside + the writable files, and every required grader was available.""" + blocking = [] + if not all(c["passed"] for c in checks): + blocking.append("checks-failed") + if lint_errors: + blocking.append("lint-error") + if unsafe: + blocking.append("unsafe-edits") + if any(c.get("unavailable") for c in checks): + blocking.append("grader-unavailable") + return blocking + + def grade(scenario: dict, workspace: Path, wright: str, scratch: Path | None = None) -> dict: entry = workspace / scenario["entry"] scratch = scratch or workspace.parent / f"{workspace.name}-grading" @@ -184,17 +212,20 @@ def grade(scenario: dict, workspace: Path, wright: str, scratch: Path | None = N checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] lint_errors = sum(1 for f in state["lint"] if f.get("severity") == "error") passed = all(c["passed"] for c in checks) + unsafe = unsafe_edits(scenario, workspace) + blocking = usable_verdict(checks, lint_errors, unsafe) auth = state["authorities"] return { "checks": checks, "passed": passed, - "usable": passed and lint_errors == 0, + "usable": not blocking, + "usableReason": blocking, "failedLayers": sorted({c["layer"] for c in checks if not c["passed"]}), "diagnostics": state.get("diagnostics", []), "authorities": {k: v for k, v in auth.items() if k != "disagreement"}, "disagreement": auth.get("disagreement"), "lintFindings": sorted({f["code"] for f in state["lint"]}), - "unsafeEdits": unsafe_edits(scenario, workspace), + "unsafeEdits": unsafe, "unverifiedRuntimeClaims": scenario.get("runtimeOnly", []), "grader": {"hash": grader_hash(scenario), "oracle": oracle_version()}, } diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py index b5f1e795..162de014 100644 --- a/benchmarks/agent/bench_report.py +++ b/benchmarks/agent/bench_report.py @@ -24,8 +24,7 @@ def wilson(k: int, n: int, z: float = 1.96) -> tuple[float, float]: def label(result: dict) -> str: - c = result["condition"] - return f"{c['wright']}/{c['knowledge']}/{c['network']}" + return result["condition"]["label"] def load(dirs: list[Path]) -> list[dict]: @@ -46,6 +45,20 @@ def rate(k: int, n: int) -> str: return f"{k}/{n} [{lo:.2f}-{hi:.2f}]" +def rate_runs(runs: list[dict]) -> str: + """Usable rate of a group. With two or more scenarios the interval is a clustered bootstrap, because trials of one scenario + are correlated; a single scenario falls back to Wilson.""" + k, n = sum(1 for r in runs if r.get("usable")), len(runs) + per: dict[str, list[int]] = defaultdict(list) + for r in runs: + per[r["scenario"]].append(1 if r.get("usable") else 0) + if len(per) < 2: + return rate(k, n) + from bench_score import cluster_interval + lo, hi = cluster_interval(per) + return f"{k}/{n} [{lo / 100:.2f}-{hi / 100:.2f}]" + + def total_tokens(result: dict) -> int | None: return (result.get("usage") or {}).get("totalTokens") @@ -60,7 +73,7 @@ def group_rows(runs: list[dict]) -> dict: "tokensPerUsable": (sum(tokens) / len(usable)) if tokens and usable and len(tokens) == len(runs) else None, "peakContext": mean(peaks) if peaks else None, "seconds": mean(r["agent"]["seconds"] for r in runs), - "usedWright": sum(1 for r in runs if r.get("wrightUse", {}).get("invocations")), + "usedTool": sum(1 for r in runs if any(u["invocations"] for u in r.get("toolUse", {}).values())), } @@ -68,20 +81,20 @@ def fmt(value, digits: int = 0) -> str: return "n/a" if value is None else f"{value:,.{digits}f}" -def paired(runs: list[dict]) -> list[str]: +def paired(runs: list[dict], reference: str = BASELINE) -> list[str]: by_key: dict[tuple, dict] = {(r["scenario"], r["agent"]["id"], r["_trial"], label(r)): r for r in runs} lines = [] - cells = sorted({label(r) for r in runs} - {BASELINE}) + cells = sorted({label(r) for r in runs} - {reference}) for agent in sorted({r["agent"]["id"] for r in runs}): for cell in cells: - pairs = [(by_key[(s, a, t, BASELINE)], r) for (s, a, t, c), r in by_key.items() if a == agent and c == cell and (s, a, t, BASELINE) in by_key] + pairs = [(by_key[(s, a, t, reference)], r) for (s, a, t, c), r in by_key.items() if a == agent and c == cell and (s, a, t, reference) in by_key] if not pairs: continue gain = sum(1 for b, r in pairs if r.get("usable") and not b.get("usable")) loss = sum(1 for b, r in pairs if b.get("usable") and not r.get("usable")) both = [(total_tokens(b), total_tokens(r)) for b, r in pairs if b.get("usable") and r.get("usable") and total_tokens(b) and total_tokens(r)] saving = f"{mean(1 - r / b for b, r in both):+.0%} tokens (n={len(both)})" if both else "no both-usable pairs" - lines.append(f"| {agent} | {cell} vs {BASELINE} | {len(pairs)} | +{gain} / -{loss} | {saving} |") + lines.append(f"| {agent} | {cell} vs {reference} | {len(pairs)} | +{gain} / -{loss} | {saving} |") return lines @@ -105,7 +118,7 @@ def diagnostics(runs: list[dict], invalid: list[dict]) -> list[str]: base = [r for r in runs if label(r) == BASELINE] if base and sum(1 for r in base if r.get("usable")) / len(base) >= HEADROOM: notes.append(f"HEADROOM: baseline `{BASELINE}` usable rate is at least {HEADROOM:.0%}; the tasks cannot show a gain.") - infra = [r for r in runs if r["agent"]["exit"] != 0 or r.get("infraRetries")] + infra = [r for r in runs if r["status"] in ("agent-error", "timeout") or r.get("infraRetries")] if infra: notes.append(f"INFRASTRUCTURE: {len(infra)} run(s) exited non-zero, timed out, or needed an infrastructure retry.") if invalid: @@ -134,12 +147,13 @@ def regrade_notes(runs: list[dict], wright: str, load_scenario) -> list[str]: return [f"GRADER: unstable verdict on {len(unstable)} workspace(s): {unstable}"] if unstable else ["GRADER: consistent on every regraded workspace."] -def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, dict]: - invalid = [r for r in results if "invalid" in r] - runs = [r for r in results if "invalid" not in r] - out = ["# Agent benchmark report", "", f"{len(runs)} valid run(s), {len(invalid)} invalid.", ""] - summary: dict = {"cells": {}} - out += ["## Outcome by agent and condition", "", "| agent | condition | runs | usable | passed | used wright | tokens/run | tokens per usable | peak context | s/run |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] +def render(results: list[dict], regrade: list[str] | None = None, reference: str = BASELINE) -> tuple[str, dict]: + invalid = [r for r in results if r["status"] == "invalid"] + infrastructure = [r for r in results if r["status"] == "provider-interrupted"] + runs = [r for r in results if r["status"] not in ("invalid", "provider-interrupted")] + out = ["# Agent benchmark report", "", f"{len(runs)} valid run(s), {len(invalid)} invalid, {len(infrastructure)} infrastructure failures excluded.", ""] + summary: dict = {"cells": {}, "infrastructureFailures": len(infrastructure)} + out += ["## Outcome by agent and condition", "", "| agent | condition | runs | usable | passed | used a tool | tokens/run | tokens per usable | peak context | s/run |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] for agent in sorted({r["agent"]["id"] for r in runs}): for cell in sorted({label(r) for r in runs}): group = [r for r in runs if r["agent"]["id"] == agent and label(r) == cell] @@ -147,7 +161,7 @@ def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, continue row = group_rows(group) summary["cells"][f"{agent}|{cell}"] = row - out.append(f"| {agent} | {cell} | {row['n']} | {rate(row['usable'], row['n'])} | {row['passed']}/{row['n']} | {row['usedWright']}/{row['n']} | " + out.append(f"| {agent} | {cell} | {row['n']} | {rate_runs(group)} | {row['passed']}/{row['n']} | {row['usedTool']}/{row['n']} | " f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} |") out += ["", "## By scenario", "", "| scenario | agent | condition | usable |", "| --- | --- | --- | --- |"] groups: dict[tuple, list[dict]] = defaultdict(list) @@ -155,6 +169,12 @@ def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, groups[(r["scenario"], r["agent"]["id"], label(r))].append(r) for (scenario, agent, cell), g in sorted(groups.items()): out.append(f"| {scenario} | {agent} | {cell} | {rate(sum(1 for r in g if r.get('usable')), len(g))} |") + out += ["", "## By language", "", "| language | condition | usable |", "| --- | --- | --- |"] + for language in sorted({r["language"] for r in runs}): + for cell in sorted({label(r) for r in runs}): + g = [r for r in runs if r["language"] == language and label(r) == cell] + if g: + out.append(f"| {language} | {cell} | {rate_runs(g)} |") splits = sorted({r["split"] for r in runs if r.get("split")}) if splits: out += ["", "## By split", "", "| split | condition | usable |", "| --- | --- | --- |"] @@ -162,10 +182,10 @@ def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, for cell in sorted({label(r) for r in runs}): g = [r for r in runs if r.get("split") == split and label(r) == cell] if g: - out.append(f"| {split} | {cell} | {rate(sum(1 for r in g if r.get('usable')), len(g))} |") - pairs = paired(runs) + out.append(f"| {split} | {cell} | {rate_runs(g)} |") + pairs = paired(runs, reference) if pairs: - out += ["", f"## Paired against `{BASELINE}` (same scenario, agent, trial)", "", "| agent | comparison | pairs | usable gained/lost | tokens where both usable |", "| --- | --- | --- | --- | --- |", *pairs] + out += ["", f"## Paired against `{reference}` (same scenario, agent, trial)", "", "| agent | comparison | pairs | usable gained/lost | tokens where both usable |", "| --- | --- | --- | --- | --- |", *pairs] exp = expectations(runs) if exp: out += ["", "## Expectation rates (pass/(pass+fail); n/a and unavailable excluded)", "", "| condition | expectations |", "| --- | --- |", *exp] @@ -184,25 +204,33 @@ def render(results: list[dict], regrade: list[str] | None = None) -> tuple[str, out += [f"| {cell} | {' | '.join(str(v.get(k, 0)) for k in keys)} |" for cell, v in sorted(fr.items())] tok: dict[str, dict[str, list[int]]] = defaultdict(lambda: defaultdict(list)) for r in runs: - for cmd, t in (r.get("wrightUse", {}).get("outputTokensEstimate") or {}).items(): - tok[cmd]["tokens"].append(t) + for tool, use in (r.get("toolUse") or {}).items(): + for cmd, t in (use.get("outputTokensEstimate") or {}).items(): + tok[f"{tool} {cmd}"]["tokens"].append(t) if tok: - out += ["", "## Wright output size per command (estimated tokens per run that used it)", "", "| command | runs | mean | max |", "| --- | --- | --- | --- |"] + out += ["", "## Tool output size per command (estimated tokens per run that used it)", "", "| command | runs | mean | max |", "| --- | --- | --- | --- |"] out += [f"| {cmd} | {len(v['tokens'])} | {mean(v['tokens']):.0f} | {max(v['tokens'])} |" for cmd, v in sorted(tok.items())] notes = diagnostics(runs, invalid) + (regrade or []) + if infrastructure: + notes.append(f"INFRASTRUCTURE: {len(infrastructure)} provider/infrastructure failures (exit 75) excluded from outcome metrics.") + out += ["", "## Provider/infrastructure failures", "", "| agent | scenario | condition | artifact |", "| --- | --- | --- | --- |"] + out += [f"| {r['agent']['id']} | {r['scenario']} | {label(r)} | {r['_dir']} |" for r in infrastructure] + missing_context = sum(1 for r in runs if not (r.get("context") or {}).get("reported")) + if missing_context: + notes.append(f"CONTEXT: {missing_context} run(s) lack observed loaded-context data; installed skills are not proof of loading.") out += ["", "## Diagnostics", ""] + ([f"- {n}" for n in notes] or ["- none"]) summary["diagnostics"] = notes summary["stdev"] = {k: pstdev([r["agent"]["seconds"] for r in runs if f"{r['agent']['id']}|{label(r)}" == k]) for k in summary["cells"]} if runs else {} return "\n".join(out) + "\n", summary -def main(dirs: list[Path], wright: str, regrade: bool, load_scenario) -> int: +def main(dirs: list[Path], wright: str, regrade: bool, load_scenario, reference: str = BASELINE) -> int: results = load(dirs) if not results: print("no results found") return 1 - notes = regrade_notes([r for r in results if "invalid" not in r], wright, load_scenario) if regrade else None - text, summary = render(results, notes) + notes = regrade_notes([r for r in results if r["status"] != "invalid"], wright, load_scenario) if regrade else None + text, summary = render(results, notes, reference) (dirs[0] / "report.md").write_text(text) (dirs[0] / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") print(text) diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py new file mode 100644 index 00000000..7e1e140b --- /dev/null +++ b/benchmarks/agent/bench_score.py @@ -0,0 +1,153 @@ +"""Wright Agent Score (#467): per-language tracks, scenario macro-average, clustered bootstrap interval, and the score card.""" + +from __future__ import annotations + +import json +import random +from collections import defaultdict +from math import comb +from pathlib import Path + +CONTRACT = "wright-agent-score/v1" +CANONICAL = "wright+wright-skill/none/off" +MIN_SCENARIOS = 8 +BOOTSTRAP_DRAWS = 10_000 +BOOTSTRAP_SEED = 467 +METHOD = f"two-stage percentile bootstrap (scenarios, then trials within a scenario), {BOOTSTRAP_DRAWS} draws, seed {BOOTSTRAP_SEED}" +EXCLUDED = ("provider-interrupted", "invalid", "agent-error") # reported separately; a timeout is the agent's own outcome and counts +TRACKS = {"workshop": "Wright Workshop Agent Score", "opy": "Wright OPY Agent Score"} + + +def cluster_interval(outcomes: dict[str, list[int]], draws: int = BOOTSTRAP_DRAWS, seed: int = BOOTSTRAP_SEED) -> tuple[float, float]: + """95% interval of the macro-averaged usable rate, resampling scenarios first and trials within each chosen scenario.""" + names = sorted(outcomes) + rng = random.Random(seed) + means = [] + for _ in range(draws): + total = 0.0 + for name in (rng.choice(names) for _ in names): + trials = outcomes[name] + total += sum(rng.choice(trials) for _ in trials) / len(trials) + means.append(100 * total / len(names)) + means.sort() + return means[int(0.025 * draws)], means[int(0.975 * draws) - 1] + + +def pass_power_k(usable: int, valid: int, k: int) -> float: + """Probability that k distinct valid trials of a scenario are all usable.""" + return comb(usable, k) / comb(valid, k) if valid >= k else 0.0 + + +def identity_of(run: dict) -> dict: + env = run["environment"] + info = run.get("agentInfo") or {} + return { + "wrightSha256": env.get("wrightSha256"), "wright": env.get("wright"), + "skills": {name: s["sha256"] for name, s in sorted((env.get("skills") or {}).items())}, + "suite": (env.get("suite") or {}).get("hash"), + "agent": run["agent"]["id"], "model": info.get("model"), "effort": info.get("effort"), + "protocol": run.get("protocol"), + } + + +def card(results: list[dict], language: str, expected: list[str]) -> dict: + """The score card of one language track, or a refusal when the runs are not one comparable environment.""" + track = [r for r in results if r["language"] == language and r["condition"]["label"] == CANONICAL and r.get("split") == "test"] + excluded = defaultdict(list) + valid = [] + for run in track: + (excluded[run["status"]] if run["status"] in EXCLUDED else valid).append(run) + if not valid: + return {"contract": CONTRACT, "track": TRACKS[language], "refused": "no valid canonical test runs for this language"} + identities = {json.dumps(identity_of(r), sort_keys=True) for r in valid} + if len(identities) > 1: + differing = sorted({k for a in map(json.loads, identities) for b in map(json.loads, identities) for k in a if a[k] != b[k]}) + return {"contract": CONTRACT, "track": TRACKS[language], "refused": f"runs come from more than one environment; they differ in: {', '.join(differing)}"} + identity = json.loads(next(iter(identities))) + per: dict[str, list[int]] = defaultdict(list) + by_scenario: dict[str, list[dict]] = defaultdict(list) + for run in valid: + per[run["scenario"]].append(1 if run.get("usable") else 0) + by_scenario[run["scenario"]].append(run) + counts = {len(v) for v in per.values()} + provisional = [] + missing = sorted(set(expected) - set(per)) + if missing: + provisional.append(f"missing held-out scenarios: {', '.join(missing)}") + if len(counts) > 1: + provisional.append(f"unequal valid trials per scenario: {sorted(counts)}") + if len(expected) < MIN_SCENARIOS: + provisional.append(f"fewer than {MIN_SCENARIOS} held-out scenarios in the suite ({len(expected)})") + k = min(counts) + rates = {s: sum(v) / len(v) for s, v in per.items()} + lo, hi = cluster_interval(per) + families: dict[str, list[float]] = defaultdict(list) + layers: dict[str, int] = defaultdict(int) + for s, runs in by_scenario.items(): + families[runs[0]["family"]].append(rates[s]) + for run in runs: + for layer in run.get("failedLayers") or []: + layers[layer] += 1 + tokens = [(r.get("usage") or {}).get("totalTokens") for r in valid] + usable_tokens = [t for t, r in zip(tokens, valid) if r.get("usable") and t] + return { + "contract": CONTRACT, "track": TRACKS[language], "scoreVersion": "v1", "language": language, + "score": round(100 * sum(rates.values()) / len(rates), 1), "ci95": [round(lo, 1), round(hi, 1)], "ciMethod": METHOD, + "trialsPerScenario": k, "scenarios": len(per), "validRuns": len(valid), + "passPowK": round(100 * sum(pass_power_k(sum(v), len(v), k) for v in per.values()) / len(per), 1), + "provisional": provisional, + "identity": identity, + "suite": (valid[0]["environment"].get("suite") or {}), + "harness": sorted({r["environment"].get("harness") for r in valid if r["environment"].get("harness")}), + "networkEnforcement": sorted({r.get("networkEnforcement") for r in valid}), + "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") for r in valid}), + "exclusions": {status: len(runs) for status, runs in sorted(excluded.items())}, + "perScenario": [{"scenario": s, "usable": sum(v), "valid": len(v), "rate": round(rates[s], 3)} for s, v in sorted(per.items())], + "byFamily": {f: round(100 * sum(v) / len(v), 1) for f, v in sorted(families.items())}, + "failureLayers": dict(sorted(layers.items())), + "secondary": { + "tokensPerRun": round(sum(t for t in tokens if t) / max(1, sum(1 for t in tokens if t))) if any(tokens) else None, + "tokensPerUsableResult": round(sum(t for t in tokens if t) / len(usable_tokens)) if usable_tokens else None, + "secondsPerRun": round(sum(r["agent"]["seconds"] for r in valid) / len(valid), 1), + }, + } + + +def render(c: dict) -> str: + if "refused" in c: + return f"{c['track']}: no score. {c['refused']}\n" + ident = c["identity"] + lines = [ + f"{c['track']} {c['scoreVersion']}", "", + f"Agent: {ident['agent']}", f"Model: {ident['model'] or 'not recorded'}", f"Inference: effort {ident['effort'] or 'not recorded'}", "", + f"Score: {c['score']}", f"95% CI: {c['ci95'][0]}-{c['ci95'][1]} ({c['ciMethod']})", + f"Pass^{c['trialsPerScenario']}: {c['passPowK']} (secondary)", + f"Trials: {c['trialsPerScenario']} per scenario ({c['validRuns']} valid runs)", f"Scenarios: {c['scenarios']} held-out", "", + f"Suite: {c['suite'].get('version')} {c['suite'].get('hash')}", f"Wright: {ident['wright']} sha256 {ident['wrightSha256']}", + f"Skills: {', '.join(f'{n} {h}' for n, h in ident['skills'].items()) or 'none'}", f"Harness: {', '.join(c['harness']) or 'not recorded'}", + f"Network: {', '.join(x or 'not recorded' for x in c['networkEnforcement'])}", f"File writes: {', '.join(x or 'not recorded' for x in c['fileWriteEnforcement'])}", + f"Excluded: {c['exclusions'] or 'none'}", + ] + if c["provisional"]: + lines += ["", "PROVISIONAL: " + "; ".join(c["provisional"])] + if "declared-only" in c["networkEnforcement"]: + lines += ["", "Network was off by declaration only for some runs; the canonical condition asks for it to be disabled."] + lines += ["", "By scenario: " + ", ".join(f"{p['scenario']} {p['usable']}/{p['valid']}" for p in c["perScenario"]), "By family: " + ", ".join(f"{f} {v}" for f, v in c["byFamily"].items()), + "Failure layers: " + (", ".join(f"{k} {v}" for k, v in c["failureLayers"].items()) or "none")] + return "\n".join(lines) + "\n" + + +def main(dirs: list[Path], languages: list[str], expected_by_language: dict[str, list[str]], out: Path | None) -> int: + from bench_report import load + results = load(dirs) + status = 0 + cards = [] + for language in languages: + c = card(results, language, expected_by_language[language]) + cards.append(c) + print(render(c)) + status = max(status, 2 if "refused" in c else 0) + target = out or dirs[0] + (target / "score.json").write_text(json.dumps({"contract": CONTRACT, "cards": cards}, indent=2) + "\n") + (target / "score.txt").write_text("\n".join(render(c) for c in cards)) + return status diff --git a/benchmarks/agent/bench_trace.py b/benchmarks/agent/bench_trace.py index 3ffe417b..a46c3f8e 100644 --- a/benchmarks/agent/bench_trace.py +++ b/benchmarks/agent/bench_trace.py @@ -39,8 +39,8 @@ def envelope_summary(envelope: dict) -> dict: def append_event(event: dict) -> None: - with open(os.environ["WRIGHT_BENCH_TRACE"], "a") as trace: - trace.write(json.dumps(event) + "\n") + with open(os.environ["BENCH_TOOL_TRACE"], "a") as trace: + trace.write(json.dumps({"tool": os.environ["BENCH_TOOL_NAME"], **event}) + "\n") def sidecar_name() -> str: @@ -48,10 +48,12 @@ def sidecar_name() -> str: def shim_main(argv: list[str]) -> int: - """Run the real Wright and record the call. `serve` sessions are teed line by line.""" - real = os.environ["WRIGHT_BENCH_REAL"] + """Run the real tool (`wright` or `overpy`, named by argv[0]) and record the call. Wright `serve` sessions are teed line by line.""" + tool, argv = argv[0], argv[1:] + os.environ["BENCH_TOOL_NAME"] = tool + real = os.environ[f"BENCH_TOOL_REAL_{tool.upper()}"] started = time.time() - if command_of(argv) == "serve": + if tool == "wright" and command_of(argv) == "serve": return serve_tee(real, argv, started) proc = subprocess.run([real, *argv], capture_output=True) sys.stdout.buffer.write(proc.stdout) @@ -59,12 +61,12 @@ def shim_main(argv: list[str]) -> int: sys.stderr.buffer.write(proc.stderr) sys.stderr.flush() name = sidecar_name() - sidecar = Path(os.environ["WRIGHT_BENCH_SIDECAR"]) + sidecar = Path(os.environ["BENCH_TOOL_SIDECAR"]) sidecar.mkdir(parents=True, exist_ok=True) (sidecar / f"{name}.out").write_bytes(proc.stdout) (sidecar / f"{name}.err").write_bytes(proc.stderr) envelope = None - if wants_json(argv): + if tool == "wright" and wants_json(argv): try: envelope = envelope_summary(json.loads(proc.stdout)) except json.JSONDecodeError: @@ -83,7 +85,7 @@ def serve_tee(real: str, argv: list[str], started: float) -> int: def log(direction: str, line: bytes) -> None: counts[direction] += 1 - append_event({"type": "serve", "dir": direction, "t": time.time(), "line": line.decode(errors="replace")[:2000]}) + append_event({"type": "serve", "dir": direction, "t": time.time(), "line": line.decode(errors="replace")}) def pump() -> None: for line in proc.stdout: @@ -146,7 +148,16 @@ def finish(self) -> list[dict]: return self.events +def tool_events(events: list[dict], tool: str) -> list[dict]: + return [e for e in events if e.get("tool") == tool] + + def summarize_trace(events: list[dict]) -> dict: + """Per tool: invocations, failures, and output size, for every tool that was called.""" + return {tool: summarize_tool(tool_events(events, tool)) for tool in sorted({e["tool"] for e in events if "tool" in e})} + + +def summarize_tool(events: list[dict]) -> dict: calls = [e for e in events if e["type"] == "call"] by_command: dict[str, int] = {} for call in calls: @@ -164,6 +175,8 @@ def summarize_trace(events: list[dict]) -> dict: def friction(events: list[dict]) -> dict: + """Wright friction; other tools are summarized by `summarize_trace` only.""" + events = tool_events(events, "wright") calls = [e for e in events if e["type"] == "call"] seen: list[tuple] = [] repeats = 0 @@ -171,13 +184,21 @@ def friction(events: list[dict]) -> dict: key = tuple(call["argv"]) repeats += key in seen seen.append(key) - serve_responses = [json.loads(e["line"]) for e in events if e["type"] == "serve" and e["dir"] == "res" and e["line"].startswith("{")] + serve_responses = [] + unparsed = 0 + for event in events: + if event["type"] == "serve" and event["dir"] == "res": + try: + serve_responses.append(json.loads(event["line"])) + except json.JSONDecodeError: + unparsed += 1 return { "usageErrors": sum(1 for c in calls if c["exit"] == 2), "unknownSubcommands": sum(1 for c in calls if "unrecognized subcommand" in c.get("stderrHead", "")), "helpLookups": sum(1 for c in calls if any(a in ("--help", "-h", "help") for a in c["argv"])), "retriesAfterUnsupported": sum(1 for i, c in enumerate(calls) if c["exit"] >= 3 and tuple(c["argv"]) in [tuple(x["argv"]) for x in calls[i + 1:]]), "malformedServeRequests": sum(1 for r in serve_responses if r.get("error", {}).get("code") == "malformed-request"), + "unparsedServeResponses": unparsed, "identicalRepeats": repeats, "callsToFirstSuccess": next((i + 1 for i, c in enumerate(calls) if c["exit"] == 0 and not any(a in ("--help", "-h", "--version") for a in c["argv"])), None), } @@ -200,6 +221,7 @@ def expectation(status: str, detail: str = "") -> dict: def detect_expectations(events: list[dict], snapshots: list[dict], scenario: dict, final_sha256: str | None) -> dict: """SPEC-414 E01-E12 over the Wright trace. E05, E07, E09, E10 need the agent transcript: `unavailable`.""" + events = tool_events(events, "wright") calls = [e for e in events if e["type"] == "call"] ops = serve_ops(events) used = bool(calls) diff --git a/benchmarks/agent/bench_wiki.py b/benchmarks/agent/bench_wiki.py new file mode 100644 index 00000000..a2dc4684 --- /dev/null +++ b/benchmarks/agent/bench_wiki.py @@ -0,0 +1,113 @@ +"""Snapshot of the Workshop wiki Markdown mirror for the benchmark's `wiki` knowledge level (#414, SPEC-414).""" + +from __future__ import annotations + +import hashlib +import json +import re +import subprocess +import time +from concurrent.futures import ThreadPoolExecutor +from datetime import datetime, timezone +from pathlib import Path + +BASE = "https://md.wrightkit.dev" +USER_AGENT = "wright-agent-bench/1 (+https://github.com/wrightkit/wright)" +SLUG = re.compile(r"^[A-Za-z0-9_-]+$") +NOTICE = ( + "Workshop.codes wiki content, rendered to Markdown by the mirror recorded in SNAPSHOT.json.\n" + "Use and redistribution follow the Workshop.codes Terms of Service (https://workshop.codes/tos).\n" + "This snapshot is for local benchmark runs; do not commit or redistribute it.\n" +) + + +def fetch(url: str, attempts: int = 4) -> bytes: + """GET through curl: the mirror rejects Python's HTTP client fingerprint with 403. Retries slow or failed requests.""" + error = "" + for attempt in range(attempts): + proc = subprocess.run(["curl", "-fsSL", "-m", "60", "-A", USER_AGENT, url], capture_output=True) + if proc.returncode == 0 and proc.stdout: + return proc.stdout + error = proc.stderr.decode(errors="replace").strip() or "empty response" + time.sleep(2 * (attempt + 1)) + raise SystemExit(f"fetch failed for {url}: {error}") + + +CATEGORIES = ("actions", "values", "events", "constants", "references") +FRONT = re.compile(r"\A---\n(.*?)\n---\n", re.S) + + +def front_matter(markdown: bytes) -> dict: + match = FRONT.match(markdown.decode(errors="replace")) + fields = {} + for line in (match.group(1).splitlines() if match else []): + key, _, value = line.partition(": ") + fields[key.strip()] = value.strip().strip('"') + return fields + + +def category_slugs(base: str, category: str) -> list[str]: + """Article slugs of one category page. The manifest lists only the first upstream page, so categories are the source.""" + page = fetch(f"{base}/wiki/categories/{category}").decode() + return list(dict.fromkeys(re.findall(r"\]\([^)]*/wiki/articles/([^)]+)\)", page))) + + +def snapshot(base: str, out: Path, categories: tuple[str, ...] = CATEGORIES, delay: float = 0.1, workers: int = 4) -> dict: + """Crawl the given categories once and write the files plus SNAPSHOT.json with content hashes.""" + if (out / "SNAPSHOT.json").exists(): + raise SystemExit(f"{out} already holds a snapshot; snapshots are pinned, so choose a new directory") + articles = out / "articles" + articles.mkdir(parents=True, exist_ok=True) + slugs: dict[str, list[str]] = {} + for category in categories: + listing = category_slugs(base, category) + if not listing: + raise SystemExit(f"category {category!r} listed no articles") + for slug in listing: + if not SLUG.match(slug): + raise SystemExit(f"refusing unsafe slug {slug!r}") + slugs.setdefault(slug, []).append(category) + + def one(item: tuple[str, list[str]]) -> dict: + slug, cats = item + target = articles / f"{slug}.md" + body = target.read_bytes() if target.is_file() and target.stat().st_size else fetch(f"{base}/wiki/articles/{slug}") # resumes an interrupted crawl + target.write_bytes(body) + meta = front_matter(body) + time.sleep(delay) + return { + "slug": slug, "categories": cats, "title": meta.get("title"), "updatedAt": meta.get("updated_at"), + "contentHash": meta.get("content_hash"), "sha256": hashlib.sha256(body).hexdigest(), + } + + with ThreadPoolExecutor(max_workers=workers) as pool: + documents = list(pool.map(one, slugs.items())) + (out / "NOTICE.txt").write_text(NOTICE) + documents.sort(key=lambda d: d["slug"]) + identity_text = "\n".join(f"{d['slug']} {d['sha256']}" for d in documents) + record = { + "source": base, "fetchedAt": datetime.now(timezone.utc).isoformat(timespec="seconds"), "categories": list(categories), + "documents": documents, "snapshotSha256": hashlib.sha256(identity_text.encode()).hexdigest(), + } + (out / "SNAPSHOT.json").write_text(json.dumps(record, indent=2) + "\n") + return record + + +def load_snapshot(wiki_dir: Path) -> dict: + record = json.loads((wiki_dir / "SNAPSHOT.json").read_text()) + for doc in record["documents"]: + if not SLUG.fullmatch(doc["slug"]): + raise SystemExit(f"refusing unsafe slug {doc['slug']!r}") + path = wiki_dir / "articles" / f"{doc['slug']}.md" + if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest() != doc["sha256"]: + raise SystemExit(f"snapshot content mismatch: {path}") + identity_text = "\n".join(f"{d['slug']} {d['sha256']}" for d in sorted(record["documents"], key=lambda d: d["slug"])) + if hashlib.sha256(identity_text.encode()).hexdigest() != record["snapshotSha256"]: + raise SystemExit(f"snapshot identity mismatch: {wiki_dir}") + return record + + +def identity(wiki_dir: Path) -> dict: + """The verified identity of a snapshot, recorded in every result that used it.""" + record = load_snapshot(wiki_dir) + return {k: record[k] for k in ("source", "fetchedAt", "snapshotSha256")} | {"documents": len(record["documents"])} diff --git a/benchmarks/agent/matrix.example.json b/benchmarks/agent/matrix.example.json index c99d62d9..bd7f9da2 100644 --- a/benchmarks/agent/matrix.example.json +++ b/benchmarks/agent/matrix.example.json @@ -1,16 +1,71 @@ { "agents": [ - {"id": "example-model", "cmd": "BENCH_MODEL=example-model python3 benchmarks/agent/adapters/claude_code.py"} + { + "id": "pi-gpt-6-luna", + "cmd": "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/wright/benchmarks/agent/adapters/pi.py" + }, + { + "id": "pi-gemini-3.6-flash", + "cmd": "BENCH_MODEL=antigravity/gemini-3.6-flash BENCH_PI_EXTENSIONS=/abs/path/pi-antigravity BENCH_PI_WEB_EXTENSIONS=/abs/path/pi-web-access python3 /abs/path/wright/benchmarks/agent/adapters/pi.py" + }, + { + "id": "devin-swe-2-max", + "cmd": "BENCH_MODEL=swe-2-max python3 /abs/path/wright/benchmarks/agent/adapters/devin.py" + } ], "cells": [ - {"wright": "none", "knowledge": "none", "network": "off"}, - {"wright": "bin", "knowledge": "none", "network": "off"}, - {"wright": "bin+skill", "knowledge": "none", "network": "off"}, - {"wright": "none", "knowledge": "web", "network": "on"}, - {"wright": "none", "knowledge": "wiki", "network": "off"} + { + "tool": "none", + "skills": [], + "knowledge": "none", + "network": "off" + }, + { + "tool": "wright", + "skills": [], + "knowledge": "none", + "network": "off" + }, + { + "tool": "wright", + "skills": [ + "wright-skill" + ], + "knowledge": "none", + "network": "off" + }, + { + "tool": "none", + "skills": [], + "knowledge": "web", + "network": "on" + }, + { + "tool": "none", + "skills": [], + "knowledge": "wiki", + "network": "off" + }, + { + "tool": "none", + "skills": [ + "workshop-skill" + ], + "knowledge": "none", + "network": "off" + } ], "trials": 5, "parallel": 2, "seed": 1, - "options": {"skill_dir": "../skills/skills/wright", "wiki_dir": "path/to/pinned/wiki", "env_pass": ["HOME", "ANTHROPIC_API_KEY"]} + "options": { + "wiki_dir": "/abs/path/pinned-wiki", + "env_pass": [ + "HOME" + ], + "skill_dirs": { + "wright-skill": "/abs/path/skills/skills/wright", + "workshop-skill": "/abs/path/local/workshop-wiki" + } + } } diff --git a/benchmarks/agent/matrix.pilot.example.json b/benchmarks/agent/matrix.pilot.example.json new file mode 100644 index 00000000..5311b38c --- /dev/null +++ b/benchmarks/agent/matrix.pilot.example.json @@ -0,0 +1,62 @@ +{ + "agents": [ + { + "id": "pi-gpt-6-luna", + "cmd": "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/wright/benchmarks/agent/adapters/pi.py" + } + ], + "cells": [ + { + "tool": "none", + "skills": [], + "knowledge": "none", + "network": "off" + }, + { + "tool": "wright", + "skills": [], + "knowledge": "none", + "network": "off" + }, + { + "tool": "wright", + "skills": [ + "wright-skill" + ], + "knowledge": "none", + "network": "off" + }, + { + "tool": "none", + "skills": [], + "knowledge": "wiki", + "network": "off" + }, + { + "tool": "none", + "skills": [ + "workshop-skill" + ], + "knowledge": "none", + "network": "off" + } + ], + "scenarios": [ + "ana-paintball", + "widow-headshots" + ], + "trials": 3, + "parallel": 1, + "seed": 1, + "options": { + "wiki_dir": "/abs/path/pinned-wiki", + "env_pass": [ + "HOME" + ], + "timeout": 3600, + "skill_dirs": { + "wright-skill": "/abs/path/skills/skills/wright", + "workshop-skill": "/abs/path/local/workshop-wiki" + } + } +} diff --git a/benchmarks/agent/scenarios/ana-paintball/negative/upstream-rejects-name/mode.opy b/benchmarks/agent/scenarios/ana-paintball/negative/hallucinated-name/mode.opy similarity index 100% rename from benchmarks/agent/scenarios/ana-paintball/negative/upstream-rejects-name/mode.opy rename to benchmarks/agent/scenarios/ana-paintball/negative/hallucinated-name/mode.opy diff --git a/benchmarks/agent/scenarios/ana-paintball/scenario.json b/benchmarks/agent/scenarios/ana-paintball/scenario.json index 446d335d..d941bf0c 100644 --- a/benchmarks/agent/scenarios/ana-paintball/scenario.json +++ b/benchmarks/agent/scenarios/ana-paintball/scenario.json @@ -138,7 +138,7 @@ "piercing-via-raycast" ] }, - "upstream-rejects-name": { + "hallucinated-name": { "fails": [ "ana", "ffa", @@ -151,7 +151,9 @@ "sleep-cooldown-2.5", "sleep-dart", "target-25", - "upstream-valid" + "upstream-valid", + "wright-check", + "wright-compiles" ] } } diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/forgets-self-kills/mode.opy b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/forgets-self-kills/mode.opy new file mode 100644 index 00000000..0a4d109e --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/forgets-self-kills/mode.opy @@ -0,0 +1,24 @@ +settings { + "main": {"description": "First to seven"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +globalvar targetScore +playervar score + +rule "configure match": + targetScore = 7 + hudHeader(getAllPlayers(), "First to {0}".format(targetScore), HudPosition.TOP, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "force soldier": + @Event eachPlayer + eventPlayer.startForcingHero(Hero.SOLDIER) + +rule "score on elimination": + @Event playerEarnedElimination + attacker.score += 1 + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= targetScore + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/no-hud/mode.opy b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/no-hud/mode.opy new file mode 100644 index 00000000..c416e677 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/negative/no-hud/mode.opy @@ -0,0 +1,24 @@ +settings { + "main": {"description": "First to seven"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +globalvar targetScore +playervar score + +rule "configure match": + targetScore = 7 + +rule "force soldier": + @Event eachPlayer + eventPlayer.startForcingHero(Hero.SOLDIER) + +rule "score on elimination": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.score += 1 + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= targetScore + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/prompt.md b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/prompt.md new file mode 100644 index 00000000..d0171519 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/prompt.md @@ -0,0 +1,10 @@ +Create `mode.opy`, an Overwatch Workshop free-for-all game mode written in OverPy, in this directory. + +Requirements: + +- Every player is forced to play Soldier: 76. +- An elimination of another player scores one point for the attacker. Self-inflicted deaths score nothing. +- A HUD visible to everyone shows the winning score. +- The first player to reach 7 points wins the match. + +Make sure the finished project is valid and has no remaining diagnostics. diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/reference/mode.opy b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/reference/mode.opy new file mode 100644 index 00000000..da888568 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/reference/mode.opy @@ -0,0 +1,25 @@ +settings { + "main": {"description": "First to seven"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +globalvar targetScore +playervar score + +rule "configure match": + targetScore = 7 + hudHeader(getAllPlayers(), "First to {0}".format(targetScore), HudPosition.TOP, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "force soldier": + @Event eachPlayer + eventPlayer.startForcingHero(Hero.SOLDIER) + +rule "score on elimination": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.score += 1 + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= targetScore + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/scenario.json b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/scenario.json new file mode 100644 index 00000000..4156c1a4 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/scenario.json @@ -0,0 +1,106 @@ +{ + "id": "greenfield-opy-elimination-race", + "family": "greenfield", + "language": "opy", + "entry": "mode.opy", + "split": "test", + "writable": [ + "mode.opy" + ], + "runtimeOnly": [ + "Elimination scoring, the HUD, and the winner declaration behave as intended in a live match." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "opy-rs" + }, + { + "id": "upstream-valid", + "kind": "oracle", + "layer": "agent" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "opy-rs" + }, + { + "id": "tracks-score-per-player", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "player:\\s*\\n\\s*0:" + ] + }, + { + "id": "forces-soldier", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Hero\\(Soldier: 76\\)" + ] + }, + { + "id": "scores-on-elimination", + "kind": "compiled-contains", + "layer": "agent", + "any": [ + "Player Earned Elimination", + "Player Dealt Final Blow" + ] + }, + { + "id": "excludes-self-kills", + "kind": "compiled-contains", + "layer": "agent", + "any": [ + "Attacker != Victim", + "Compare\\(Attacker, !=, Victim\\)" + ] + }, + { + "id": "has-hud", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Create HUD Text" + ] + }, + { + "id": "target-is-7", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "\\b7\\b" + ] + }, + { + "id": "declares-winner", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Declare Player Victory" + ] + }, + { + "id": "no-waitless-loop", + "kind": "lint", + "layer": "wright", + "code": "while-without-wait" + } + ], + "negatives": { + "forgets-self-kills": { + "fails": [ + "excludes-self-kills" + ] + }, + "no-hud": { + "fails": [ + "has-hud" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/greenfield-opy-elimination-race/seed/.gitkeep b/benchmarks/agent/scenarios/greenfield-opy-elimination-race/seed/.gitkeep new file mode 100644 index 00000000..e69de29b diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/drops-an-action/mode.ws b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/drops-an-action/mode.ws new file mode 100644 index 00000000..2bb7776a --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/drops-an-action/mode.ws @@ -0,0 +1,71 @@ +settings +{ + main + { + Description: "Reward demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: streak +} + +subroutines { + 0: grantReward +} + +rule ("grantReward") { + event { + Subroutine; + grantReward; + } + actions { + Modify Player Variable(Event Player, streak, Add, 1); + Set Ultimate Charge(Event Player, 100); + Big Message(Event Player, Custom String("Streak {0}", (Event Player).streak)); + } +} + +rule ("reward on elimination") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Call Subroutine(grantReward); + } +} + +rule ("reward on assist") { + event { + Player Dealt Final Blow; + All; + All; + } + actions { + Call Subroutine(grantReward); + } +} + +rule ("reset streak") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Event Player, streak, 0); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/only-one-rule-calls-it/mode.ws b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/only-one-rule-calls-it/mode.ws new file mode 100644 index 00000000..54d09eb9 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/only-one-rule-calls-it/mode.ws @@ -0,0 +1,75 @@ +settings +{ + main + { + Description: "Reward demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: streak +} + +subroutines { + 0: grantReward +} + +rule ("grantReward") { + event { + Subroutine; + grantReward; + } + actions { + Modify Player Variable(Event Player, streak, Add, 1); + Set Ultimate Charge(Event Player, 100); + Big Message(Event Player, Custom String("Streak {0}", (Event Player).streak)); + Heal(Event Player, Null, 50); + } +} + +rule ("reward on elimination") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Call Subroutine(grantReward); + } +} + +rule ("reward on assist") { + event { + Player Dealt Final Blow; + All; + All; + } + actions { + Modify Player Variable(Attacker, streak, Add, 1); + Set Ultimate Charge(Attacker, 100); + Big Message(Attacker, Custom String("Streak {0}", (Attacker).streak)); + Heal(Attacker, Null, 50); + } +} + +rule ("reset streak") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Event Player, streak, 0); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/unchanged-seed/mode.ws b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/unchanged-seed/mode.ws new file mode 100644 index 00000000..4d6e2fbf --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/negative/unchanged-seed/mode.ws @@ -0,0 +1,61 @@ +settings +{ + main + { + Description: "Reward demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: streak +} + +rule ("reward on elimination") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Modify Player Variable(Attacker, streak, Add, 1); + Set Ultimate Charge(Attacker, 100); + Big Message(Attacker, Custom String("Streak {0}", (Attacker).streak)); + Heal(Attacker, Null, 50); + } +} + +rule ("reward on assist") { + event { + Player Dealt Final Blow; + All; + All; + } + actions { + Modify Player Variable(Attacker, streak, Add, 1); + Set Ultimate Charge(Attacker, 100); + Big Message(Attacker, Custom String("Streak {0}", (Attacker).streak)); + Heal(Attacker, Null, 50); + } +} + +rule ("reset streak") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Event Player, streak, 0); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/prompt.md b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/prompt.md new file mode 100644 index 00000000..604421e6 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/prompt.md @@ -0,0 +1,3 @@ +`mode.ws` is an Overwatch Workshop game mode written as raw Workshop script. The same four actions are repeated in two rules. Move them into one subroutine and call it from both rules, without changing what the mode does. + +Keep every existing rule, and make sure the project stays valid. diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/reference/mode.ws b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/reference/mode.ws new file mode 100644 index 00000000..c12be39e --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/reference/mode.ws @@ -0,0 +1,72 @@ +settings +{ + main + { + Description: "Reward demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: streak +} + +subroutines { + 0: grantReward +} + +rule ("grantReward") { + event { + Subroutine; + grantReward; + } + actions { + Modify Player Variable(Event Player, streak, Add, 1); + Set Ultimate Charge(Event Player, 100); + Big Message(Event Player, Custom String("Streak {0}", (Event Player).streak)); + Heal(Event Player, Null, 50); + } +} + +rule ("reward on elimination") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Call Subroutine(grantReward); + } +} + +rule ("reward on assist") { + event { + Player Dealt Final Blow; + All; + All; + } + actions { + Call Subroutine(grantReward); + } +} + +rule ("reset streak") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Event Player, streak, 0); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/scenario.json b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/scenario.json new file mode 100644 index 00000000..733207a0 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/scenario.json @@ -0,0 +1,90 @@ +{ + "id": "modify-opy-extract-subroutine-ws", + "family": "modification", + "language": "workshop", + "entry": "mode.ws", + "split": "test", + "writable": [ + "mode.ws" + ], + "runtimeOnly": [ + "The refactored subroutine keeps the same in-game behavior." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "workshop-rs" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "workshop-rs" + }, + { + "id": "subroutine-defined", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "subroutines\\s*\\{\\s*\\n?\\s*0:" + ] + }, + { + "id": "called-from-both-rules", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Call Subroutine[\\s\\S]*Call Subroutine" + ] + }, + { + "id": "duplication-removed", + "kind": "compiled-absent", + "layer": "agent", + "all": [ + "(Set Ultimate Charge[\\s\\S]*){2}" + ] + }, + { + "id": "rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"reward on elimination\"\\)", + "rule \\(\"reward on assist\"\\)", + "rule \\(\"reset streak\"\\)" + ] + }, + { + "id": "all-four-actions-kept", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Modify Player Variable\\([^;]*streak", + "Set Ultimate Charge", + "Big Message", + "Heal\\(" + ] + } + ], + "negatives": { + "only-one-rule-calls-it": { + "fails": [ + "called-from-both-rules", + "duplication-removed" + ] + }, + "drops-an-action": { + "fails": [ + "all-four-actions-kept" + ] + }, + "unchanged-seed": { + "fails": [ + "called-from-both-rules", + "duplication-removed", + "subroutine-defined" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/seed/mode.ws b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/seed/mode.ws new file mode 100644 index 00000000..4d6e2fbf --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine-ws/seed/mode.ws @@ -0,0 +1,61 @@ +settings +{ + main + { + Description: "Reward demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: streak +} + +rule ("reward on elimination") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Modify Player Variable(Attacker, streak, Add, 1); + Set Ultimate Charge(Attacker, 100); + Big Message(Attacker, Custom String("Streak {0}", (Attacker).streak)); + Heal(Attacker, Null, 50); + } +} + +rule ("reward on assist") { + event { + Player Dealt Final Blow; + All; + All; + } + actions { + Modify Player Variable(Attacker, streak, Add, 1); + Set Ultimate Charge(Attacker, 100); + Big Message(Attacker, Custom String("Streak {0}", (Attacker).streak)); + Heal(Attacker, Null, 50); + } +} + +rule ("reset streak") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Event Player, streak, 0); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/drops-an-action/mode.opy b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/drops-an-action/mode.opy new file mode 100644 index 00000000..d6b7be93 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/drops-an-action/mode.opy @@ -0,0 +1,24 @@ +settings { + "main": {"description": "Reward demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar streak + +def grantReward(): + @Name "grantReward" + eventPlayer.streak += 1 + eventPlayer.setUltCharge(100) + bigMessage(eventPlayer, "Streak {0}".format(eventPlayer.streak)) + +rule "reward on elimination": + @Event playerEarnedElimination + grantReward() + +rule "reward on assist": + @Event playerDealtFinalBlow + grantReward() + +rule "reset streak": + @Event playerDied + eventPlayer.streak = 0 diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/only-one-rule-calls-it/mode.opy b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/only-one-rule-calls-it/mode.opy new file mode 100644 index 00000000..2fbd138f --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/only-one-rule-calls-it/mode.opy @@ -0,0 +1,28 @@ +settings { + "main": {"description": "Reward demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar streak + +def grantReward(): + @Name "grantReward" + eventPlayer.streak += 1 + eventPlayer.setUltCharge(100) + bigMessage(eventPlayer, "Streak {0}".format(eventPlayer.streak)) + heal(eventPlayer, null, 50) + +rule "reward on elimination": + @Event playerEarnedElimination + grantReward() + +rule "reward on assist": + @Event playerDealtFinalBlow + attacker.streak += 1 + attacker.setUltCharge(100) + bigMessage(attacker, "Streak {0}".format(attacker.streak)) + heal(attacker, null, 50) + +rule "reset streak": + @Event playerDied + eventPlayer.streak = 0 diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/unchanged-seed/mode.opy b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/unchanged-seed/mode.opy new file mode 100644 index 00000000..61332458 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/negative/unchanged-seed/mode.opy @@ -0,0 +1,24 @@ +settings { + "main": {"description": "Reward demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar streak + +rule "reward on elimination": + @Event playerEarnedElimination + attacker.streak += 1 + attacker.setUltCharge(100) + bigMessage(attacker, "Streak {0}".format(attacker.streak)) + heal(attacker, null, 50) + +rule "reward on assist": + @Event playerDealtFinalBlow + attacker.streak += 1 + attacker.setUltCharge(100) + bigMessage(attacker, "Streak {0}".format(attacker.streak)) + heal(attacker, null, 50) + +rule "reset streak": + @Event playerDied + eventPlayer.streak = 0 diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/prompt.md b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/prompt.md new file mode 100644 index 00000000..08d3bdd7 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/prompt.md @@ -0,0 +1,3 @@ +`mode.opy` is an Overwatch Workshop game mode written in OverPy. The same four actions are repeated in two rules. Move them into one subroutine and call it from both rules, without changing what the mode does. + +Keep every existing rule, and make sure the project stays valid. diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/reference/mode.opy b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/reference/mode.opy new file mode 100644 index 00000000..94badd15 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/reference/mode.opy @@ -0,0 +1,25 @@ +settings { + "main": {"description": "Reward demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar streak + +def grantReward(): + @Name "grantReward" + eventPlayer.streak += 1 + eventPlayer.setUltCharge(100) + bigMessage(eventPlayer, "Streak {0}".format(eventPlayer.streak)) + heal(eventPlayer, null, 50) + +rule "reward on elimination": + @Event playerEarnedElimination + grantReward() + +rule "reward on assist": + @Event playerDealtFinalBlow + grantReward() + +rule "reset streak": + @Event playerDied + eventPlayer.streak = 0 diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/scenario.json b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/scenario.json new file mode 100644 index 00000000..16209fff --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/scenario.json @@ -0,0 +1,95 @@ +{ + "id": "modify-opy-extract-subroutine", + "family": "modification", + "language": "opy", + "entry": "mode.opy", + "split": "test", + "writable": [ + "mode.opy" + ], + "runtimeOnly": [ + "The refactored subroutine keeps the same in-game behavior." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "opy-rs" + }, + { + "id": "upstream-valid", + "kind": "oracle", + "layer": "agent" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "opy-rs" + }, + { + "id": "subroutine-defined", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "subroutines\\s*\\{\\s*\\n?\\s*0:" + ] + }, + { + "id": "called-from-both-rules", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Call Subroutine[\\s\\S]*Call Subroutine" + ] + }, + { + "id": "duplication-removed", + "kind": "compiled-absent", + "layer": "agent", + "all": [ + "(Set Ultimate Charge[\\s\\S]*){2}" + ] + }, + { + "id": "rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"reward on elimination\"\\)", + "rule \\(\"reward on assist\"\\)", + "rule \\(\"reset streak\"\\)" + ] + }, + { + "id": "all-four-actions-kept", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Modify Player Variable\\([^;]*streak", + "Set Ultimate Charge", + "Big Message", + "Heal\\(" + ] + } + ], + "negatives": { + "only-one-rule-calls-it": { + "fails": [ + "called-from-both-rules", + "duplication-removed" + ] + }, + "drops-an-action": { + "fails": [ + "all-four-actions-kept" + ] + }, + "unchanged-seed": { + "fails": [ + "called-from-both-rules", + "duplication-removed", + "subroutine-defined" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-extract-subroutine/seed/mode.opy b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/seed/mode.opy new file mode 100644 index 00000000..61332458 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-extract-subroutine/seed/mode.opy @@ -0,0 +1,24 @@ +settings { + "main": {"description": "Reward demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar streak + +rule "reward on elimination": + @Event playerEarnedElimination + attacker.streak += 1 + attacker.setUltCharge(100) + bigMessage(attacker, "Streak {0}".format(attacker.streak)) + heal(attacker, null, 50) + +rule "reward on assist": + @Event playerDealtFinalBlow + attacker.streak += 1 + attacker.setUltCharge(100) + bigMessage(attacker, "Streak {0}".format(attacker.streak)) + heal(attacker, null, 50) + +rule "reset streak": + @Event playerDied + eventPlayer.streak = 0 diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/changes-the-target/mode.ws b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/changes-the-target/mode.ws new file mode 100644 index 00000000..6cd541a5 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/changes-the-target/mode.ws @@ -0,0 +1,86 @@ +settings +{ + main + { + Description: "Kill race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: kills +} + +rule ("force heroes") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Ana)); + } +} + +rule ("count kill") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, kills, Add, 1); + } +} + +rule ("restore health") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Heal(Attacker, Null, 50); + } +} + +rule ("kills display") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Eliminations: {0}", (Event Player).kills), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).kills >= 7; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/drops-a-rule/mode.ws b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/drops-a-rule/mode.ws new file mode 100644 index 00000000..2bc923e8 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/drops-a-rule/mode.ws @@ -0,0 +1,75 @@ +settings +{ + main + { + Description: "Kill race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: kills +} + +rule ("force heroes") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Ana)); + } +} + +rule ("count kill") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, kills, Add, 1); + } +} + +rule ("kills display") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Eliminations: {0}", (Event Player).kills), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).kills >= 5; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/no-hud/mode.ws b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/no-hud/mode.ws new file mode 100644 index 00000000..3fd15826 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/negative/no-hud/mode.ws @@ -0,0 +1,75 @@ +settings +{ + main + { + Description: "Kill race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: kills +} + +rule ("force heroes") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Ana)); + } +} + +rule ("count kill") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, kills, Add, 1); + } +} + +rule ("restore health") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Heal(Attacker, Null, 50); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).kills >= 5; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/prompt.md b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/prompt.md new file mode 100644 index 00000000..83ddb2a8 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/prompt.md @@ -0,0 +1,5 @@ +`mode.ws` is an Overwatch Workshop game mode written as raw Workshop script. Make this change and nothing else: + +- Show every player their own elimination count in a HUD, updated as they score. + +Leave the existing rules' behavior untouched, and make sure the project stays valid. diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/reference/mode.ws b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/reference/mode.ws new file mode 100644 index 00000000..cde1cb58 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/reference/mode.ws @@ -0,0 +1,86 @@ +settings +{ + main + { + Description: "Kill race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: kills +} + +rule ("force heroes") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Ana)); + } +} + +rule ("count kill") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, kills, Add, 1); + } +} + +rule ("restore health") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Heal(Attacker, Null, 50); + } +} + +rule ("kills display") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Eliminations: {0}", (Event Player).kills), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).kills >= 5; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/scenario.json b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/scenario.json new file mode 100644 index 00000000..e7543523 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/scenario.json @@ -0,0 +1,78 @@ +{ + "id": "modify-opy-kill-hud-ws", + "family": "modification", + "language": "workshop", + "entry": "mode.ws", + "split": "test", + "writable": [ + "mode.ws" + ], + "runtimeOnly": [ + "The HUD text updates and renders as intended in a live match." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "workshop-rs" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "workshop-rs" + }, + { + "id": "hud-shows-own-eliminations", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Create HUD Text\\((Event Player|All Players)[^;]*(\\(Event Player\\)|Event Player)\\.\\w*kills" + ] + }, + { + "id": "rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"force heroes\"\\)", + "rule \\(\"count kill\"\\)", + "rule \\(\"restore health\"\\)", + "rule \\(\"declare winner\"\\)" + ] + }, + { + "id": "target-unchanged", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + ">= 5" + ] + }, + { + "id": "health-rule-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Heal\\(" + ] + } + ], + "negatives": { + "no-hud": { + "fails": [ + "hud-shows-own-eliminations" + ] + }, + "drops-a-rule": { + "fails": [ + "health-rule-preserved", + "rules-preserved" + ] + }, + "changes-the-target": { + "fails": [ + "target-unchanged" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/seed/mode.ws b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/seed/mode.ws new file mode 100644 index 00000000..3fd15826 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud-ws/seed/mode.ws @@ -0,0 +1,75 @@ +settings +{ + main + { + Description: "Kill race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: kills +} + +rule ("force heroes") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Ana)); + } +} + +rule ("count kill") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, kills, Add, 1); + } +} + +rule ("restore health") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Heal(Attacker, Null, 50); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).kills >= 5; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/changes-the-target/mode.opy b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/changes-the-target/mode.opy new file mode 100644 index 00000000..0ad43151 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/changes-the-target/mode.opy @@ -0,0 +1,31 @@ +settings { + "main": {"description": "Kill race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +#!define TARGET_KILLS 7 + +playervar kills + +rule "force heroes": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.startForcingHero(Hero.ANA) + +rule "count kill": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.kills += 1 + +rule "restore health": + @Event playerEarnedElimination + heal(attacker, null, 50) + +rule "kills display": + @Event eachPlayer + hudHeader(eventPlayer, "Eliminations: {0}".format(eventPlayer.kills), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.kills >= TARGET_KILLS + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/drops-a-rule/mode.opy b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/drops-a-rule/mode.opy new file mode 100644 index 00000000..ac17461e --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/drops-a-rule/mode.opy @@ -0,0 +1,27 @@ +settings { + "main": {"description": "Kill race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +#!define TARGET_KILLS 5 + +playervar kills + +rule "force heroes": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.startForcingHero(Hero.ANA) + +rule "count kill": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.kills += 1 + +rule "kills display": + @Event eachPlayer + hudHeader(eventPlayer, "Eliminations: {0}".format(eventPlayer.kills), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.kills >= TARGET_KILLS + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/no-hud/mode.opy b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/no-hud/mode.opy new file mode 100644 index 00000000..835bfdab --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/negative/no-hud/mode.opy @@ -0,0 +1,27 @@ +settings { + "main": {"description": "Kill race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +#!define TARGET_KILLS 5 + +playervar kills + +rule "force heroes": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.startForcingHero(Hero.ANA) + +rule "count kill": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.kills += 1 + +rule "restore health": + @Event playerEarnedElimination + heal(attacker, null, 50) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.kills >= TARGET_KILLS + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/prompt.md b/benchmarks/agent/scenarios/modify-opy-kill-hud/prompt.md new file mode 100644 index 00000000..403b3eb4 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/prompt.md @@ -0,0 +1,5 @@ +`mode.opy` is an Overwatch Workshop game mode written in OverPy. Make this change and nothing else: + +- Show every player their own elimination count in a HUD, updated as they score. + +Leave the existing rules' behavior untouched, and make sure the project stays valid. diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/reference/mode.opy b/benchmarks/agent/scenarios/modify-opy-kill-hud/reference/mode.opy new file mode 100644 index 00000000..506d1d57 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/reference/mode.opy @@ -0,0 +1,31 @@ +settings { + "main": {"description": "Kill race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +#!define TARGET_KILLS 5 + +playervar kills + +rule "force heroes": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.startForcingHero(Hero.ANA) + +rule "count kill": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.kills += 1 + +rule "restore health": + @Event playerEarnedElimination + heal(attacker, null, 50) + +rule "kills display": + @Event eachPlayer + hudHeader(eventPlayer, "Eliminations: {0}".format(eventPlayer.kills), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.kills >= TARGET_KILLS + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/scenario.json b/benchmarks/agent/scenarios/modify-opy-kill-hud/scenario.json new file mode 100644 index 00000000..76200878 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/scenario.json @@ -0,0 +1,83 @@ +{ + "id": "modify-opy-kill-hud", + "family": "modification", + "language": "opy", + "entry": "mode.opy", + "split": "test", + "writable": [ + "mode.opy" + ], + "runtimeOnly": [ + "The HUD text updates and renders as intended in a live match." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "opy-rs" + }, + { + "id": "upstream-valid", + "kind": "oracle", + "layer": "agent" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "opy-rs" + }, + { + "id": "hud-shows-own-eliminations", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Create HUD Text\\((Event Player|All Players)[^;]*(\\(Event Player\\)|Event Player)\\.\\w*kills" + ] + }, + { + "id": "rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"force heroes\"\\)", + "rule \\(\"count kill\"\\)", + "rule \\(\"restore health\"\\)", + "rule \\(\"declare winner\"\\)" + ] + }, + { + "id": "target-unchanged", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + ">= 5" + ] + }, + { + "id": "health-rule-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Heal\\(" + ] + } + ], + "negatives": { + "no-hud": { + "fails": [ + "hud-shows-own-eliminations" + ] + }, + "drops-a-rule": { + "fails": [ + "health-rule-preserved", + "rules-preserved" + ] + }, + "changes-the-target": { + "fails": [ + "target-unchanged" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/modify-opy-kill-hud/seed/mode.opy b/benchmarks/agent/scenarios/modify-opy-kill-hud/seed/mode.opy new file mode 100644 index 00000000..835bfdab --- /dev/null +++ b/benchmarks/agent/scenarios/modify-opy-kill-hud/seed/mode.opy @@ -0,0 +1,27 @@ +settings { + "main": {"description": "Kill race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +#!define TARGET_KILLS 5 + +playervar kills + +rule "force heroes": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.startForcingHero(Hero.ANA) + +rule "count kill": + @Event playerEarnedElimination + @Condition attacker != victim + attacker.kills += 1 + +rule "restore health": + @Event playerEarnedElimination + heal(attacker, null, 50) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.kills >= TARGET_KILLS + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fix-without-diagnosis/mode.ws b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fix-without-diagnosis/mode.ws new file mode 100644 index 00000000..d80344b8 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fix-without-diagnosis/mode.ws @@ -0,0 +1,58 @@ +settings +{ + main + { + Description: "Score race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: score +} + +rule ("credit killer") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + } +} + +rule ("show score") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Score: {0}", (Event Player).score), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).score >= 10; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/answer.json new file mode 100644 index 00000000..9e2217b4 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/answer.json @@ -0,0 +1 @@ +{"rule": "declare winner"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/mode.ws b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/mode.ws new file mode 100644 index 00000000..a2f3367b --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/fixes-wrong-rule/mode.ws @@ -0,0 +1,58 @@ +settings +{ + main + { + Description: "Score race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: score +} + +rule ("credit killer") { + event { + Player Died; + All; + All; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + } +} + +rule ("show score") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Score: {0}", (Event Player).score), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).score >= 10; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/answer.json new file mode 100644 index 00000000..5fc36344 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/answer.json @@ -0,0 +1 @@ +{"rule": "credit killer"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/mode.ws b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/mode.ws new file mode 100644 index 00000000..c646401f --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/negative/removes-the-scoring/mode.ws @@ -0,0 +1,47 @@ +settings +{ + main + { + Description: "Score race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: score +} + +rule ("show score") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Score: {0}", (Event Player).score), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).score >= 10; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/prompt.md b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/prompt.md new file mode 100644 index 00000000..3a4c2788 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/prompt.md @@ -0,0 +1,3 @@ +Players report that in `mode.ws` (an Overwatch Workshop game mode written as raw Workshop script) jumping off the map scores a point for the player who jumped, so people win by falling. + +Find the root cause, fix it with the smallest change that keeps the intended behavior (an elimination scores one point for the attacker), and record the name of the offending rule as `{"rule": ""}` in `answer.json`. Make sure the project has no remaining problems. diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/answer.json new file mode 100644 index 00000000..5fc36344 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/answer.json @@ -0,0 +1 @@ +{"rule": "credit killer"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/mode.ws b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/mode.ws new file mode 100644 index 00000000..dca2fb52 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/reference/mode.ws @@ -0,0 +1,61 @@ +settings +{ + main + { + Description: "Score race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: score +} + +rule ("credit killer") { + event { + Player Died; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + } +} + +rule ("show score") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Score: {0}", (Event Player).score), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).score >= 10; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/scenario.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/scenario.json new file mode 100644 index 00000000..60542d87 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/scenario.json @@ -0,0 +1,80 @@ +{ + "id": "repair-opy-self-kill-score-ws", + "family": "diagnosis", + "language": "workshop", + "entry": "mode.ws", + "split": "test", + "writable": [ + "mode.ws", + "answer.json" + ], + "runtimeOnly": [ + "Falling off the map no longer scores in a live match." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "workshop-rs" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "workshop-rs" + }, + { + "id": "self-kills-excluded", + "kind": "compiled-contains", + "layer": "agent", + "any": [ + "Attacker != Victim", + "Compare\\(Attacker, !=, Victim\\)", + "Player Earned Elimination", + "Player Dealt Final Blow" + ] + }, + { + "id": "still-scores-the-attacker", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Modify Player Variable\\(Attacker, score, Add, 1\\)" + ] + }, + { + "id": "other-rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"show score\"\\)", + "rule \\(\"declare winner\"\\)" + ] + }, + { + "id": "root-cause-named", + "kind": "answer", + "layer": "agent", + "key": "rule", + "expected": "credit killer" + } + ], + "negatives": { + "fixes-wrong-rule": { + "fails": [ + "root-cause-named", + "self-kills-excluded" + ] + }, + "removes-the-scoring": { + "fails": [ + "self-kills-excluded", + "still-scores-the-attacker" + ] + }, + "fix-without-diagnosis": { + "fails": [ + "root-cause-named" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/seed/mode.ws b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/seed/mode.ws new file mode 100644 index 00000000..a2f3367b --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score-ws/seed/mode.ws @@ -0,0 +1,58 @@ +settings +{ + main + { + Description: "Score race" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + player: + 0: score +} + +rule ("credit killer") { + event { + Player Died; + All; + All; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + } +} + +rule ("show score") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Score: {0}", (Event Player).score), Null, Null, Left, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + (Event Player).score >= 10; + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fix-without-diagnosis/mode.opy b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fix-without-diagnosis/mode.opy new file mode 100644 index 00000000..49535bd5 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fix-without-diagnosis/mode.opy @@ -0,0 +1,19 @@ +settings { + "main": {"description": "Score race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar score + +rule "credit killer": + @Event playerEarnedElimination + attacker.score += 1 + +rule "show score": + @Event eachPlayer + hudHeader(eventPlayer, "Score: {0}".format(eventPlayer.score), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= 10 + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/answer.json new file mode 100644 index 00000000..9e2217b4 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/answer.json @@ -0,0 +1 @@ +{"rule": "declare winner"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/mode.opy b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/mode.opy new file mode 100644 index 00000000..cfc177fb --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/fixes-wrong-rule/mode.opy @@ -0,0 +1,19 @@ +settings { + "main": {"description": "Score race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar score + +rule "credit killer": + @Event playerDied + attacker.score += 1 + +rule "show score": + @Event eachPlayer + hudHeader(eventPlayer, "Score: {0}".format(eventPlayer.score), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= 10 + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/answer.json new file mode 100644 index 00000000..5fc36344 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/answer.json @@ -0,0 +1 @@ +{"rule": "credit killer"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/mode.opy b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/mode.opy new file mode 100644 index 00000000..372b08cd --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/negative/removes-the-scoring/mode.opy @@ -0,0 +1,19 @@ +settings { + "main": {"description": "Score race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar score + +rule "credit killer": + @Event playerDied + wait(0.016) + +rule "show score": + @Event eachPlayer + hudHeader(eventPlayer, "Score: {0}".format(eventPlayer.score), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= 10 + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/prompt.md b/benchmarks/agent/scenarios/repair-opy-self-kill-score/prompt.md new file mode 100644 index 00000000..cd7db3cb --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/prompt.md @@ -0,0 +1,3 @@ +Players report that in `mode.opy` (an Overwatch Workshop game mode written in OverPy) jumping off the map scores a point for the player who jumped, so people win by falling. + +Find the root cause, fix it with the smallest change that keeps the intended behavior (an elimination scores one point for the attacker), and record the name of the offending rule as `{"rule": ""}` in `answer.json`. Make sure the project has no remaining problems. diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/answer.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/answer.json new file mode 100644 index 00000000..5fc36344 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/answer.json @@ -0,0 +1 @@ +{"rule": "credit killer"} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/mode.opy b/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/mode.opy new file mode 100644 index 00000000..4526a182 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/reference/mode.opy @@ -0,0 +1,20 @@ +settings { + "main": {"description": "Score race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar score + +rule "credit killer": + @Event playerDied + @Condition attacker != victim + attacker.score += 1 + +rule "show score": + @Event eachPlayer + hudHeader(eventPlayer, "Score: {0}".format(eventPlayer.score), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= 10 + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/scenario.json b/benchmarks/agent/scenarios/repair-opy-self-kill-score/scenario.json new file mode 100644 index 00000000..d5726445 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/scenario.json @@ -0,0 +1,85 @@ +{ + "id": "repair-opy-self-kill-score", + "family": "diagnosis", + "language": "opy", + "entry": "mode.opy", + "split": "test", + "writable": [ + "mode.opy", + "answer.json" + ], + "runtimeOnly": [ + "Falling off the map no longer scores in a live match." + ], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "opy-rs" + }, + { + "id": "upstream-valid", + "kind": "oracle", + "layer": "agent" + }, + { + "id": "wright-compiles", + "kind": "wright-compile", + "layer": "opy-rs" + }, + { + "id": "self-kills-excluded", + "kind": "compiled-contains", + "layer": "agent", + "any": [ + "Attacker != Victim", + "Compare\\(Attacker, !=, Victim\\)", + "Player Earned Elimination", + "Player Dealt Final Blow" + ] + }, + { + "id": "still-scores-the-attacker", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "Modify Player Variable\\(Attacker, score, Add, 1\\)" + ] + }, + { + "id": "other-rules-preserved", + "kind": "compiled-contains", + "layer": "agent", + "all": [ + "rule \\(\"show score\"\\)", + "rule \\(\"declare winner\"\\)" + ] + }, + { + "id": "root-cause-named", + "kind": "answer", + "layer": "agent", + "key": "rule", + "expected": "credit killer" + } + ], + "negatives": { + "fixes-wrong-rule": { + "fails": [ + "root-cause-named", + "self-kills-excluded" + ] + }, + "removes-the-scoring": { + "fails": [ + "self-kills-excluded", + "still-scores-the-attacker" + ] + }, + "fix-without-diagnosis": { + "fails": [ + "root-cause-named" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/repair-opy-self-kill-score/seed/mode.opy b/benchmarks/agent/scenarios/repair-opy-self-kill-score/seed/mode.opy new file mode 100644 index 00000000..cfc177fb --- /dev/null +++ b/benchmarks/agent/scenarios/repair-opy-self-kill-score/seed/mode.opy @@ -0,0 +1,19 @@ +settings { + "main": {"description": "Score race"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +playervar score + +rule "credit killer": + @Event playerDied + attacker.score += 1 + +rule "show score": + @Event eachPlayer + hudHeader(eventPlayer, "Score: {0}".format(eventPlayer.score), HudPosition.LEFT, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "declare winner": + @Event eachPlayer + @Condition eventPlayer.score >= 10 + declarePlayerVictory(eventPlayer) diff --git a/benchmarks/agent/scenarios/understand-opy-events-ws/negative/text-search-answers/answer.json b/benchmarks/agent/scenarios/understand-opy-events-ws/negative/text-search-answers/answer.json new file mode 100644 index 00000000..406dc782 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events-ws/negative/text-search-answers/answer.json @@ -0,0 +1 @@ +{"deathRules": ["death count", "kill credit"], "subroutineCallers": {"awardPoint": ["kill credit"], "refreshLeader": ["kill credit"], "resetTimer": ["start match"]}, "neverUsed": [], "multiWriterGlobals": []} diff --git a/benchmarks/agent/scenarios/understand-opy-events-ws/prompt.md b/benchmarks/agent/scenarios/understand-opy-events-ws/prompt.md new file mode 100644 index 00000000..f3ca0c7d --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events-ws/prompt.md @@ -0,0 +1,8 @@ +`mode.ws` is an Overwatch Workshop game mode written as raw Workshop script. Do not modify it. + +Answer these questions about it by writing `answer.json` in this directory, with exactly these keys: + +- `deathRules`: the names of the rules whose event is `Player Died`, sorted alphabetically. +- `subroutineCallers`: an object that maps each subroutine name to the sorted list of rule names that call it. +- `neverUsed`: the names of the declared variables that no rule or subroutine reads or writes, sorted alphabetically. +- `multiWriterGlobals`: the names of the global variables written by more than one rule, counting writes made through subroutines that a rule calls, sorted alphabetically. diff --git a/benchmarks/agent/scenarios/understand-opy-events-ws/reference/answer.json b/benchmarks/agent/scenarios/understand-opy-events-ws/reference/answer.json new file mode 100644 index 00000000..97e569f2 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events-ws/reference/answer.json @@ -0,0 +1,23 @@ +{ + "deathRules": [ + "death count" + ], + "subroutineCallers": { + "awardPoint": [ + "kill credit" + ], + "refreshLeader": [ + "kill credit" + ], + "resetTimer": [ + "restart on empty", + "start match" + ] + }, + "neverUsed": [ + "spareFlag" + ], + "multiWriterGlobals": [ + "roundTimer" + ] +} diff --git a/benchmarks/agent/scenarios/understand-opy-events-ws/scenario.json b/benchmarks/agent/scenarios/understand-opy-events-ws/scenario.json new file mode 100644 index 00000000..2cd0e905 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events-ws/scenario.json @@ -0,0 +1,73 @@ +{ + "id": "understand-opy-events-ws", + "family": "understanding", + "language": "workshop", + "entry": "mode.ws", + "split": "test", + "writable": [ + "answer.json" + ], + "runtimeOnly": [], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "workshop-rs" + }, + { + "id": "death-rules", + "kind": "answer", + "layer": "agent", + "key": "deathRules", + "expected": [ + "death count" + ] + }, + { + "id": "subroutine-callers", + "kind": "answer", + "layer": "agent", + "key": "subroutineCallers", + "expected": { + "awardPoint": [ + "kill credit" + ], + "refreshLeader": [ + "kill credit" + ], + "resetTimer": [ + "restart on empty", + "start match" + ] + } + }, + { + "id": "never-used", + "kind": "answer", + "layer": "agent", + "key": "neverUsed", + "expected": [ + "spareFlag" + ] + }, + { + "id": "multi-writer-globals", + "kind": "answer", + "layer": "agent", + "key": "multiWriterGlobals", + "expected": [ + "roundTimer" + ] + } + ], + "negatives": { + "text-search-answers": { + "fails": [ + "death-rules", + "multi-writer-globals", + "never-used", + "subroutine-callers" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/understand-opy-events-ws/seed/mode.ws b/benchmarks/agent/scenarios/understand-opy-events-ws/seed/mode.ws new file mode 100644 index 00000000..b59ef0a0 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events-ws/seed/mode.ws @@ -0,0 +1,152 @@ +settings +{ + main + { + Description: "Round demo" + } + modes + { + Deathmatch + { + enabled maps + { + Workshop Island + } + } + } +} + +variables { + global: + 0: matchStarted + 1: roundTimer + 2: leader + 3: spareFlag + 4: bonusPoints + player: + 0: kills + 1: deaths + 2: shield +} + +subroutines { + 0: awardPoint + 1: refreshLeader + 2: resetTimer +} + +rule ("awardPoint") { + event { + Subroutine; + awardPoint; + } + actions { + Modify Player Variable(Event Player, kills, Add, 1); + Modify Global Variable(bonusPoints, Add, 1); + } +} + +rule ("refreshLeader") { + event { + Subroutine; + refreshLeader; + } + actions { + Set Global Variable(leader, First Of(Sorted Array(All Players(All Teams), Multiply(-1, (Current Array Element).kills)))); + } +} + +rule ("resetTimer") { + event { + Subroutine; + resetTimer; + } + actions { + Set Global Variable(roundTimer, 300); + } +} + +rule ("start match") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(matchStarted, True); + Call Subroutine(resetTimer); + } +} + +rule ("countdown") { + event { + Ongoing - Global; + } + conditions { + Global.matchStarted == True; + } + actions { + While(Compare(Global.roundTimer, >, 0)); + Wait(1, Ignore Condition); + Modify Global Variable(roundTimer, Subtract, 1); + End; + } +} + +rule ("restart on empty") { + event { + Ongoing - Global; + } + conditions { + Number Of Players(All Teams) == 0; + } + actions { + Call Subroutine(resetTimer); + } +} + +rule ("kill credit") { + event { + Player Earned Elimination; + All; + All; + } + actions { + Call Subroutine(awardPoint); + Call Subroutine(refreshLeader); + } +} + +rule ("death count") { + event { + Player Died; + All; + All; + } + actions { + Modify Player Variable(Event Player, deaths, Add, 1); + } +} + +rule ("leader hud") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Create HUD Text(Event Player, Custom String("Leader: {0}", Global.leader), Null, Null, Top, 0, Color(White), Null, Null, Visible To and String, Default Visibility); + } +} + +rule ("spawn shield") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Has Spawned(Event Player) == True; + } + actions { + Set Player Variable(Event Player, shield, 100); + } +} diff --git a/benchmarks/agent/scenarios/understand-opy-events/negative/text-search-answers/answer.json b/benchmarks/agent/scenarios/understand-opy-events/negative/text-search-answers/answer.json new file mode 100644 index 00000000..406dc782 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events/negative/text-search-answers/answer.json @@ -0,0 +1 @@ +{"deathRules": ["death count", "kill credit"], "subroutineCallers": {"awardPoint": ["kill credit"], "refreshLeader": ["kill credit"], "resetTimer": ["start match"]}, "neverUsed": [], "multiWriterGlobals": []} diff --git a/benchmarks/agent/scenarios/understand-opy-events/prompt.md b/benchmarks/agent/scenarios/understand-opy-events/prompt.md new file mode 100644 index 00000000..a0160dfa --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events/prompt.md @@ -0,0 +1,8 @@ +`mode.opy` is an Overwatch Workshop game mode written in OverPy. Do not modify it. + +Answer these questions about it by writing `answer.json` in this directory, with exactly these keys: + +- `deathRules`: the names of the rules whose event is `playerDied`, sorted alphabetically. +- `subroutineCallers`: an object that maps each subroutine name to the sorted list of rule names that call it. +- `neverUsed`: the names of the declared variables that no rule or subroutine reads or writes, sorted alphabetically. +- `multiWriterGlobals`: the names of the global variables written by more than one rule, counting writes made through subroutines that a rule calls, sorted alphabetically. diff --git a/benchmarks/agent/scenarios/understand-opy-events/reference/answer.json b/benchmarks/agent/scenarios/understand-opy-events/reference/answer.json new file mode 100644 index 00000000..97e569f2 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events/reference/answer.json @@ -0,0 +1,23 @@ +{ + "deathRules": [ + "death count" + ], + "subroutineCallers": { + "awardPoint": [ + "kill credit" + ], + "refreshLeader": [ + "kill credit" + ], + "resetTimer": [ + "restart on empty", + "start match" + ] + }, + "neverUsed": [ + "spareFlag" + ], + "multiWriterGlobals": [ + "roundTimer" + ] +} diff --git a/benchmarks/agent/scenarios/understand-opy-events/scenario.json b/benchmarks/agent/scenarios/understand-opy-events/scenario.json new file mode 100644 index 00000000..f8f89dfe --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events/scenario.json @@ -0,0 +1,78 @@ +{ + "id": "understand-opy-events", + "family": "understanding", + "language": "opy", + "entry": "mode.opy", + "split": "test", + "writable": [ + "answer.json" + ], + "runtimeOnly": [], + "checks": [ + { + "id": "valid-project", + "kind": "check", + "layer": "opy-rs" + }, + { + "id": "upstream-valid", + "kind": "oracle", + "layer": "agent" + }, + { + "id": "death-rules", + "kind": "answer", + "layer": "agent", + "key": "deathRules", + "expected": [ + "death count" + ] + }, + { + "id": "subroutine-callers", + "kind": "answer", + "layer": "agent", + "key": "subroutineCallers", + "expected": { + "awardPoint": [ + "kill credit" + ], + "refreshLeader": [ + "kill credit" + ], + "resetTimer": [ + "restart on empty", + "start match" + ] + } + }, + { + "id": "never-used", + "kind": "answer", + "layer": "agent", + "key": "neverUsed", + "expected": [ + "spareFlag" + ] + }, + { + "id": "multi-writer-globals", + "kind": "answer", + "layer": "agent", + "key": "multiWriterGlobals", + "expected": [ + "roundTimer" + ] + } + ], + "negatives": { + "text-search-answers": { + "fails": [ + "death-rules", + "multi-writer-globals", + "never-used", + "subroutine-callers" + ] + } + } +} diff --git a/benchmarks/agent/scenarios/understand-opy-events/seed/mode.opy b/benchmarks/agent/scenarios/understand-opy-events/seed/mode.opy new file mode 100644 index 00000000..3101e3c0 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-opy-events/seed/mode.opy @@ -0,0 +1,61 @@ +settings { + "main": {"description": "Round demo"}, + "gamemodes": {"ffa": {"enabledMaps": ["workshopIsland"]}} +} + +globalvar matchStarted +globalvar roundTimer +globalvar leader +globalvar spareFlag +globalvar bonusPoints + +playervar kills +playervar deaths +playervar shield + +# spareFlag was used by an older timer; nothing uses it now. + +def awardPoint(): + @Name "awardPoint" + eventPlayer.kills += 1 + bonusPoints += 1 + +def refreshLeader(): + @Name "refreshLeader" + leader = sorted(getAllPlayers(), lambda p: -p.kills)[0] + +def resetTimer(): + @Name "resetTimer" + roundTimer = 300 + +rule "start match": + matchStarted = true + resetTimer() + +rule "countdown": + @Condition matchStarted == true + while roundTimer > 0: + wait(1) + roundTimer -= 1 + +rule "restart on empty": + @Condition len(getAllPlayers()) == 0 + resetTimer() + +rule "kill credit": + @Event playerEarnedElimination + awardPoint() + refreshLeader() + +rule "death count": + @Event playerDied + eventPlayer.deaths += 1 + +rule "leader hud": + @Event eachPlayer + hudHeader(eventPlayer, "Leader: {0}".format(leader), HudPosition.TOP, 0, Color.WHITE, HudReeval.VISIBILITY_AND_STRING) + +rule "spawn shield": + @Event eachPlayer + @Condition eventPlayer.hasSpawned() + eventPlayer.shield = 100 diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py new file mode 100644 index 00000000..f2e42d8c --- /dev/null +++ b/benchmarks/agent/test_adapters.py @@ -0,0 +1,116 @@ +import sys +import io +import json +from unittest.mock import patch, MagicMock +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent / "adapters")) + +import devin +import pi +import codex +import agy + + +class PiAdapterTest(unittest.TestCase): + def test_usage_row_counts_cache_in_context(self): + message = {"usage": {"input": 586, "output": 5, "cacheRead": 400, "cacheWrite": 100, "reasoning": 3}} + row = pi.usage_row(message, 272_000, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["cache_write"], row["context"], row["context_limit"]), (586, 400, 100, 1086, 272_000)) + self.assertEqual((row["output"], row["reasoning"]), (2, 3)) + self.assertEqual(sum(row[key] or 0 for key in ("input", "output", "cache_read", "cache_write", "reasoning")), 1091) + + def test_recovered_provider_error_does_not_fail_completed_task(self): + outage = {"role": "assistant", "stopReason": "error", "errorMessage": "WebSocket error"} + recovered = {"role": "assistant", "stopReason": "stop", "content": [{"type": "text", "text": "done"}]} + for messages, expected in (([outage, recovered], 0), ([recovered, outage], 75)): + with self.subTest(expected=expected): + proc = MagicMock() + proc.stdin = io.StringIO() + proc.stdout = io.StringIO("".join(json.dumps({"type": "message_end", "message": m}) + "\n" for m in messages)) + proc.stderr = io.StringIO() + proc.wait.return_value = 0 + env = { + "HOME": "/isolated", "BENCH_MODEL": "provider/model", "BENCH_RUN_DIR": "/run", + "BENCH_KNOWLEDGE": "none", "BENCH_USAGE": "/run/usage", + "BENCH_TRANSCRIPT": "/run/transcript", "BENCH_CONTEXT": "/run/context", "BENCH_AGENT_INFO": "/run/agent-info", + } + with ( + patch.dict(pi.os.environ, env, clear=True), + patch.object(pi.sys, "stdin", io.StringIO("task")), + patch.object(pi.sys, "stdout", io.StringIO()), + patch.object(pi.sys, "stderr", io.StringIO()), + patch.object(pi.shutil, "which", return_value="pi"), + patch.object(pi.Path, "mkdir"), + patch.object(pi.Path, "is_file", return_value=False), + patch.object(pi.Path, "write_text"), + patch.object(pi, "context_limit", return_value=None), + patch.object(pi.subprocess, "Popen", return_value=proc), + patch("builtins.open", side_effect=lambda *a, **kw: io.StringIO()), + ): + self.assertEqual(pi.main(), expected) + + def test_loaded_skills_and_final_text(self): + system = {"sections": {"skills": "wrightother"}} + self.assertEqual(pi.skill_names(system), ["wright", "other"]) + self.assertEqual(pi.skill_names({"sections": {}}), []) + self.assertEqual(pi.message_text({"content": [{"type": "text", "text": "a"}, {"type": "tool_use"}, {"type": "text", "text": "b"}]}), "ab") + + +class DevinAdapterTest(unittest.TestCase): + export = {"steps": [ + {"source": "system", "message": '\n'}, + {"source": "system", "message": ( + "\n" + "- **wright**: A guide. (source: /w/.agents/skills/wright/SKILL.md)\n" + "- **devin-cli**: Docs. (source: /h/share/devin/docs)\n" + "- **upload-secrets**: Secrets. (source: builtin:upload-secrets)\n" + "- **context7-mcp**: Docs. (source: /h/cli/plugins/cache/x/skills/context7-mcp/SKILL.md)\n")}, + {"source": "user", "message": "task"}, + {"source": "agent", "timestamp": "2026-09-30T16:18:40+00:00", "metrics": {"prompt_tokens": 1000, "completion_tokens": 20, "cached_tokens": 600}}, + {"source": "agent", "timestamp": "2026-09-30T16:18:41+00:00", "message": "no metrics"}, + ]} + + def test_usage_rows_split_cached_prompt_tokens(self): + rows, _, _ = devin.parse_export(self.export) + self.assertEqual(len(rows), 1) + self.assertEqual((rows[0]["input"], rows[0]["cache_read"], rows[0]["output"], rows[0]["context"]), (400, 600, 20, 1000)) + + def test_loaded_context_excludes_builtins_and_lists_plugins_apart(self): + _, loaded, plugins = devin.parse_export(self.export) + self.assertEqual(loaded, ["AGENTS", "wright"]) + self.assertEqual(plugins, ["context7-mcp"]) + + def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): + base = {"read_config_from": {"claude": True}, "permissions": {"allow": ["Exec(*)"]}, "agent": {"model": "old"}} + closed = devin.isolated_config(base, "swe-2-max", web=False) + opened = devin.isolated_config(base, "swe-2-max", web=True) + self.assertFalse(any(closed["read_config_from"].values())) + self.assertEqual(closed["agent"]["model"], "swe-2-max") + self.assertIn("web_search", closed["permissions"]["deny"]) + self.assertNotIn("web_search", opened["permissions"]["deny"]) + self.assertIn("mcp_call_tool", opened["permissions"]["deny"]) + self.assertNotIn("allow", closed["permissions"]) + + +class NativeAdapterUsageTest(unittest.TestCase): + def test_codex_inclusive_counts_are_split_without_counting_reasoning_twice(self): + row = codex.usage_row({"input_tokens": 100, "cached_input_tokens": 60, "output_tokens": 20, "reasoning_output_tokens": 12}, 1.0, 272000) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["reasoning"], row["context"]), (40, 60, 8, 12, 100)) + self.assertEqual(sum(row[key] or 0 for key in ("input", "cache_read", "cache_write", "output", "reasoning")), 120) + + def test_agy_cache_is_exclusive_and_thinking_is_included_in_output(self): + row = agy.usage_row({"input_tokens": 278, "cache_read_tokens": 30214, "output_tokens": 4, "thinking_tokens": 3}, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["reasoning"], row["context"]), (278, 30214, 1, 3, 30492)) + self.assertIsNone(row["context_limit"]) + + def test_codex_builtin_skills_are_separate_from_observed_project_skills(self): + text = ('- `r0` = `/isolated/.codex/skills/.system`\n- `r1` = `/workspace/.agents/skills`\n' + '- openai-docs: Builtin. (file: r0/openai-docs/SKILL.md)\n' + '- wright: Project. (file: r1/wright/SKILL.md)\n') + self.assertEqual(codex.loaded_skills(text), (["wright"], ["openai-docs"])) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index c75bd3e4..5bd0d966 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -1,7 +1,9 @@ import argparse +import hashlib import json import os import shutil +import sys import tempfile import unittest from pathlib import Path @@ -10,6 +12,7 @@ import bench_grade import bench_report import bench_trace +import wiki_skill WRIGHT = os.environ.get("WRIGHT_BIN", str(agent_bench.ROOT / "target/debug/wright")) SCENARIO = "repair-runaway-loop" @@ -26,33 +29,193 @@ def setUp(self): self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")).resolve() self.addCleanup(shutil.rmtree, self.out, True) - def trial(self, agent_cmd: str, wright: str = "bin", scenario: str = SCENARIO, **options) -> dict: + def trial(self, agent_cmd: str, tool: str = "wright", skills: tuple = (), knowledge: str = "none", scenario: str = SCENARIO, **options) -> dict: args = argparse.Namespace(**{ "wright": str(Path(WRIGHT).resolve()), "agent_id": "fake", "agent_cmd": agent_cmd, "timeout": 60, "infra_retries": 2, - "env_pass": [], "canary_cmd": None, "skill_dir": None, "wiki_dir": None, **options, + "env_pass": [], "canary_cmd": None, "skill_dirs": {}, "wiki_dir": None, "check_ancestors": False, **options, }) - cell = {"wright": wright, "knowledge": "none", "network": "off"} - return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{wright}") + cell = agent_bench.normalize_cell({"tool": tool, "skills": list(skills), "knowledge": knowledge, "network": "off"}) + return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{tool}") + + def test_wiki_snapshot_is_copied_and_identified(self): + snapshot = self.out / "snapshot" + (snapshot / "articles").mkdir(parents=True) + (snapshot / "articles/wait-until.md").write_text("# Wait Until\n") + article_hash = hashlib.sha256((snapshot / "articles/wait-until.md").read_bytes()).hexdigest() + snapshot_hash = hashlib.sha256(f"wait-until {article_hash}".encode()).hexdigest() + (snapshot / "SNAPSHOT.json").write_text(json.dumps({"source": "https://mirror.example", "fetchedAt": "2026-09-30T00:00:00+00:00", "snapshotSha256": snapshot_hash, "documents": [{"slug": "wait-until", "sha256": article_hash}]})) + result = self.trial("ls wiki/articles > listing.txt", knowledge="wiki", wiki_dir=snapshot) + self.assertEqual(result["environment"]["wiki"]["snapshotSha256"], snapshot_hash) + self.assertEqual(result["environment"]["wiki"]["documents"], 1) + self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/listing.txt").read_text().strip(), "wait-until.md") + self.assertNotIn("wiki", " ".join(result["unsafeEdits"])) + + def test_skills_are_installed_identified_and_the_only_ones_expected(self): + skill = self.out / "workshop-wiki" + skill.mkdir() + (skill / "SKILL.md").write_text("---\nname: workshop-wiki\n---\n") + skill_hash = wiki_skill.content_hash(skill) + with self.assertRaisesRegex(SystemExit, "requires --skill"): + self.trial("true", skills=("workshop-skill",)) + expected = self.trial("echo '{\"loaded\": [\"workshop-wiki\"]}' > \"$BENCH_CONTEXT\"; echo \"$BENCH_SKILL_DIRS\" > dirs.txt", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) + self.assertNotIn("invalid", expected) + self.assertEqual(expected["condition"]["label"], "wright+workshop-skill/none/off") + self.assertEqual(expected["environment"]["skills"]["workshop-skill"]["sha256"], skill_hash) + self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/dirs.txt").read_text().strip(), str(skill.resolve())) + stray = self.trial("echo '{\"loaded\": [\"workshop-wiki\", \"other\"]}' > \"$BENCH_CONTEXT\"", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) + self.assertIn("unexpected loaded context", stray["invalid"]) + (skill / "BUILD.json").write_text(json.dumps({"name": "workshop-wiki", "skillSha256": "0" * 64})) + with self.assertRaisesRegex(SystemExit, "skill content mismatch"): + self.trial("true", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) + + def test_the_prompt_is_exactly_the_scenario_prompt(self): + self.trial('cat > received.txt') + received = (self.out / f"{SCENARIO}-wright/workspace/received.txt").read_text() + self.assertEqual(received, (agent_bench.SCENARIOS / SCENARIO / "prompt.md").read_text()) + + def test_cell_labels_applicability_and_baseline(self): + cell = agent_bench.normalize_cell({"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"}) + self.assertEqual(agent_bench.cell_label(cell), "wright+wright-skill/none/off") + base = agent_bench.normalize_cell({"tool": "none", "skills": [], "knowledge": "none", "network": "off"}) + self.assertEqual(agent_bench.cell_label(base), bench_report.BASELINE) + opy = agent_bench.normalize_cell({"tool": "overpy", "skills": ["opy-skill"], "knowledge": "none", "network": "off"}) + self.assertTrue(agent_bench.applicable({"language": "opy"}, opy)) + self.assertFalse(agent_bench.applicable({"language": "workshop"}, opy)) + self.assertTrue(agent_bench.applicable({"language": "workshop"}, cell)) + fmt = agent_bench.normalize_cell({"tool": "none", "skills": ["workshop-format-skill"], "knowledge": "none", "network": "off"}) + self.assertTrue(agent_bench.applicable({"language": "workshop"}, fmt)) + self.assertFalse(agent_bench.applicable({"language": "opy"}, fmt)) + + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") + def test_overpy_tool_is_traced_and_wright_is_hidden(self): + scenario = "repair-opy-runaway-loop" + agent = (f"cp {reference(scenario)}/* . && overpy compile -i mode.opy -o out.txt; " + "(command -v wright >/dev/null && echo found || echo hidden) > wright.txt") + result = self.trial(agent, tool="overpy", skills=("opy-skill",), scenario=scenario, skill_dirs={"opy-skill": self.skill_dir("overpy")}) + self.assertEqual(result["toolUse"]["overpy"]["byCommand"], {"compile": 1}) + self.assertNotIn("wright", result["toolUse"]) + self.assertEqual((self.out / f"{scenario}-overpy/workspace/wright.txt").read_text().strip(), "hidden") + self.assertTrue((self.out / f"{scenario}-overpy/workspace/out.txt").is_file()) + self.assertEqual(result["condition"]["label"], "overpy+opy-skill/none/off") + with self.assertRaisesRegex(SystemExit, "does not apply"): + self.trial("true", tool="overpy", scenario=SCENARIO) + + def skill_dir(self, name: str) -> Path: + directory = self.out / "skills" / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "SKILL.md").write_text(f"---\nname: {name}\n---\n") + return directory + + def test_result_records_protocol_identity_and_status(self): + result = self.trial(f"cp {reference()}/* .") + self.assertEqual(result["status"], "completed") + self.assertEqual(result["protocol"], {"timeoutSeconds": 60, "infraRetries": 2}) + env = result["environment"] + self.assertEqual(env["wrightSha256"], hashlib.sha256(Path(WRIGHT).read_bytes()).hexdigest()) + self.assertEqual((env["suite"]["version"], len(env["suite"]["hash"])), ("v1", 64)) + self.assertTrue(env["harness"]) + interrupted = self.trial("exit 75", infra_retries=0) + self.assertEqual(interrupted["status"], "provider-interrupted") + timeout = self.trial("sleep 5", timeout=1) + self.assertEqual((timeout["status"], timeout["agent"]["exit"]), ("timeout", None)) + self.assertEqual(self.trial("exit 3")["status"], "agent-error") + info = self.trial('echo \'{"agent": "fake", "model": "m"}\' > "$BENCH_AGENT_INFO"') + self.assertEqual(info["agentInfo"]["model"], "m") + + def test_unsafe_edits_block_usable_even_when_every_check_passes(self): + result = self.trial(f"cp {reference()}/* . && echo stray > stray.txt") + self.assertTrue(result["passed"]) + self.assertFalse(result["usable"]) + self.assertEqual(result["usableReason"], ["unsafe-edits"]) + clean = self.trial(f"cp {reference()}/* .") + self.assertEqual((clean["usable"], clean["usableReason"]), (True, [])) + + def test_an_unavailable_required_grader_makes_the_run_invalid_not_a_failure(self): + from unittest.mock import patch + scenario = "repair-opy-runaway-loop" + with patch.object(bench_grade, "oracle_available", return_value=False): + result = self.trial(f"cp {reference(scenario)}/* .", scenario=scenario) + self.assertEqual(result["status"], "invalid") + self.assertIn("grader", result["invalid"]) + self.assertIn("grader-unavailable", result["usableReason"]) + + def test_matrix_skips_inapplicable_cells_and_stops_after_repeated_interruptions(self): + import contextlib + import io + config = self.out / "matrix.json" + config.write_text(json.dumps({ + "agents": [{"id": "fake", "cmd": "exit 75"}], + "cells": [{"tool": "none", "skills": ["opy-skill"], "knowledge": "none", "network": "off"}], + "scenarios": ["repair-runaway-loop", "repair-opy-runaway-loop"], "trials": 4, "parallel": 1, "seed": 1, + "options": {"skill_dirs": {"opy-skill": str(self.skill_dir("overpy"))}, "timeout": 30, "infra_retries": 0}, + })) + args = argparse.Namespace(config=config, out=self.out / "m", wright=str(Path(WRIGHT).resolve()), skill_dirs={}, wiki_dir=None, env_pass=[], + check_ancestors=False, canary_cmd=None, timeout=30, infra_retries=0, file_sandbox=False) + printed = io.StringIO() + with contextlib.redirect_stdout(printed): + code = agent_bench.cmd_matrix(args) + self.assertEqual(code, 3) + self.assertIn("not applicable: repair-runaway-loop", printed.getvalue()) + self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 2) # the third and fourth jobs were left unattempted + self.assertIn("left unattempted", printed.getvalue()) def test_scenarios_are_solvable_and_not_vacuous(self): self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) - def test_levels_differ_only_in_wright_availability(self): + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_agent_file_writes_cannot_escape_the_trial_directory(self): + code = 'from pathlib import Path; Path("allowed.txt").write_text("ok"); Path("../../escaped.txt").write_text("bad")' + import shlex + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True) + self.assertNotEqual(result["agent"]["exit"], 0) + self.assertFalse((self.out / "escaped.txt").exists()) + self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/allowed.txt").read_text(), "ok") + self.assertEqual(result["fileWriteEnforcement"], "trial-directory-only") + + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_host_instructions_are_unreadable_but_workspace_instructions_are_allowed(self): + import shlex + host = self.out / "AGENTS.md" + host.write_text("host-only instructions") + code = ( + 'from pathlib import Path; ' + 'Path("AGENTS.md").write_text("workspace instructions"); ' + 'assert Path("AGENTS.md").read_text() == "workspace instructions"; ' + f'Path({str(host)!r}).read_text()' + ) + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True) + self.assertNotEqual(result["agent"]["exit"], 0) + self.assertIn("PermissionError", (self.out / f"{SCENARIO}-wright/agent.log").read_text()) + self.assertEqual(host.read_text(), "host-only instructions") + self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/AGENTS.md").read_text(), "workspace instructions") + + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" - none = self.trial(agent, wright="none") - assisted = self.trial(agent, wright="bin") - self.assertEqual(none["wrightUse"]["invocations"], 0) + none = self.trial(agent, tool="none") + assisted = self.trial(agent, tool="wright") + self.assertEqual(none["toolUse"], {}) self.assertIn("missing-wright.txt", none["unsafeEdits"]) - self.assertEqual(assisted["wrightUse"]["byCommand"], {"check": 1}) + self.assertEqual(assisted["toolUse"]["wright"]["byCommand"], {"check": 1}) self.assertTrue(assisted["passed"]) self.assertEqual(assisted["unsafeEdits"], []) - def test_canary_rejects_reachable_wright_under_none(self): - cell = {"wright": "none", "knowledge": "none", "network": "off"} - args = argparse.Namespace(canary_cmd=None) + def test_canary_rejects_a_tool_that_is_not_part_of_the_condition(self): + cell = agent_bench.normalize_cell({"tool": "none", "skills": [], "knowledge": "none", "network": "off"}) + args = argparse.Namespace(canary_cmd=None, check_ancestors=False) env = {"PATH": str(Path(WRIGHT).resolve().parent)} self.assertIn("reachable", agent_bench.canaries(cell, env, self.out, args)) - self.assertIsNone(agent_bench.canaries({**cell, "wright": "bin"}, env, self.out, args)) + self.assertIsNone(agent_bench.canaries({**cell, "tool": "wright"}, env, self.out, args)) + + def test_canary_rejects_instruction_files_above_the_workspace(self): + outside = Path(tempfile.mkdtemp()).resolve() # outside the repository, whose own AGENTS.md would match + self.addCleanup(shutil.rmtree, outside, True) + workspace = outside / "repo/runs/w" + workspace.mkdir(parents=True) + cell = agent_bench.normalize_cell({"tool": "wright", "skills": [], "knowledge": "none", "network": "on"}) + args = argparse.Namespace(canary_cmd=None, check_ancestors=True) + self.assertIsNone(agent_bench.canaries(cell, {"PATH": ""}, workspace, args)) + (outside / "repo/AGENTS.md").write_text("instructions") + self.assertIn("AGENTS.md", agent_bench.canaries(cell, {"PATH": ""}, workspace, args)) def test_network_canary_invalidates_run(self): result = self.trial("true", canary_cmd="true") @@ -63,16 +226,16 @@ def test_environment_is_scrubbed(self): os.environ["BENCH_LEAK_PROBE"] = "leak" self.addCleanup(os.environ.pop, "BENCH_LEAK_PROBE", None) self.trial("env > env.txt") - env = (self.out / f"{SCENARIO}-bin/workspace/env.txt").read_text() + env = (self.out / f"{SCENARIO}-wright/workspace/env.txt").read_text() self.assertNotIn("BENCH_LEAK_PROBE", env) self.assertIn("BENCH_KNOWLEDGE=none", env) - self.assertIn(f"HOME={self.out}/{SCENARIO}-bin/home", env) + self.assertIn(f"HOME={self.out}/{SCENARIO}-wright/home", env) def test_trace_records_envelope_and_serve_sessions(self): agent = (f"cp {reference()}/* . && wright lint mode.ws -f json >/dev/null; " "printf '{\"op\":\"capabilities\"}\\n{\"op\":\"lint\"}\\n' | wright serve mode.ws >/dev/null") result = self.trial(agent) - events = bench_trace.read_events(self.out / f"{SCENARIO}-bin/wright-trace.jsonl") + events = bench_trace.read_events(self.out / f"{SCENARIO}-wright/tool-trace.jsonl") lint = next(e for e in events if e["type"] == "call" and bench_trace.command_of(e["argv"]) == "lint") self.assertEqual(lint["envelope"]["command"], "lint") self.assertRegex(lint["envelope"]["inputIdentity"], r"^[0-9a-f]{64}$") @@ -157,7 +320,7 @@ def test_oracle_disagreement_is_reported(self): class DetectorTest(unittest.TestCase): def call(self, argv, exit_code=0, t=0.0, **extra): - return {"type": "call", "t": t, "argv": argv, "exit": exit_code, "seconds": 0.1, "stdoutBytes": 40, "stderrBytes": 0, "stderrHead": "", "envelope": None, **extra} + return {"tool": "wright", "type": "call", "t": t, "argv": argv, "exit": exit_code, "seconds": 0.1, "stdoutBytes": 40, "stderrBytes": 0, "stderrHead": "", "envelope": None, **extra} def test_discovery_and_structured_output(self): good = bench_trace.detect_expectations([self.call(["--help"]), self.call(["check", "m.ws", "-f", "json"])], [], {}, None) @@ -171,6 +334,29 @@ def test_retry_storm_and_friction(self): friction = bench_trace.friction(events) self.assertEqual((friction["usageErrors"], friction["unknownSubcommands"], friction["identicalRepeats"]), (1, 1, 2)) + def test_friction_reports_unparseable_serve_responses_without_losing_errors(self): + responses = ['{"result":"unfinished', '{"error":{"code":"malformed-request"}}', '{"result":{}}'] + events = [{"tool": "wright", "type": "serve", "dir": "res", "line": line} for line in responses] + result = bench_trace.friction(events) + self.assertEqual(result["malformedServeRequests"], 1) + self.assertEqual(result["unparsedServeResponses"], 1) + + def test_serve_trace_preserves_large_json_responses(self): + import io + from types import SimpleNamespace + from unittest.mock import Mock, patch + response = json.dumps({"result": {"value": "x" * 4096}}) + "\n" + process = SimpleNamespace(stdin=io.BytesIO(), stdout=io.BytesIO(response.encode()), wait=lambda: 0) + output = SimpleNamespace(buffer=io.BytesIO(), flush=lambda: None) + with patch.object(bench_trace.subprocess, "Popen", return_value=process), \ + patch.object(bench_trace.sys, "stdin", SimpleNamespace(buffer=io.BytesIO())), \ + patch.object(bench_trace.sys, "stdout", output), \ + patch.object(bench_trace, "append_event", Mock()) as capture: + self.assertEqual(bench_trace.serve_tee("wright", ["serve"], 0), 0) + recorded = next(call.args[0] for call in capture.call_args_list if call.args[0]["type"] == "serve") + self.assertEqual(json.loads(recorded["line"]), json.loads(response)) + self.assertEqual(output.buffer.getvalue(), response.encode()) + def test_unused_wright_is_not_applicable(self): result = bench_trace.detect_expectations([], [], {"stabilityRisk": True}, None) self.assertEqual(result["E03"]["status"], "na") @@ -178,12 +364,13 @@ def test_unused_wright_is_not_applicable(self): class ReportTest(unittest.TestCase): - def result(self, cell, trial, usable, tokens, split=None): - wright, knowledge, network = cell.split("/") + def result(self, cell, trial, usable, tokens, split=None, scenario="s", language="opy", status="completed"): + tool = cell.split("/")[0].split("+")[0] return { - "contract": "wright-agent-bench/v2", "scenario": "s", "split": split, "_trial": trial, "_dir": Path("d"), - "condition": {"wright": wright, "knowledge": knowledge, "network": network}, "agent": {"id": "m", "exit": 0, "seconds": 1.0}, - "usable": usable, "passed": usable, "usage": {"totalTokens": tokens, "peakContext": tokens // 2}, "wrightUse": {"invocations": 1 if wright != "none" else 0}, + "contract": "wright-agent-bench/v3", "scenario": scenario, "family": "diagnosis", "language": language, "split": split, "_trial": trial, "_dir": Path("d"), + "condition": {"label": cell}, "agent": {"id": "m", "exit": 0, "seconds": 1.0}, "status": status, + "usable": usable, "passed": usable, "usage": {"totalTokens": tokens, "peakContext": tokens // 2}, + "toolUse": {"wright": {"invocations": 1}} if tool == "wright" else {}, } def test_wilson_interval(self): @@ -193,21 +380,31 @@ def test_wilson_interval(self): self.assertEqual(bench_report.wilson(0, 0), (0.0, 0.0)) def test_paired_efficiency_counts_only_both_usable_and_failures_cost(self): - runs = [self.result("none/none/off", 1, True, 1000), self.result("bin/none/off", 1, True, 600), - self.result("none/none/off", 2, False, 900), self.result("bin/none/off", 2, True, 700)] + runs = [self.result("none/none/off", 1, True, 1000), self.result("wright/none/off", 1, True, 600), + self.result("none/none/off", 2, False, 900), self.result("wright/none/off", 2, True, 700)] text, summary = bench_report.render(runs) self.assertIn("+1 / -0", text) self.assertIn("+40% tokens (n=1)", text) - self.assertEqual(summary["cells"]["m|bin/none/off"]["tokensPerUsable"], 650) + self.assertEqual(summary["cells"]["m|wright/none/off"]["tokensPerUsable"], 650) self.assertEqual(summary["cells"]["m|none/none/off"]["tokensPerUsable"], 1900) def test_headroom_and_invalid_runs_are_reported(self): runs = [self.result("none/none/off", t, True, 100) for t in range(1, 5)] - runs.append({**self.result("bin/none/off", 1, True, 100), "invalid": "canary"}) + runs.append({**self.result("wright/none/off", 1, True, 100, status="invalid"), "invalid": "canary"}) text, _ = bench_report.render(runs) self.assertIn("HEADROOM", text) self.assertIn("INVALID: 1 run(s) excluded", text) + def test_provider_failures_do_not_count_as_agent_failures(self): + good = self.result("none/none/off", 1, True, 100) + provider_failure = self.result("none/none/off", 2, False, 0) + provider_failure.update(status="provider-interrupted") + provider_failure["agent"]["exit"] = 75 + text, summary = bench_report.render([good, provider_failure]) + self.assertEqual(summary["cells"]["m|none/none/off"]["n"], 1) + self.assertEqual(summary["infrastructureFailures"], 1) + self.assertIn("excluded from outcome metrics", text) + if __name__ == "__main__": unittest.main() diff --git a/benchmarks/agent/test_score.py b/benchmarks/agent/test_score.py new file mode 100644 index 00000000..cb9ec03c --- /dev/null +++ b/benchmarks/agent/test_score.py @@ -0,0 +1,106 @@ +import json +import shutil +import tempfile +import unittest +from pathlib import Path + +import bench_score + + +def run(scenario, trial, usable, label=bench_score.CANONICAL, status="completed", split="test", language="opy", sha="a" * 64, model="m"): + return { + "scenario": scenario, "family": "diagnosis" if scenario.endswith("1") else "modification", "language": language, "split": split, + "condition": {"label": label}, "status": status, "usable": usable, "failedLayers": [] if usable else ["agent"], + "agent": {"id": "codex", "seconds": 10.0}, "agentInfo": {"model": model, "effort": "high"}, "protocol": {"timeoutSeconds": 60, "infraRetries": 0}, + "environment": {"wright": "wright 0.5.0", "wrightSha256": sha, "skills": {"wright-skill": {"sha256": "b" * 64}}, "suite": {"version": "v1", "hash": "c" * 64}, "harness": "abc"}, + "networkEnforcement": "declared-only", "fileWriteEnforcement": "trial-directory-only", "usage": {"totalTokens": 1000}, "_trial": trial, + } + + +SCENARIOS = [f"s{i}" for i in range(8)] + + +def runs(usable_scenarios, trials=3, **kw): + return [run(s, t, s in usable_scenarios, **kw) for s in SCENARIOS for t in range(1, trials + 1)] + + +class ScoreTest(unittest.TestCase): + def test_interval_is_deterministic_and_degenerate_when_nothing_varies(self): + outcomes = {"a": [1, 1, 1], "b": [0, 0, 0]} + self.assertEqual(bench_score.cluster_interval(outcomes), bench_score.cluster_interval(outcomes)) + lo, hi = bench_score.cluster_interval(outcomes) + self.assertTrue(0 <= lo <= 50 <= hi <= 100) + self.assertEqual(bench_score.cluster_interval({"a": [1, 1], "b": [1, 1]}), (100.0, 100.0)) + + def test_pass_power_k(self): + self.assertAlmostEqual(bench_score.pass_power_k(2, 3, 2), 1 / 3) + self.assertEqual(bench_score.pass_power_k(3, 3, 3), 1.0) + self.assertEqual(bench_score.pass_power_k(1, 3, 2), 0.0) + + def test_score_macro_averages_scenarios_with_equal_weight(self): + data = runs({"s0", "s1", "s2", "s3"}) + data += [run("s0", 4, True)] # an extra trial must not change a scenario's weight beyond its own rate + c = bench_score.card([r for r in data if r["scenario"] != "s0" or r["_trial"] <= 3], "opy", SCENARIOS) + self.assertEqual(c["score"], 50.0) + self.assertEqual(c["scenarios"], 8) + self.assertEqual(c["trialsPerScenario"], 3) + self.assertEqual(c["provisional"], []) + self.assertEqual(c["passPowK"], 50.0) + unequal = bench_score.card(data, "opy", SCENARIOS) + self.assertTrue(any("unequal valid trials" in p for p in unequal["provisional"])) + self.assertEqual(unequal["score"], 50.0) + + def test_only_canonical_test_runs_of_the_language_count(self): + data = runs({"s0"}) + runs(set(SCENARIOS), label="wright/none/off") + runs(set(SCENARIOS), split="train") + runs(set(SCENARIOS), language="workshop") + c = bench_score.card(data, "opy", SCENARIOS) + self.assertEqual(c["score"], 12.5) + self.assertEqual(c["validRuns"], 24) + + def test_interrupted_invalid_and_agent_error_runs_are_excluded_and_published(self): + data = runs({"s0", "s1"}) + data += [run("s0", 4, False, status="provider-interrupted"), run("s1", 4, False, status="invalid"), run("s2", 4, False, status="agent-error"), run("s3", 4, False, status="timeout")] + c = bench_score.card(data, "opy", SCENARIOS) + self.assertEqual(c["exclusions"], {"agent-error": 1, "invalid": 1, "provider-interrupted": 1}) + self.assertEqual(c["validRuns"], 25) # a timeout is the agent's own outcome and counts + + def test_runs_from_different_environments_are_refused(self): + mixed = runs({"s0"}) + [run("s0", 9, True, sha="d" * 64)] + c = bench_score.card(mixed, "opy", SCENARIOS) + self.assertIn("wrightSha256", c["refused"]) + self.assertIn("no score", bench_score.render(c)) + self.assertIn("model", bench_score.card(runs({"s0"}) + [run("s1", 9, True, model="other")], "opy", SCENARIOS)["refused"]) + + def test_small_suites_and_missing_scenarios_are_provisional(self): + c = bench_score.card(runs(set(SCENARIOS))[:21], "opy", SCENARIOS) + self.assertTrue(any("missing held-out scenarios" in p for p in c["provisional"])) + few = ["s0", "s1"] + small = bench_score.card([r for r in runs(set(SCENARIOS)) if r["scenario"] in few], "opy", few) + self.assertTrue(any("fewer than 8" in p for p in small["provisional"])) + self.assertIn("PROVISIONAL", bench_score.render(small)) + + def test_card_discloses_network_enforcement_and_identity(self): + c = bench_score.card(runs(set(SCENARIOS)), "opy", SCENARIOS) + text = bench_score.render(c) + self.assertEqual(c["networkEnforcement"], ["declared-only"]) + self.assertIn("declaration only", text) + for needle in ("Suite:", "Wright:", "Skills:", "wright-skill", "Model: m", "95% CI"): + self.assertIn(needle, text) + json.dumps(c) + + def test_main_writes_the_machine_readable_card_and_exit_status(self): + root = Path(tempfile.mkdtemp()) + self.addCleanup(shutil.rmtree, root, True) + for i, r in enumerate(runs(set(SCENARIOS))): + d = root / f"r{i}-{r['_trial']}" + d.mkdir() + payload = {k: v for k, v in r.items() if k != "_trial"} + (d / "result.json").write_text(json.dumps({"contract": "wright-agent-bench/v3", **payload})) + self.assertEqual(bench_score.main([root], ["opy"], {"opy": SCENARIOS}, None), 0) + card = json.loads((root / "score.json").read_text()) + self.assertEqual(card["contract"], "wright-agent-score/v1") + self.assertEqual(card["cards"][0]["score"], 100.0) + self.assertEqual(bench_score.main([root], ["workshop"], {"workshop": SCENARIOS}, None), 2) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmarks/agent/test_wiki.py b/benchmarks/agent/test_wiki.py new file mode 100644 index 00000000..cdf0708e --- /dev/null +++ b/benchmarks/agent/test_wiki.py @@ -0,0 +1,98 @@ +import argparse +import json +import shutil +import tempfile +import threading +import unittest +from http.server import BaseHTTPRequestHandler, HTTPServer +from pathlib import Path + +import agent_bench +import bench_wiki + +ARTICLES = { + "wait-until": b"---\ntitle: Wait Until\nupdated_at: 2026-01-01T00:00:00Z\ncontent_hash: aa\n---\n\n# Wait Until\nbody\n", + "count-of": b"---\ntitle: Count Of\nupdated_at: 2026-01-02T00:00:00Z\ncontent_hash: bb\n---\n\n# Count Of\nbody\n", +} + + +class Handler(BaseHTTPRequestHandler): + categories = {"actions": list(ARTICLES), "values": ["count-of"]} # count-of appears in two categories + + def do_GET(self): + path = self.path + if path.startswith("/wiki/categories/") and path.rsplit("/", 1)[1] in self.categories: + body = "\n".join(f"- [{s}](https://mirror.example/wiki/articles/{s})" for s in self.categories[path.rsplit("/", 1)[1]]).encode() + elif path.startswith("/wiki/articles/") and path.rsplit("/", 1)[1] in ARTICLES: + body = ARTICLES[path.rsplit("/", 1)[1]] + else: + self.send_error(404) + return + self.send_response(200) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *args): + pass + + +class WikiSnapshotTest(unittest.TestCase): + def setUp(self): + self.server = HTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=self.server.serve_forever, daemon=True).start() + self.addCleanup(self.server.server_close) + self.addCleanup(self.server.shutdown) + self.base = f"http://127.0.0.1:{self.server.server_port}" + self.tmp = Path(tempfile.mkdtemp()) + self.addCleanup(shutil.rmtree, self.tmp, True) + + def crawl(self, name: str) -> dict: + return bench_wiki.snapshot(self.base, self.tmp / name, categories=("actions", "values"), delay=0) + + def test_snapshot_crawls_categories_once_per_article_with_a_stable_identity(self): + first, second = self.crawl("a"), self.crawl("b") + self.assertEqual((self.tmp / "a/articles/wait-until.md").read_bytes(), ARTICLES["wait-until"]) + self.assertTrue((self.tmp / "a/NOTICE.txt").is_file()) + self.assertEqual([d["slug"] for d in first["documents"]], ["count-of", "wait-until"]) + self.assertEqual(next(d for d in first["documents"] if d["slug"] == "count-of")["categories"], ["actions", "values"]) + self.assertEqual(first["documents"][1]["contentHash"], "aa") + self.assertEqual(first["snapshotSha256"], second["snapshotSha256"]) + self.assertEqual(bench_wiki.identity(self.tmp / "a")["documents"], 2) + + def test_snapshots_are_pinned_and_slugs_are_checked(self): + self.crawl("a") + with self.assertRaises(SystemExit): + self.crawl("a") + Handler.categories["evil"] = ["../etc"] + self.addCleanup(Handler.categories.pop, "evil") + with self.assertRaises(SystemExit): + bench_wiki.snapshot(self.base, self.tmp / "c", categories=("evil",), delay=0) + with self.assertRaises(SystemExit): + bench_wiki.snapshot(self.base, self.tmp / "d", categories=("missing",), delay=0) + + def test_wiki_level_needs_a_snapshot(self): + cell = {"tool": "none", "skills": [], "knowledge": "wiki", "network": "off"} + with self.assertRaises(SystemExit): + agent_bench.check_cell(cell, argparse.Namespace(skill_dirs={}, wiki_dir=self.tmp)) + self.crawl("snap") + agent_bench.check_cell(cell, argparse.Namespace(skill_dirs={}, wiki_dir=self.tmp / "snap")) + + def test_pinned_snapshot_refuses_changed_missing_content_and_wrong_identity(self): + record = self.crawl("snap") + snap = self.tmp / "snap" + note = snap / "articles/wait-until.md" + for mutation in (lambda: note.write_text("changed"), lambda: note.rename(snap / "moved.md")): + with self.subTest(mutation=mutation): + note.write_bytes(ARTICLES["wait-until"]) + mutation() + with self.assertRaisesRegex(SystemExit, "snapshot content mismatch"): + bench_wiki.identity(snap) + note.write_bytes(ARTICLES["wait-until"]) + record["snapshotSha256"] = "0" * 64 + (snap / "SNAPSHOT.json").write_text(json.dumps(record)) + with self.assertRaisesRegex(SystemExit, "snapshot identity mismatch"): + bench_wiki.identity(snap) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmarks/agent/test_wiki_skill.py b/benchmarks/agent/test_wiki_skill.py new file mode 100644 index 00000000..699b7f32 --- /dev/null +++ b/benchmarks/agent/test_wiki_skill.py @@ -0,0 +1,119 @@ +import hashlib +import json +import shutil +import tempfile +import unittest +from pathlib import Path + +import wiki_skill + +CATALOG = { + "actions": [{"id": "playEffect", "aliases": {"en-US": "Play Effect"}}, {"id": "evaluateOnce", "aliases": {"en-US": "Evaluate Once"}}, + {"id": "createHudText", "aliases": {"en-US": "Create HUD Text"}}], + "values": [], "events": [], "enums": [{"domain": "HudPosition"}], +} +MANIFEST = {"aliases": [{"source": "evalOnce", "target": "evaluateOnce"}], "functions": [{"id": "playEffect"}, {"id": "evaluateOnce"}, {"id": "hudText", "catalogId": "createHudText"}]} +UPSTREAM = "playEffect evalOnce HudPosition hudText" + + +def article(title: str, slug: str, body: str) -> str: + return f"---\ntitle: {title}\ndescription: Workshop.code wiki article\nslug: {slug}\ntags: []\nupdated_at: 2026-07-10T18:15:21.022Z\ncontent_hash: abc\n---\n\n# {title}\n\n{body}\n" + + +class WikiSkillTest(unittest.TestCase): + def setUp(self): + self.tmp = Path(tempfile.mkdtemp()) + self.addCleanup(shutil.rmtree, self.tmp, True) + snap = self.tmp / "snap" + (snap / "articles").mkdir(parents=True) + docs = [ + ("Play Effect", "play-effect", ["actions"], "Plays an effect. It stops when the player dies."), + ("Evaluate Once", "evaluate-once", ["actions"], "Evaluates a value once."), + ("HUD Position", "hud-position", ["constants"], "Where a HUD text appears on screen."), + ("Unknown Thing", "unknown-thing", ["references"], "A reference table that has no catalog entry."), + ("Create HUD Text", "create-hud-text", ["actions"], "Create HUD Text"), + ] + records = [] + for title, slug, cats, body in docs: + raw = article(title, slug, body).encode() + (snap / "articles" / f"{slug}.md").write_bytes(raw) + records.append({"slug": slug, "categories": cats, "title": title, "updatedAt": "2026-07-10T18:15:21.022Z", "contentHash": "abc", "sha256": hashlib.sha256(raw).hexdigest()}) + identity_text = "\n".join(f"{d['slug']} {d['sha256']}" for d in sorted(records, key=lambda d: d["slug"])) + (snap / "SNAPSHOT.json").write_text(json.dumps({"snapshotSha256": hashlib.sha256(identity_text.encode()).hexdigest(), "documents": records})) + self.snap, self.out = snap, self.tmp / "workshop-wiki" + + def build(self, upstream: str = UPSTREAM) -> dict: + return wiki_skill.build(self.snap, self.out, CATALOG, MANIFEST, upstream) + + def test_layout_and_progressive_disclosure(self): + stats = self.build() + skill = (self.out / "SKILL.md").read_text() + self.assertIn("name: workshop-wiki", skill) + self.assertLess(len(skill.splitlines()), 40) + self.assertEqual(stats["articles"], 5) + categories = (self.out / "references/categories.md").read_text() + self.assertIn("actions.md", categories) + index = (self.out / "references/actions.md").read_text() + self.assertIn("[Play Effect](articles/play-effect.md)", index) + note = (self.out / "references/articles/play-effect.md").read_text() + self.assertNotIn("Workshop.code wiki article", note) + self.assertIn("overpy: playEffect", note) + self.assertIn("It stops when the player dies.", note) + + def test_overpy_spelling_is_upstream_name_and_only_when_verified(self): + stats = self.build() + index = (self.out / "references/actions.md").read_text() + self.assertIn("overpy `playEffect`", index) + self.assertIn("overpy `evalOnce`", index) # the upstream spelling, not the canonical id + self.assertNotIn("evaluateOnce", index.replace("Evaluate Once", "")) + self.assertIn("overpy `HudPosition`", (self.out / "references/constants.md").read_text()) + self.assertIn("overpy `hudText`", index) # through the manifest's catalogId, not the Workshop name + self.assertEqual(stats["overpy"], 4) + self.assertEqual(stats["catalogMatched"], 4) + + def test_unverified_spellings_are_dropped_not_guessed(self): + stats = self.build(upstream="playEffect") + self.assertEqual(stats["overpy"], 1) + self.assertEqual(stats["overpyUnverified"], 3) + self.assertNotIn("overpy `evalOnce`", (self.out / "references/actions.md").read_text()) + + def test_identity_is_stable_and_output_is_not_overwritten(self): + first = self.build() + with self.assertRaises(SystemExit): + self.build() + shutil.rmtree(self.out) + self.assertEqual(self.build()["skillSha256"], first["skillSha256"]) + with self.assertRaises(SystemExit): + wiki_skill.build(self.snap, self.tmp / "other-name", CATALOG, MANIFEST, UPSTREAM) + + def test_changed_snapshot_is_refused_before_writing_a_skill(self): + (self.snap / "articles/play-effect.md").write_text("changed") + with self.assertRaisesRegex(SystemExit, "snapshot content mismatch"): + self.build() + self.assertFalse(self.out.exists()) + + def test_skill_identity_refuses_changed_or_added_content(self): + record = self.build() + self.assertEqual(wiki_skill.identity(self.out), record) + note = self.out / "references/articles/play-effect.md" + original = note.read_text() + note.write_text("changed") + with self.assertRaisesRegex(SystemExit, "wiki skill content mismatch"): + wiki_skill.identity(self.out) + note.write_text(original) + (self.out / "references/extra.md").write_text("extra context") + with self.assertRaisesRegex(SystemExit, "wiki skill content mismatch"): + wiki_skill.identity(self.out) + + def test_summary_takes_the_first_sentence_without_markup(self): + self.assertEqual(wiki_skill.summary("> Source: https://x\n\nPlays an **effect**. It stops early."), "Plays an effect") + self.assertEqual(wiki_skill.summary("# T\n\nshort"), "") + self.assertEqual(wiki_skill.summary("Waits until the condition is true."), "Waits until the condition is true") + + def test_stub_articles_fall_back_to_their_syntax_line(self): + body = "# Create HUD Text\n\nCreate HUD Text\n\n```\nCreate HUD Text(All Players(All Teams), Custom String(\"x\"));\n```\n" + self.assertTrue(wiki_skill.summary(body, "Create HUD Text").startswith("syntax `Create HUD Text(")) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmarks/agent/wiki_skill.py b/benchmarks/agent/wiki_skill.py new file mode 100644 index 00000000..34689e4c --- /dev/null +++ b/benchmarks/agent/wiki_skill.py @@ -0,0 +1,163 @@ +"""Generate the `workshop-wiki` skill from a pinned wiki snapshot (#414, SPEC-414). + +The skill is community knowledge that works without Wright: a short SKILL.md, one index per category, and one file +per article. It is built deterministically from `SNAPSHOT.json`; nothing is written by hand except SKILL.md. Spellings +for OverPy come from the workshop-rs catalog and the opy-rs manifest and are kept only when the pinned upstream +compiler's own data contains them, so the skill never invents an alias. +""" + +from __future__ import annotations + +import hashlib +import json +import re +from datetime import datetime, timezone +from pathlib import Path + +import bench_grade +import bench_wiki + +KIND = {"actions": "action", "values": "value", "events": "event", "constants": "constant", "references": "reference"} +SKILL_NAME = "workshop-wiki" +DESCRIPTION = ( + "Use when you are unsure of the exact name, parameters, or behavior of an Overwatch Workshop action, value, event, " + "or constant, in raw Workshop script or OverPy, or hit a Workshop quirk such as timing, event semantics, or a HUD or " + "effect limit. Look it up in the local community wiki notes instead of guessing names or semantics." +) +SKILL_MD = f"""--- +name: {SKILL_NAME} +description: {DESCRIPTION} +--- + +# Workshop wiki notes + +Local benchmark material only. The source content remains subject to the Workshop.codes Terms of Service +(https://workshop.codes/tos); this generated skill grants no redistribution or AI-training permission. + +Community-written notes on Workshop actions, values, events, constants, and references, stored as one small file per +article. They are not exhaustive and may be out of date: check the `updated` date, and prefer a compiler or validator +result over a note when the two disagree. + +## Find a note without reading everything + +1. `references/categories.md` lists the categories (a few lines). +2. `references/.md` indexes one category, one line per article: title, the OverPy spelling when known, + last update, and a short summary. Search it instead of reading it: `grep -i "keyword" references/*.md`. +3. `references/articles/.md` is the note itself. Read only the ones you need. + +## Use them well + +- Notes use Workshop names (`Play Effect`). In OverPy the same thing is usually camelCase (`playEffect`); an index line + shows the OverPy spelling only when it could be verified, so a missing one means unknown, not absent. Control flow + (`If`, `Loop`), assignments (`Set Global Variable`, `Modify …`), and operators are OverPy syntax, not functions. +- A note describes behavior and pitfalls. It does not replace checking that your code is valid. +- Do not load a whole category index or many articles at once; look up what the task needs. +""" + + +def norm(text: str) -> str: + return re.sub(r"[^a-z0-9]", "", text.lower()) + + +def catalog_index(catalog: dict) -> dict[str, dict]: + """Normalized English name -> item, over actions, values, events, and enum domains.""" + index: dict[str, dict] = {} + for group in ("actions", "values", "events"): + for item in catalog[group]: + index.setdefault(norm(item["aliases"].get("en-US", item["id"])), {"id": item["id"], "group": group}) + for enum in catalog["enums"]: + index.setdefault(norm(enum["domain"]), {"id": enum["domain"], "group": "enums"}) + return index + + +def upstream_spellings(manifest: dict) -> dict[str, str]: + """Workshop catalog id -> the OverPy spelling of the pinned upstream compiler. + + The manifest maps a catalog id to its OverPy function id (they differ, for example `createHudText` and `hudText`), + and an alias maps a canonical id to the spelling upstream actually uses.""" + upstream = {a["target"]: a["source"] for a in manifest["aliases"]} + spellings: dict[str, str] = {} + for function in manifest["functions"]: + name = upstream.get(function["id"], function["id"]) + spellings.setdefault(function.get("catalogId", function["id"]), name) + return spellings + + +def verified(name: str, upstream_source: str) -> bool: + return re.search(rf"(? str: + """First sentence of the article text without markup; for a stub article, its first code snippet line.""" + for paragraph in re.split(r"\n\s*\n", re.sub(r"```.*?```", "", body, flags=re.S)): + text = re.sub(r"\s+", " ", re.sub(r"<[^>]+>|[`*_>#\[\]]|\(https?://[^)]*\)", "", paragraph)).strip() + if len(text) > 20 and text.lower() != title.lower() and not text.lower().startswith(("source:", "title:", "updated:")): + sentence = re.split(r"(?<=[.!?])\s", text)[0] + return sentence[:110].rstrip(" .,;:") + ("…" if len(sentence) > 110 else "") + snippet = re.search(r"```[^\n]*\n(.+?)\n", body) + return f"syntax `{snippet.group(1).strip()[:90]}`" if snippet else "" + + +def article_file(doc: dict, raw: str, opy: str | None, catalog_id: str | None) -> str: + body = raw.split("\n---\n", 1)[1].lstrip("\n") if raw.startswith("---\n") else raw + head = [f"title: {doc['title']}", f"category: {', '.join(doc['categories'])}", f"updated: {(doc['updatedAt'] or '')[:10]}", f"content_hash: {doc['contentHash']}"] + if catalog_id: + head.append(f"catalog_id: {catalog_id}") + if opy: + head.append(f"overpy: {opy}") + return "---\n" + "\n".join(head) + "\n---\n\n" + body + + +def build(snapshot: Path, out: Path, catalog: dict, manifest: dict, upstream_source: str) -> dict: + if out.name != SKILL_NAME: + raise SystemExit(f"the output directory must be named {SKILL_NAME!r} so it installs under that skill name") + if out.exists(): + raise SystemExit(f"{out} exists; generated skills are not overwritten") + record = bench_wiki.load_snapshot(snapshot) + by_name, spelling = catalog_index(catalog), upstream_spellings(manifest) + (out / "references/articles").mkdir(parents=True) + index: dict[str, list[str]] = {} + stats = {"articles": 0, "catalogMatched": 0, "overpy": 0, "overpyUnverified": 0} + for doc in record["documents"]: + raw = (snapshot / "articles" / f"{doc['slug']}.md").read_text() + match = by_name.get(norm(doc["title"] or doc["slug"])) + opy = None + if match: + candidate = spelling.get(match["id"], match["id"] if match["group"] in ("enums", "events") else None) + if candidate and verified(candidate, upstream_source): + opy = candidate + elif candidate: + stats["overpyUnverified"] += 1 + (out / "references/articles" / f"{doc['slug']}.md").write_text(article_file(doc, raw, opy, match["id"] if match else None)) + body = raw.split("\n---\n", 1)[1] if raw.startswith("---\n") else raw + line = f"- [{doc['title']}](articles/{doc['slug']}.md)" + (f" · overpy `{opy}`" if opy else "") + f" · {(doc['updatedAt'] or '')[:7]} · {summary(body, doc['title'] or '')}" + for category in doc["categories"]: + index.setdefault(category, []).append(line) + stats["articles"] += 1 + stats["catalogMatched"] += bool(match) + stats["overpy"] += bool(opy) + for category, lines in index.items(): + (out / f"references/{category}.md").write_text(f"# {category} ({len(lines)})\n\n" + "\n".join(sorted(lines, key=str.lower)) + "\n") + (out / "references/categories.md").write_text("# Categories\n\n" + "\n".join(f"- [{c}]({c}.md): {len(index[c])} {KIND.get(c, c)} notes" for c in sorted(index)) + "\n") + (out / "SKILL.md").write_text(SKILL_MD) + build_record = {"name": SKILL_NAME, "snapshotSha256": record["snapshotSha256"], "builtAt": datetime.now(timezone.utc).isoformat(timespec="seconds"), "skillSha256": content_hash(out), **stats} + (out / "BUILD.json").write_text(json.dumps(build_record, indent=2) + "\n") + return build_record + + +def content_hash(skill_dir: Path) -> str: + return hashlib.sha256("".join(f"{p.relative_to(skill_dir)}{hashlib.sha256(p.read_bytes()).hexdigest()}" for p in sorted(skill_dir.rglob("*.md"))).encode()).hexdigest() + + +def identity(skill_dir: Path) -> dict: + record = json.loads((skill_dir / "BUILD.json").read_text()) + if record["name"] != SKILL_NAME or content_hash(skill_dir) != record["skillSha256"]: + raise SystemExit(f"wiki skill content mismatch: {skill_dir}") + return record + + +def upstream_source_text() -> str: + path = bench_grade.ORACLE / "node_modules/overpy/overpy.js" + if not path.is_file(): + raise SystemExit("upstream oracle not installed; run `agent_bench.py setup-oracle`") + return path.read_text() diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 1d672960..65989ea6 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -1,6 +1,6 @@ # Agent Benchmark -- Contract: `wright-agent-bench/v2` +- Contracts: `wright-agent-bench/v3` (a run result) and `wright-agent-score/v1` (a score card) - Harness: [`benchmarks/agent/agent_bench.py`](../benchmarks/agent/agent_bench.py) - Design and requirements: [`SPEC-414`](specs/SPEC-414-agent-benchmark-comparison.md) @@ -15,8 +15,8 @@ evidence of correctness (use `--trials`). - The scenario workspace: the seed project only. - The scenario prompt, delivered on the agent's stdin. It states the requirement in user terms and never names Wright commands or Workshop APIs. -- In the `bin` condition, the released `wright` CLI and `wright serve` - session on `PATH`. +- In the `wright` tool condition, the released `wright` CLI and `wright serve` + session on `PATH`; in the `overpy` tool condition, the `overpy` compiler. Not allowed in the primary condition: a Workshop/OverPy/OSTW system prompt or skill pack, a generated API reference, or task-specific hints. The agent, @@ -25,29 +25,67 @@ on a vendor. Other conditions below are experiments and are labeled as such. ## Conditions -A cell is `//`. The task and prompt are identical -in every cell. +A cell is `tool[+skill...]/knowledge/network`, for example `wright+wright-skill/none/off`; +the baseline is `none/none/off`. The task and prompt are identical in every cell. | Factor | Levels | | --- | --- | -| `wright` | `none`: no directory providing `wright` is on `PATH`. `bin`: `wright` on `PATH` through a tracing shim. `bin+skill`: `bin`, plus the guide directory given by `--skill-dir`, which the adapter installs. | -| `knowledge` | `none`; `wiki`: `--wiki-dir` is linked read-only as `./wiki` (never counted as an edit); `web`: the adapter enables its web tools. | +| `tool` | `none`: neither tool is reachable. `wright`: `wright` on `PATH` through a tracing shim. `overpy`: the pinned `overpy` compiler through a tracing shim (OverPy scenarios only). A tool that is not part of the condition is hidden from `PATH`, and a canary fails the run if it is reachable. | +| `skills` | Any of `wright-skill` (how to use Wright), `workshop-skill` (the progressive wiki knowledge skill, see below), `opy-skill` (how to write OverPy and use `overpy`; OverPy scenarios only), and `workshop-format-skill` (the raw Workshop source format; Workshop scenarios only, a local benchmark control). Each is given by `--skill-dir NAME=DIR` and installed through the agent's skill mechanism. | +| `knowledge` | `none`; `wiki`: a pinned snapshot given by `--wiki-dir`, copied into the workspace as `./wiki` (a real copy, because tools such as `rg` do not follow symlinks; never counted as an edit); `web`: the adapter enables its web tools. | | `network` | `off` or `on`; `web` requires `on`. | +Language decides which cells apply: the `overpy` tool, `opy-skill`, and +`workshop-format-skill` apply only to scenarios of their language, and `matrix` +skips the other pairs and prints them as not applicable. They are not failures +and are not counted. + +`agent_bench.py wiki-snapshot [--dir DIR]` builds the `wiki` snapshot from the +mirror at `md.wrightkit.dev`: it lists category pages and fetches each distinct article once +(through `curl`, because the mirror rejects Python's HTTP client with 403), and +writes `articles/`, `NOTICE.txt`, and `SNAPSHOT.json` with per-document +hashes and a `snapshotSha256`. A snapshot is never overwritten, a run with +`knowledge` `wiki` verifies the article hashes and snapshot identity before starting, +and the result records the verified identity in `environment.wiki`. +The default categories are actions, values, events, constants, and references; +add `tutorials` through `--categories` for a separate second-tier experiment. +The mirror's manifest is incomplete and is not the crawl source. + +The `workshop-skill` is built by `agent_bench.py wiki-skill`. It is a separate skill (generated name `workshop-wiki`) with a short `SKILL.md`, category +indexes, and individual articles. It takes a pinned snapshot, the workshop-rs catalog, +and the opy-rs manifest; OverPy spellings are included only when found in the pinned +upstream oracle. The generated skill is community guidance, not canonical semantic +authority. Its content hash is recorded with its build record in `environment.skills`; content that +differs from its `BUILD.json` is refused. The `./wiki` copy is not read-only, and `./wiki` +is not counted as an edit, so edits there are not detected. + +```sh +python3 benchmarks/agent/agent_bench.py wiki-skill \ + --snapshot /abs/path/pinned-wiki --out-dir /abs/path/local/workshop-wiki \ + --catalog /abs/path/workshop-rs/crates/workshop-rs/src/catalog/data/catalog.json \ + --opy-manifest /abs/path/opy-rs/crates/opy-rs/src/manifest/data/manifest.json +``` + +The output directory must be named `workshop-wiki` and must not exist. Run +`setup-oracle` first. Snapshots and derived skills are local benchmark material; +do not commit or distribute them. The [Workshop.codes Terms of Service](https://workshop.codes/tos) +apply to the source content; generating a skill grants no additional permission. + Each run is scrubbed: a fresh `HOME`, an allowlisted environment (`--env-pass` names host variables to keep), and no host instruction files. Two canaries run before the agent; a failed canary marks the run `invalid` and it is excluded -from results: `wright` must not be reachable under level `none`, and +from results: a tool outside the condition must not be reachable, and `--canary-cmd` (a command that must fail when the network is `off`) must fail -in the agent environment. A determined agent can still find a Wright binary +in the agent environment. A determined agent can still find a tool binary elsewhere on disk, so run `none` in a clean environment when that matters. ## Adapters `--agent-cmd` is a shell command run in the workspace with the prompt on stdin. The harness describes the cell through environment variables, and the adapter -enforces it: `BENCH_WRIGHT`, `BENCH_KNOWLEDGE`, `BENCH_NETWORK`, -`BENCH_SKILL_DIR` (only for `bin+skill`), `BENCH_HOST_PATH` (the unscrubbed +enforces it: `BENCH_TOOL`, `BENCH_SKILLS` (names), `BENCH_SKILL_DIRS` (one +directory per installed skill, separated by the path separator), `BENCH_KNOWLEDGE`, +`BENCH_NETWORK`, `BENCH_HOST_PATH` (the unscrubbed `PATH`, for locating the agent binary itself; do not pass it to the agent), and `BENCH_RUN_DIR`. The adapter reports, all optional: @@ -55,13 +93,65 @@ enforces it: `BENCH_WRIGHT`, `BENCH_KNOWLEDGE`, `BENCH_NETWORK`, | --- | --- | | `BENCH_USAGE` | JSONL, one row per model turn: `t` (epoch seconds), `input`, `output`, `cache_read`, `cache_write`, `reasoning`, `context`, `context_limit` | | `BENCH_TRANSCRIPT` | normalized JSONL of the agent's events | -| `BENCH_CONTEXT` | `{"loaded": [...]}`; a loaded item other than the expected guide invalidates the run | +| `BENCH_CONTEXT` | `{"loaded": [...]}`; a loaded item other than the condition's skills (by the name in each `SKILL.md`) invalidates the run | +| `BENCH_AGENT_INFO` | `{"agent", "model", "effort", "tools", ...}`: the agent system actually used and the tools it exposed, copied into the result as `agentInfo`, so harness differences can be disclosed | + +An adapter that cannot observe loaded context omits `loaded` and reports its +audit limitation instead. The harness and report mark that field unreported; +installing a skill is not evidence that the agent loaded it. Built-in skills +belong to each agent's default context and are recorded separately when observable. Exit code 75 means a provider or infrastructure failure: the harness retries -the trial (`--infra-retries`) and records the retries. Any other non-zero exit -is reported as an agent or infrastructure failure in the report diagnostics. -[`adapters/claude_code.py`](../benchmarks/agent/adapters/claude_code.py) is the -reference adapter. +the trial (`--infra-retries`) and records the retries. Every result has a +`status`: `completed`, `provider-interrupted` (exit 75 after the retries), +`timeout`, `agent-error` (any other non-zero exit), or `invalid` (a canary, an +unexpected loaded context, or an unavailable required grader). + +The scenario `prompt.md` is the only text the harness gives an agent: it adds no +system prompt, hint, or Workshop context. Each agent keeps its own default system +prompt, which is part of what is measured. The adapter, not the harness, must +keep host configuration out of the run, and it reports what loaded through +`BENCH_CONTEXT`. + +| Adapter | Agent | Notes | +| --- | --- | --- | +| [`claude_code.py`](../benchmarks/agent/adapters/claude_code.py) | Claude Code | `BENCH_MODEL` selects the model. Removes web tools unless knowledge is `web`. | +| [`pi.py`](../benchmarks/agent/adapters/pi.py) | pi | `BENCH_MODEL=provider/id`. Disables context files, extensions, and skill discovery, and loads the guide explicitly. Providers registered by an extension need `BENCH_PI_EXTENSIONS`; web tools come from `BENCH_PI_WEB_EXTENSIONS`. | +| [`devin.py`](../benchmarks/agent/adapters/devin.py) | Devin CLI | `BENCH_MODEL` (for example `swe-2-max`). Runs with an isolated `HOME` holding only the Devin credentials, a config that reads no other tool's rules or skills, and MCP tools denied. Managed plugin skills are listed apart from `loaded`. | +| [`codex.py`](../benchmarks/agent/adapters/codex.py) | Codex CLI | `BENCH_MODEL` and `BENCH_THINKING`. Isolated `HOME`/`CODEX_HOME`, user config and exec rules ignored, workspace skills installed under `.agents/skills`. Session token events supply per-model-call usage and observed skill context; built-ins are listed apart. Web search, Apps, and plugin discovery are disabled; any observed MCP call invalidates the trial. | +| [`agy.py`](../benchmarks/agent/adapters/agy.py) | Antigravity CLI | `BENCH_MODEL` (for example `gemini-3.8-flash-high`) and `BENCH_THINKING`. Isolated `HOME` with only authentication files, workspace skills under `.agents/skills`, MCP/browser access denied and URL reads denied outside `web`. Observed built-in web tools under network `off` stop and invalidate the trial; URL permissions do not cover search. Streaming step usage is recorded. The CLI does not export observed loaded skills or context limits; these remain unreported. | + +The pi adapter also uses an isolated `HOME` containing only its authentication +files. Explicit provider extensions remain referenced by path, not copied with +host settings. Reasoning is counted once: pi, Codex, and Antigravity include reasoning +in their output counters, so adapters split it out before aggregation. Antigravity +reports uncached input separately from cached reads; Codex reports inclusive input. + +Set the model in `--agent-cmd` (the environment is scrubbed), and give the +adapter and skill paths as absolute paths, because the agent runs in the +workspace: `--agent-cmd "BENCH_MODEL=openai-codex/gpt-6-luna python3 /abs/path/adapters/pi.py"`. +Pass `--env-pass HOME` when the agent authenticates from the real home directory. + +Two checks protect the context. The workspace must not sit below a directory +that holds instruction files (`AGENTS.md`, `CLAUDE.md`, and similar), because +agents discover them by walking up; the default `--out` is +`~/.cache/wright-agent-bench` for that reason, and a violation marks the run +`invalid` (`--no-ancestor-check` disables it). Network `off` is enforced only +when `--canary-cmd` is given and fails inside the agent environment; without it +the result records `networkEnforcement: declared-only`, which is what the shell +tools of pi and Devin provide today. + +On macOS, `--file-sandbox` applies `sandbox-exec` to the adapter and all descendant +processes: filesystem writes are restricted to that trial's output directory +(plus device streams), and `TMPDIR` points inside it. Unsupported hosts fail +instead of silently running without protection. This protects host files; it +blocks reads of `AGENTS.md`, `CLAUDE.md`, and `GEMINI.md` outside the trial workspace +to prevent tool-path rule discovery from contaminating context. Other host reads +remain possible, and network isolation is not enforced. Model +account usage, CPU and disk consumption remain shared with the host. Provider +failures returned as exit 75 are listed separately and excluded from outcome +metrics. The harness never edits the task prompt: network `off` and the workspace +boundary are enforced or disclosed, not stated to the agent. ## Scenarios @@ -72,7 +162,7 @@ to form a passing solution), and optional `negative//` overlays. | Field | Meaning | | --- | --- | -| `id`, `family`, `language` | Identity; `language` is `workshop`, `opy`, or `ostw`, and a scenario may use a language only once its owner declares the needed capability supported | +| `id`, `family`, `language` | Identity; `language` is `workshop`, `opy`, or `ostw`; a scenario may use a language only once its owner declares the needed capability supported, and `language` selects its score track and which language tools and skills apply | | `entry` | Source file that Wright checks and that is snapshotted after each write (`watch` overrides the file list) | | `writable` | Files the agent may change; any other change is reported as an unsafe edit | | `runtimeOnly` | Claims that only the Overwatch runtime can verify; reported as unverified, never as passed | @@ -119,16 +209,34 @@ job; running agents does not. ```sh python3 benchmarks/agent/agent_bench.py run --agent-id LABEL --agent-cmd CMD \ - --wright-level bin --knowledge none --network off --trials 5 + --tool wright --skills wright-skill --skill-dir wright-skill=DIR \ + --knowledge none --network off --trials 5 python3 benchmarks/agent/agent_bench.py matrix matrix.json # agents x cells x scenarios x trials -python3 benchmarks/agent/agent_bench.py report target/agent-bench [--regrade] +python3 benchmarks/agent/agent_bench.py report target/agent-bench [--regrade] [--reference none/none/off] +python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agent Score cards ``` [`matrix.example.json`](../benchmarks/agent/matrix.example.json) is the Tier 1 matrix. `matrix.json` lists `agents` (`{id, cmd}`), `cells`, optional `scenarios`, `trials`, `parallel`, `seed` (run order is shuffled by it), and `options` -(`skill_dir`, `wiki_dir`, `env_pass`, ...). Finished runs are skipped, so an +(`skill_dirs` as `{name: dir}`, `wiki_dir`, `env_pass`, ...). Cells not applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs are skipped, so an interrupted matrix resumes. +For a local offline-declared pilot, use +[`matrix.pilot.example.json`](../benchmarks/agent/matrix.pilot.example.json): one +agent at a time, five cells including `workshop-skill`, two scenarios with +different requirement families, three trials, and a 3600-second timeout. Replace +the absolute path placeholders and choose the adapter/model before running. This +is 30 trials per agent, not the full-suite evaluation. Start with a single trial +to check cost, usage, `context.loaded`, and `networkEnforcement`, then use a fresh +output directory for the randomized matrix. Complete pi GPT before switching to +pi Gemini and then Devin; Gemini needs its provider extension. + +This pilot omits `web`; it does not establish network isolation when enforcement +is `declared-only`. It measures baseline headroom and variance before guide tuning. +Skill-retrieval attribution (files and content tokens read), wiki-only scenarios, +and web URL contamination analysis remain separate follow-up work; a successful +pilot does not establish those requirements. + ## Result `agent_bench.py run` writes `///-/result.json`, @@ -136,18 +244,19 @@ with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | Field | Meaning | | --- | --- | -| `condition`, `agent`, `environment`, `grader` | Cell, agent label/command/exit/seconds, OS/Python/Wright version, grader hash and oracle version | -| `checks`, `passed`, `usable`, `failedLayers` | Per-check outcome; `usable` is `passed` with no error-severity lint finding | +| `condition`, `agent`, `agentInfo`, `protocol`, `status` | Cell with its `label`; agent label/command/exit/seconds; the system and tools the adapter reports; timeout and infra retries; `completed`, `provider-interrupted`, `timeout`, `agent-error`, or `invalid` | +| `environment`, `grader` | OS/Python/Wright version and sha256, harness commit, suite version and hash, skill identities (`skills`), wiki identity, grader hash and oracle version | +| `checks`, `passed`, `usable`, `failedLayers` | Per-check outcome; `usable` is `passed` with no error-severity lint finding and no unsafe edits; `usableReason` names the blocking cause | | `authorities`, `disagreement` | Validity per authority and any Wright/oracle disagreement | | `diagnostics`, `lintFindings`, `unsafeEdits` | Remaining `wright check` diagnostics, lint rule codes, files changed outside `writable` | | `unverifiedRuntimeClaims` | The scenario's `runtimeOnly` claims | -| `wrightUse` | Invocations by subcommand, failures, exits of 3 or 4 (candidate owner or environment gaps), and estimated output tokens per command | -| `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, identical repeats; expectation E01-E12 verdicts | +| `toolUse` | Per tool (`wright`, `overpy`): invocations by subcommand, failures, exits of 3 or 4 (candidate owner or environment gaps), and estimated output tokens per command | +| `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, unparsed `serve` responses, identical repeats; expectation E01-E12 verdicts | | `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | | `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | -| `invalid`, `infraRetries` | Present when the run was excluded or retried | +| `invalid`, `infraRetries`, `networkEnforcement` | Present when the run was excluded or retried; whether network `off` was checked by a canary or only declared | -`wrightUse` is recorded per CLI invocation; `wright serve` sessions are teed +`toolUse` is recorded per CLI invocation through the shim; `wright serve` sessions are teed line by line into the trace. Comparing cells for the same scenario shows what Wright adds. Exit 3 or 4 entries, disagreements, and failed `layer` values are the input for owner Issues. @@ -159,14 +268,28 @@ output use four bytes per token; provider-reported usage is authoritative. ## Report `agent_bench.py report` writes `report.md` and `summary.json`: usable and passed -counts with Wilson 95% intervals, tokens per run and per usable result, peak -context, paired comparison against `none/none/off` (same scenario, agent, and +counts with scenario-clustered 95% intervals (resampling scenarios, then trials), tokens per run and per usable result, peak +context, paired comparison against `--reference` (default `none/none/off`) (same scenario, agent, and trial; token comparison only where both are usable), per-scenario and per-split tables, expectation rates, friction, output size per command, and diagnostics. Diagnostics flag headroom (baseline usable rate of at least 95%), infrastructure failures, invalid runs, trial variance, and, with `--regrade`, a grader that gives different verdicts on the same stored workspace. +## Score + +`agent_bench.py score` writes `score.json` and `score.txt`: one card per language +track (Wright Workshop Agent Score, Wright OPY Agent Score; contract +`wright-agent-score/v1`). A card uses only the canonical cell +`wright+wright-skill/none/off` on `test` scenarios, macro-averages the usable rate +over scenarios, and gives a two-stage bootstrap 95% interval (10,000 draws, seed +467) and Pass^k as a secondary figure. Provider-interrupted, invalid, and +agent-error runs are published as exclusions; timeouts count. The score is +refused if the runs differ in Wright binary, skill hashes, suite hash, agent, +model, effort, or protocol, and it is marked provisional with fewer than eight +held-out scenarios, missing scenarios, or unequal trials. The card discloses +`networkEnforcement` (`declared-only` or `canary-checked`). + ## Cadence The benchmark does not gate pull requests. Run it manually or on a schedule once diff --git a/docs/specs/SPEC-414-agent-benchmark-comparison.md b/docs/specs/SPEC-414-agent-benchmark-comparison.md index b306e5aa..299a8938 100644 --- a/docs/specs/SPEC-414-agent-benchmark-comparison.md +++ b/docs/specs/SPEC-414-agent-benchmark-comparison.md @@ -39,13 +39,19 @@ Pilot observations that shaped this spec (Sonnet only, 2 scenarios, 18 runs; not ### Conditions -- REQ-001: A run is defined by (Wright level, knowledge level, tool network). Wright level is `none`, `bin` - (on PATH through the tracing shim), or `bin+skill` (plus the guide from a pinned `wrightkit/skills` commit, - installed through the agent's own skill mechanism). Knowledge level is `none`, `wiki` (a read-only local - snapshot of the Workshop wiki Markdown mirror, pinned by hash, with no added index or tool), or `web` - (standard fetch and search tools). Tool network is `off` or `on`; `web` implies `on`. -- REQ-002: Tier 1 conditions are `none/none`, `bin/none`, `bin+skill/none`, `none/web`, `none/wiki`. Tier 2 is - `bin+skill/web`, `bin+skill/wiki`, `bin/wiki`. The task and prompt are identical across conditions. +- REQ-001: A run is defined by (tool, skills, knowledge, network), labelled `tool[+skill...]/knowledge/network`. + Tool is `none`, `wright` (on PATH through the tracing shim), or `overpy` (the pinned compiler through the + same shim; OverPy scenarios only). Skills is any subset of `wright-skill` and `workshop-skill` (the local + progressive wiki skill, pinned by snapshot and content hashes, usable without Wright), plus `opy-skill` + (OverPy only) and `workshop-format-skill` (Workshop only), each installed through the agent's own skill + mechanism. Knowledge is `none`, `wiki` (a local snapshot copied into the workspace, pinned by hash, with no + added index or tool), or `web` (standard fetch and search tools). Network is `off` or `on`; `web` implies + `on`. Pairs not applicable to a scenario's language are skipped, not failed. +- REQ-002: Tier 1 cells are `none/none/off`, `wright/none/off`, `wright+wright-skill/none/off`, + `none/web/on`, `none/wiki/off`, `workshop-skill/none/off`; OverPy scenarios add `overpy/none/off` and + `overpy+opy-skill/none/off` as language-appropriate controls. The canonical score cell is + `wright+wright-skill/none/off` (`wright-agent-score/v1`, per language track, defined in + `docs/agent-benchmark.md`). The task prompt is the scenario's `prompt.md`, identical across conditions. - REQ-003: Each run passes a canary before the agent starts, and a failed canary invalidates the run: `none` Wright means `command -v wright` fails and no install path is reachable; `off` means an attempted fetch from the agent's tool environment fails while the model channel still works. @@ -202,6 +208,11 @@ Pilot observations that shaped this spec (Sonnet only, 2 scenarios, 18 runs; not - Q-003 [verification]: rubric-based judgment (REQ-011) in scope for the first version, or deferred; owner QA. - Q-004 [architecture]: the harness now lives in `benchmarks/agent` as small modules beside `agent_bench.py` (grading, trace, report, adapters); confirm or redirect; owner Architect. -- Q-005 [verification]: the source of the wiki snapshot and its license and pinning method; owner QA. +- Q-005 [verification]: wiki snapshots use the mirror's category pages and verified per-article content hashes; + derived skills also verify their content hash. Both remain local under the source terms. Any distribution + needs explicit source permission and an Architect ownership decision; it is outside this benchmark batch. - Q-006 [verification]: the common tokenizer used for cross-model attribution (REQ-021) and how its error is reported; owner QA. +- Q-007 [verification]: skill-retrieval metrics beyond aggregate usage (unique files read, repeated reads, + content tokens read, and how to judge which articles were needed) require transcript normalization and a + defined estimator. Decide from the baseline pilot before adding requirements or tuning either guide.