diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e7224cb..0c61f95 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,10 +16,12 @@ on: paths-ignore: - "**/*.md" - "docs/**" + - "!benchmarks/**/*.md" pull_request: paths-ignore: - "**/*.md" - "docs/**" + - "!benchmarks/**/*.md" permissions: contents: read @@ -58,6 +60,7 @@ jobs: - 'crates/**' - 'tests/fixtures/**' - 'scripts/**' + - 'benchmarks/**' - '.github/workflows/**' dist: - 'dist/**' @@ -456,6 +459,12 @@ jobs: id: bench run: target/debug/wright-bench + - name: Validate agent benchmark scenarios + run: | + cargo build --locked -p wright-cli + python3 benchmarks/agent/agent_bench.py validate + python3 -m unittest discover -s benchmarks/agent + - name: Upload benchmark report if: always() uses: actions/upload-artifact@v7 diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py new file mode 100644 index 0000000..cc86940 --- /dev/null +++ b/benchmarks/agent/agent_bench.py @@ -0,0 +1,256 @@ +#!/usr/bin/env python3 +"""Wright agent benchmark harness (#414). Contract: docs/agent-benchmark.md.""" + +from __future__ import annotations + +import argparse +import json +import os +import platform +import shutil +import signal +import subprocess +import sys +import time +from datetime import datetime, timezone +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parent.parent +SCENARIOS = HERE / "scenarios" +RESULT_CONTRACT = "wright-agent-bench/v1" +SHIM_ENV = ("WRIGHT_BENCH_REAL", "WRIGHT_BENCH_TRACE") + + +def load_scenario(scenario_id: str) -> dict: + directory = SCENARIOS / scenario_id + scenario = json.loads((directory / "scenario.json").read_text()) + scenario["dir"] = directory + return scenario + + +def all_scenario_ids() -> list[str]: + return sorted(p.name for p in SCENARIOS.iterdir() if (p / "scenario.json").is_file()) + + +def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: + proc = subprocess.run([wright, *args, "--format", "json"], capture_output=True, text=True) + try: + return proc.returncode, json.loads(proc.stdout) + except json.JSONDecodeError: + return proc.returncode, {"diagnostics": [{"code": "harness-output", "severity": "error", "message": proc.stderr.strip() or proc.stdout.strip()}]} + + +def serve_request(wright: str, entry: Path, request: dict) -> dict: + proc = subprocess.run([wright, "serve", str(entry)], input=json.dumps(request) + "\n", capture_output=True, text=True) + try: + return json.loads(proc.stdout.splitlines()[0]) + except (IndexError, json.JSONDecodeError): + return {"error": {"code": "harness-output", "message": proc.stderr.strip()}} + + +def check_result(check: dict, passed: bool, detail: str) -> dict: + return {"id": check["id"], "kind": check["kind"], "layer": check.get("layer", "agent"), "passed": passed, "detail": detail} + + +def run_check(check: dict, workspace: Path, entry: Path, wright: str, state: dict) -> dict: + kind = check["kind"] + if kind == "check": + code, envelope = wright_json(wright, ["check", str(entry)]) + state["diagnostics"] = envelope.get("diagnostics", []) + errors = [d for d in state["diagnostics"] if d.get("severity") == "error"] + return check_result(check, code == 0 and not errors, f"exit {code}, {len(errors)} error diagnostic(s)") + if kind == "lint": + _, envelope = wright_json(wright, ["lint", str(entry)]) + findings = [f for f in envelope.get("result", {}).get("findings", []) if f["code"] == check["code"]] + return check_result(check, len(findings) <= check.get("max", 0), f"{len(findings)} '{check['code']}' finding(s)") + if kind == "symbols": + response = serve_request(wright, entry, {"op": "symbols", "kind": check["symbolKind"]}) + found = len(response.get("result", [])) + return check_result(check, found >= check.get("min", 1), f"{found} '{check['symbolKind']}' symbol(s)") + if kind in ("contains", "absent"): + path = workspace / check["file"] + text = path.read_text() if path.is_file() else "" + texts = check["text"] if isinstance(check["text"], list) else [check["text"]] + count = sum(text.count(t) for t in texts) + if kind == "contains": + passed = count >= check.get("min", 1) and count <= check.get("max", count) + else: + passed = count == 0 + return check_result(check, passed, f"{count} occurrence(s) in {check['file']}") + if kind == "answer": + path = workspace / "answer.json" + try: + actual = json.loads(path.read_text()).get(check["key"]) + except (OSError, json.JSONDecodeError, AttributeError): + actual = None + return check_result(check, actual == check["expected"], f"answer[{check['key']}] = {json.dumps(actual)}") + raise SystemExit(f"unknown check kind '{kind}'") + + +def tree(root: Path) -> dict[str, bytes]: + return {str(p.relative_to(root)): p.read_bytes() for p in sorted(root.rglob("*")) if p.is_file()} + + +def unsafe_edits(scenario: dict, workspace: Path) -> list[str]: + seed = tree(scenario["dir"] / "seed") + now = tree(workspace) + writable = set(scenario["writable"]) + return sorted(name for name in seed.keys() | now.keys() if seed.get(name) != now.get(name) and name not in writable) + + +def grade(scenario: dict, workspace: Path, wright: str) -> dict: + entry = workspace / scenario["entry"] + state: dict = {} + checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] + return { + "checks": checks, + "passed": all(c["passed"] for c in checks), + "failedLayers": sorted({c["layer"] for c in checks if not c["passed"]}), + "diagnostics": state.get("diagnostics", []), + "unsafeEdits": unsafe_edits(scenario, workspace), + "unverifiedRuntimeClaims": scenario.get("runtimeOnly", []), + } + + +def materialize(scenario: dict, workspace: Path, overlay: str | None = None) -> None: + shutil.copytree(scenario["dir"] / "seed", workspace) + if overlay: + shutil.copytree(scenario["dir"] / overlay, workspace, dirs_exist_ok=True) + + +def validate(wright: str, out: Path) -> bool: + """Each scenario must be solvable by the reference and unsolved by its seed.""" + ok = True + for scenario_id in all_scenario_ids(): + scenario = load_scenario(scenario_id) + results = {} + for name, overlay in (("seed", None), ("reference", "reference")): + workspace = out / scenario_id / name + shutil.rmtree(workspace, ignore_errors=True) + materialize(scenario, workspace, overlay) + results[name] = grade(scenario, workspace, wright) + good = results["reference"]["passed"] and not results["seed"]["passed"] and not results["reference"]["unsafeEdits"] + if not good: + ok = False + failed = [c for c in results["reference"]["checks"] if not c["passed"]] + print(f"INVALID {scenario_id}: seed passed={results['seed']['passed']}, reference failures={failed}, unsafe={results['reference']['unsafeEdits']}") + else: + print(f"ok {scenario_id}") + return ok + + +def baseline_path(path: str) -> str: + """PATH without any directory that provides a `wright` executable.""" + kept = [d for d in path.split(os.pathsep) if d and not shutil.which("wright", path=d)] + return os.pathsep.join(kept) + + +def shim_main(argv: list[str]) -> int: + start = time.monotonic() + code = subprocess.call([os.environ["WRIGHT_BENCH_REAL"], *argv]) + with open(os.environ["WRIGHT_BENCH_TRACE"], "a") as trace: + trace.write(json.dumps({"argv": argv, "exit": code, "seconds": round(time.monotonic() - start, 3)}) + "\n") + return code + + +def summarize_trace(trace: Path) -> dict: + calls = [json.loads(line) for line in trace.read_text().splitlines()] if trace.is_file() else [] + commands = [next((a for a in c["argv"] if not a.startswith("-")), "") for c in calls] + by_command: dict[str, int] = {} + for command in commands: + by_command[command] = by_command.get(command, 0) + 1 + return { + "invocations": len(calls), + "byCommand": by_command, + "failedInvocations": sum(1 for c in calls if c["exit"] != 0), + "ownerOrEnvironmentGaps": [c for c in calls if c["exit"] >= 3], + } + + +def run_trial(scenario: dict, condition: str, args: argparse.Namespace, out: Path) -> dict: + shutil.rmtree(out, ignore_errors=True) + workspace = out / "workspace" + materialize(scenario, workspace) + prompt = (scenario["dir"] / "prompt.md").read_text() + trace = out / "wright-trace.jsonl" + env = {k: v for k, v in os.environ.items() if k not in SHIM_ENV} + if condition == "wright": + shim_dir = out / "bin" + shim_dir.mkdir(parents=True) + shim = shim_dir / "wright" + shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim "$@"\n') + shim.chmod(0o755) + env.update(WRIGHT_BENCH_REAL=args.wright, WRIGHT_BENCH_TRACE=str(trace), PATH=f"{shim_dir}{os.pathsep}{baseline_path(env['PATH'])}") + else: + env["PATH"] = baseline_path(env["PATH"]) + start = time.monotonic() + proc = subprocess.Popen( + args.agent_cmd, shell=True, cwd=workspace, env=env, text=True, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + start_new_session=True, + ) + try: + stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) + agent_exit = proc.returncode + except subprocess.TimeoutExpired: + agent_exit = None + try: + os.killpg(proc.pid, signal.SIGKILL) + except ProcessLookupError: + pass + stdout, stderr = proc.communicate() + stderr = f"{stderr}\ntimeout" if stderr else "timeout" + seconds = round(time.monotonic() - start, 1) + (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") + wright_version = subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip() + result = { + "contract": RESULT_CONTRACT, + "scenario": scenario["id"], + "family": scenario["family"], + "language": scenario["language"], + "condition": condition, + "agent": {"id": args.agent_id, "command": args.agent_cmd, "exit": agent_exit, "seconds": seconds}, + "environment": {"os": platform.platform(), "python": platform.python_version(), "wright": wright_version, "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds")}, + "wrightUse": summarize_trace(trace), + **grade(scenario, workspace, args.wright), + } + (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + return result + + +def cmd_run(args: argparse.Namespace) -> int: + scenario = load_scenario(args.scenario) + ok = True + for condition in args.conditions: + for trial in range(1, args.trials + 1): + result = run_trial(scenario, condition, args, args.out / args.scenario / f"{condition}-{trial}") + ok &= result["passed"] + print(f"{args.scenario} {condition} trial {trial}: {'PASS' if result['passed'] else 'FAIL'} layers={result['failedLayers']} wright-invocations={result['wrightUse']['invocations']}") + return 0 if ok else 1 + + +def main() -> int: + if len(sys.argv) > 1 and sys.argv[1] == "shim": + return shim_main(sys.argv[2:]) + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + for name in ("validate", "run"): + p = sub.add_parser(name) + p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") + p.add_argument("--out", type=Path, default=ROOT / "target/agent-bench") + sub.choices["run"].add_argument("scenario", choices=all_scenario_ids()) + sub.choices["run"].add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace") + sub.choices["run"].add_argument("--agent-id", required=True, help="recorded agent/model/version label") + sub.choices["run"].add_argument("--conditions", nargs="+", choices=("baseline", "wright"), default=["baseline", "wright"]) + sub.choices["run"].add_argument("--trials", type=int, default=1) + sub.choices["run"].add_argument("--timeout", type=int, default=1800) + args = parser.parse_args() + args.wright = str(Path(args.wright).resolve()) + if args.command == "validate": + return 0 if validate(args.wright, args.out) else 1 + return cmd_run(args) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/scenarios/greenfield-elimination-race/prompt.md b/benchmarks/agent/scenarios/greenfield-elimination-race/prompt.md new file mode 100644 index 0000000..7eab510 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-elimination-race/prompt.md @@ -0,0 +1,10 @@ +Create `mode.ws`, an Overwatch Workshop free-for-all game mode, in this directory. + +Requirements: + +- Every player is forced to play Soldier: 76. +- An elimination of another player scores one point for the attacker. Self-inflicted deaths score nothing. +- A HUD visible to everyone shows the winning score. +- The first player to reach 7 points wins the match. + +Make sure the finished project is valid and has no remaining diagnostics. diff --git a/benchmarks/agent/scenarios/greenfield-elimination-race/reference/mode.ws b/benchmarks/agent/scenarios/greenfield-elimination-race/reference/mode.ws new file mode 100644 index 0000000..062b34a --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-elimination-race/reference/mode.ws @@ -0,0 +1,55 @@ +variables { + global: + 0: target_score + player: + 0: score +} + +rule ("configure match") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(target_score, 7); + Create HUD Text(All Players(All Teams), Custom String("First to {0}", Global.target_score), Null, Null, Top, 0, Color(White), Color(White), Color(White), Visible To and String, Default Visibility); + } +} + +rule ("force soldier") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Soldier: 76)); + } +} + +rule ("score on elimination") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Compare(Event Player.score, >=, Global.target_score); + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/greenfield-elimination-race/scenario.json b/benchmarks/agent/scenarios/greenfield-elimination-race/scenario.json new file mode 100644 index 0000000..e05ddc4 --- /dev/null +++ b/benchmarks/agent/scenarios/greenfield-elimination-race/scenario.json @@ -0,0 +1,18 @@ +{ + "id": "greenfield-elimination-race", + "family": "greenfield", + "language": "workshop", + "entry": "mode.ws", + "writable": ["mode.ws"], + "runtimeOnly": ["Elimination scoring, the HUD, and the winner declaration behave as intended in a live match."], + "checks": [ + {"id": "valid-project", "kind": "check", "layer": "workshop-rs"}, + {"id": "tracks-score-per-player", "kind": "symbols", "symbolKind": "playerVariable", "layer": "wright"}, + {"id": "forces-soldier", "kind": "contains", "file": "mode.ws", "text": "Hero(Soldier: 76)"}, + {"id": "scores-on-elimination", "kind": "contains", "file": "mode.ws", "text": ["Player Earned Elimination", "Player Dealt Final Blow"]}, + {"id": "excludes-self-kills", "kind": "contains", "file": "mode.ws", "text": ["Attacker != Victim", "Compare(Attacker, !=, Victim)"]}, + {"id": "has-hud", "kind": "contains", "file": "mode.ws", "text": "Create HUD Text"}, + {"id": "target-is-7", "kind": "contains", "file": "mode.ws", "text": ", 7"}, + {"id": "declares-winner", "kind": "contains", "file": "mode.ws", "text": "Declare Player Victory"} + ] +} diff --git a/benchmarks/agent/scenarios/greenfield-elimination-race/seed/.gitkeep b/benchmarks/agent/scenarios/greenfield-elimination-race/seed/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/benchmarks/agent/scenarios/modify-target-score/prompt.md b/benchmarks/agent/scenarios/modify-target-score/prompt.md new file mode 100644 index 0000000..3bf2fb7 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-target-score/prompt.md @@ -0,0 +1,6 @@ +`mode.ws` is an Overwatch Workshop game mode. Make these changes and nothing else: + +- The winning score becomes 7 instead of 5. +- Add a HUD visible to everyone that shows the winning score, e.g. "First to 7". + +Leave the existing rules' behavior otherwise untouched, and make sure the project stays valid. diff --git a/benchmarks/agent/scenarios/modify-target-score/reference/mode.ws b/benchmarks/agent/scenarios/modify-target-score/reference/mode.ws new file mode 100644 index 0000000..5584dad --- /dev/null +++ b/benchmarks/agent/scenarios/modify-target-score/reference/mode.ws @@ -0,0 +1,87 @@ +variables { + global: + 0: target_score + 1: match_started + player: + 0: score + 1: streak + 2: bonus +} + +rule ("configure match") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(target_score, 7); + Create HUD Text(All Players(All Teams), Custom String("First to {0}", Global.target_score), Null, Null, Top, 0, Color(White), Color(White), Color(White), Visible To and String, Default Visibility); + Set Global Variable(match_started, True); + } +} + +rule ("force soldier") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Soldier: 76)); + } +} + +rule ("score on elimination") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Global.match_started == True; + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + Modify Player Variable(Attacker, streak, Add, 1); + } +} + +rule ("reset streak on death") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Victim, streak, 0); + } +} + +rule ("streak bonus points") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Event Player.streak >= 3; + } + actions { + Modify Player Variable(Event Player, score, Add, 1); + Set Player Variable(Event Player, streak, 0); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Compare(Event Player.score, >=, Global.target_score); + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/modify-target-score/scenario.json b/benchmarks/agent/scenarios/modify-target-score/scenario.json new file mode 100644 index 0000000..3517f77 --- /dev/null +++ b/benchmarks/agent/scenarios/modify-target-score/scenario.json @@ -0,0 +1,18 @@ +{ + "id": "modify-target-score", + "family": "modification", + "language": "workshop", + "entry": "mode.ws", + "writable": ["mode.ws"], + "runtimeOnly": ["The HUD text renders as intended in a live match."], + "checks": [ + {"id": "valid-project", "kind": "check", "layer": "workshop-rs"}, + {"id": "target-is-7", "kind": "contains", "file": "mode.ws", "text": "Set Global Variable(target_score, 7);"}, + {"id": "old-target-gone", "kind": "absent", "file": "mode.ws", "text": "Set Global Variable(target_score, 5);"}, + {"id": "hud-added", "kind": "contains", "file": "mode.ws", "text": "Create HUD Text"}, + {"id": "streak-rule-preserved", "kind": "contains", "file": "mode.ws", "text": "Set Player Variable(Event Player, streak, 0);"}, + {"id": "elimination-rule-preserved", "kind": "contains", "file": "mode.ws", "text": "Modify Player Variable(Attacker, streak, Add, 1);"}, + {"id": "winner-rule-preserved", "kind": "contains", "file": "mode.ws", "text": "Declare Player Victory(Event Player);"}, + {"id": "rule-count-preserved", "kind": "contains", "file": "mode.ws", "text": "rule (", "min": 6, "max": 6} + ] +} diff --git a/benchmarks/agent/scenarios/modify-target-score/seed/mode.ws b/benchmarks/agent/scenarios/modify-target-score/seed/mode.ws new file mode 100644 index 0000000..385b48c --- /dev/null +++ b/benchmarks/agent/scenarios/modify-target-score/seed/mode.ws @@ -0,0 +1,86 @@ +variables { + global: + 0: target_score + 1: match_started + player: + 0: score + 1: streak + 2: bonus +} + +rule ("configure match") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(target_score, 5); + Set Global Variable(match_started, True); + } +} + +rule ("force soldier") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Soldier: 76)); + } +} + +rule ("score on elimination") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Global.match_started == True; + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + Modify Player Variable(Attacker, streak, Add, 1); + } +} + +rule ("reset streak on death") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Victim, streak, 0); + } +} + +rule ("streak bonus points") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Event Player.streak >= 3; + } + actions { + Modify Player Variable(Event Player, score, Add, 1); + Set Player Variable(Event Player, streak, 0); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Compare(Event Player.score, >=, Global.target_score); + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/prompt.md b/benchmarks/agent/scenarios/repair-runaway-loop/prompt.md new file mode 100644 index 0000000..a82fc5f --- /dev/null +++ b/benchmarks/agent/scenarios/repair-runaway-loop/prompt.md @@ -0,0 +1,3 @@ +Players report that as soon as a match starts using `mode.ws` (an Overwatch Workshop game mode), the server freezes and the countdown never finishes. + +Find the root cause, fix it with the smallest change that keeps the intended behavior (a counter that counts up once per second while the match runs), and record the name of the offending rule as `{"rule": ""}` in `answer.json`. Make sure the project has no remaining problems. diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/reference/answer.json b/benchmarks/agent/scenarios/repair-runaway-loop/reference/answer.json new file mode 100644 index 0000000..791c0cd --- /dev/null +++ b/benchmarks/agent/scenarios/repair-runaway-loop/reference/answer.json @@ -0,0 +1 @@ +{"rule": "tick counter"} diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/reference/mode.ws b/benchmarks/agent/scenarios/repair-runaway-loop/reference/mode.ws new file mode 100644 index 0000000..e56f9a5 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-runaway-loop/reference/mode.ws @@ -0,0 +1,40 @@ +variables { + global: + 0: countdown + 1: ticks +} + +rule ("start countdown") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(countdown, 10); + } +} + +rule ("tick counter") { + event { + Ongoing - Global; + } + actions { + While(True); + Modify Global Variable(ticks, Add, 1); + Wait(1, Ignore Condition); + End; + } +} + +rule ("run countdown") { + event { + Ongoing - Global; + } + conditions { + Global.countdown > 0; + } + actions { + Wait(1, Ignore Condition); + Modify Global Variable(countdown, Subtract, 1); + Loop If Condition Is True; + } +} diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json b/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json new file mode 100644 index 0000000..4d50193 --- /dev/null +++ b/benchmarks/agent/scenarios/repair-runaway-loop/scenario.json @@ -0,0 +1,15 @@ +{ + "id": "repair-runaway-loop", + "family": "diagnosis", + "language": "workshop", + "entry": "mode.ws", + "writable": ["mode.ws", "answer.json"], + "runtimeOnly": ["The server no longer freezes in a live match."], + "checks": [ + {"id": "valid-project", "kind": "check", "layer": "workshop-rs"}, + {"id": "no-unbounded-loop", "kind": "lint", "code": "while-without-wait", "layer": "wright"}, + {"id": "counter-still-counts", "kind": "contains", "file": "mode.ws", "text": "Modify Global Variable(ticks, Add, 1);"}, + {"id": "countdown-rule-preserved", "kind": "contains", "file": "mode.ws", "text": "Modify Global Variable(countdown, Subtract, 1);"}, + {"id": "root-cause-named", "kind": "answer", "key": "rule", "expected": "tick counter"} + ] +} diff --git a/benchmarks/agent/scenarios/repair-runaway-loop/seed/mode.ws b/benchmarks/agent/scenarios/repair-runaway-loop/seed/mode.ws new file mode 100644 index 0000000..2a415ff --- /dev/null +++ b/benchmarks/agent/scenarios/repair-runaway-loop/seed/mode.ws @@ -0,0 +1,39 @@ +variables { + global: + 0: countdown + 1: ticks +} + +rule ("start countdown") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(countdown, 10); + } +} + +rule ("tick counter") { + event { + Ongoing - Global; + } + actions { + While(True); + Modify Global Variable(ticks, Add, 1); + End; + } +} + +rule ("run countdown") { + event { + Ongoing - Global; + } + conditions { + Global.countdown > 0; + } + actions { + Wait(1, Ignore Condition); + Modify Global Variable(countdown, Subtract, 1); + Loop If Condition Is True; + } +} diff --git a/benchmarks/agent/scenarios/understand-score-flow/prompt.md b/benchmarks/agent/scenarios/understand-score-flow/prompt.md new file mode 100644 index 0000000..b4866ca --- /dev/null +++ b/benchmarks/agent/scenarios/understand-score-flow/prompt.md @@ -0,0 +1,8 @@ +`mode.ws` is an Overwatch Workshop game mode. Do not modify it. + +Answer these questions about it by writing `answer.json` in this directory, with exactly these keys: + +- `targetScore`: the number of points needed to win, as a number. +- `victoryRule`: the name of the rule that declares the winner. +- `scoreWriters`: the names of all rules that change the player variable `score`, sorted alphabetically. +- `unusedVariables`: the names of all declared variables that no rule ever reads or writes, sorted alphabetically. diff --git a/benchmarks/agent/scenarios/understand-score-flow/reference/answer.json b/benchmarks/agent/scenarios/understand-score-flow/reference/answer.json new file mode 100644 index 0000000..49a5586 --- /dev/null +++ b/benchmarks/agent/scenarios/understand-score-flow/reference/answer.json @@ -0,0 +1 @@ +{"targetScore": 5, "victoryRule": "declare winner", "scoreWriters": ["score on elimination", "streak bonus points"], "unusedVariables": ["bonus"]} diff --git a/benchmarks/agent/scenarios/understand-score-flow/scenario.json b/benchmarks/agent/scenarios/understand-score-flow/scenario.json new file mode 100644 index 0000000..82222dc --- /dev/null +++ b/benchmarks/agent/scenarios/understand-score-flow/scenario.json @@ -0,0 +1,14 @@ +{ + "id": "understand-score-flow", + "family": "understanding", + "language": "workshop", + "entry": "mode.ws", + "writable": ["answer.json"], + "checks": [ + {"id": "source-still-valid", "kind": "check", "layer": "workshop-rs"}, + {"id": "target-score", "kind": "answer", "key": "targetScore", "expected": 5}, + {"id": "victory-rule", "kind": "answer", "key": "victoryRule", "expected": "declare winner"}, + {"id": "score-writers", "kind": "answer", "key": "scoreWriters", "expected": ["score on elimination", "streak bonus points"]}, + {"id": "unused-variables", "kind": "answer", "key": "unusedVariables", "expected": ["bonus"]} + ] +} diff --git a/benchmarks/agent/scenarios/understand-score-flow/seed/mode.ws b/benchmarks/agent/scenarios/understand-score-flow/seed/mode.ws new file mode 100644 index 0000000..385b48c --- /dev/null +++ b/benchmarks/agent/scenarios/understand-score-flow/seed/mode.ws @@ -0,0 +1,86 @@ +variables { + global: + 0: target_score + 1: match_started + player: + 0: score + 1: streak + 2: bonus +} + +rule ("configure match") { + event { + Ongoing - Global; + } + actions { + Set Global Variable(target_score, 5); + Set Global Variable(match_started, True); + } +} + +rule ("force soldier") { + event { + Ongoing - Each Player; + All; + All; + } + actions { + Start Forcing Player To Be Hero(Event Player, Hero(Soldier: 76)); + } +} + +rule ("score on elimination") { + event { + Player Earned Elimination; + All; + All; + } + conditions { + Global.match_started == True; + Attacker != Victim; + } + actions { + Modify Player Variable(Attacker, score, Add, 1); + Modify Player Variable(Attacker, streak, Add, 1); + } +} + +rule ("reset streak on death") { + event { + Player Died; + All; + All; + } + actions { + Set Player Variable(Victim, streak, 0); + } +} + +rule ("streak bonus points") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Event Player.streak >= 3; + } + actions { + Modify Player Variable(Event Player, score, Add, 1); + Set Player Variable(Event Player, streak, 0); + } +} + +rule ("declare winner") { + event { + Ongoing - Each Player; + All; + All; + } + conditions { + Compare(Event Player.score, >=, Global.target_score); + } + actions { + Declare Player Victory(Event Player); + } +} diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py new file mode 100644 index 0000000..89bd91e --- /dev/null +++ b/benchmarks/agent/test_agent_bench.py @@ -0,0 +1,44 @@ +import json +import os +import shutil +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +import agent_bench + +WRIGHT = os.environ.get("WRIGHT_BIN", str(agent_bench.ROOT / "target/debug/wright")) + + +@unittest.skipUnless(Path(WRIGHT).is_file(), "build wright first or set WRIGHT_BIN") +class AgentBenchTest(unittest.TestCase): + def setUp(self): + (agent_bench.ROOT / "target").mkdir(exist_ok=True) + self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")) + self.addCleanup(shutil.rmtree, self.out, True) + + def test_scenarios_are_solvable_and_not_vacuous(self): + self.assertTrue(agent_bench.validate(WRIGHT, self.out)) + + def test_conditions_differ_only_in_wright_availability(self): + scenario = "repair-runaway-loop" + reference = agent_bench.SCENARIOS / scenario / "reference" + agent = f"cp {reference}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" + subprocess.run( + [sys.executable, agent_bench.__file__, "run", scenario, "--wright", WRIGHT, "--out", str(self.out), "--agent-id", "fake", "--agent-cmd", agent], + check=False, + capture_output=True, + ) + baseline = json.loads((self.out / scenario / "baseline-1/result.json").read_text()) + assisted = json.loads((self.out / scenario / "wright-1/result.json").read_text()) + self.assertEqual(baseline["wrightUse"]["invocations"], 0) + self.assertIn("missing-wright.txt", baseline["unsafeEdits"]) + self.assertEqual(assisted["wrightUse"]["byCommand"], {"check": 1}) + self.assertTrue(assisted["passed"]) + self.assertEqual(assisted["unsafeEdits"], []) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/README.md b/docs/README.md index 636dd5a..9049cf7 100644 --- a/docs/README.md +++ b/docs/README.md @@ -39,6 +39,8 @@ Issue contract. - [Embedding/tool API](embedding.md): programmatic session/query/edit services. - [Agent contract](agent-contract.md): versioned session requests, results, and transport mappings for coding agents and embedding consumers. +- [Agent benchmark](agent-benchmark.md): product-level benchmark contract for + general coding agents working with Wright. - [Language services & LSP](language-services.md): editor-neutral language services and LSP framing. - [Integration verification](compatibility.md): owner boundaries and diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md new file mode 100644 index 0000000..62909f1 --- /dev/null +++ b/docs/agent-benchmark.md @@ -0,0 +1,85 @@ +# Agent Benchmark + +- Contract: `wright-agent-bench/v1` +- Harness: [`benchmarks/agent/agent_bench.py`](../benchmarks/agent/agent_bench.py) + +The benchmark answers one product question: can a general coding agent, with no +Workshop-specific prompt or skill injection, use a project and Wright to +complete realistic Workshop work correctly? It measures Wright's discoverable +semantic surface; it is not a model leaderboard, and one stochastic run is not +evidence of correctness (use `--trials`). + +## Allowed agent context + +- The scenario workspace: the seed project only. +- The scenario prompt, delivered on the agent's stdin. It states the + requirement in user terms and never names Wright commands or Workshop APIs. +- In the `wright` condition, the released `wright` CLI and `wright serve` + session on `PATH`. + +Not allowed: a Workshop/OverPy/OSTW system prompt or skill pack, a generated +API reference, or task-specific hints. The agent, model, and version are +recorded (`--agent-id`); the contract does not depend on a vendor. + +## Conditions + +Both conditions use the same workspace and prompt. They differ only in `PATH`: +`baseline` removes every directory that provides a `wright` executable; +`wright` prepends a logging shim for the binary under test. A determined agent +can still find a Wright binary elsewhere on disk, so run baselines in a clean +environment when that matters. + +## Scenarios + +`benchmarks/agent/scenarios//` contains `scenario.json`, `prompt.md`, +`seed/` (the initial workspace), and `reference/` (files overlaid on the seed +to form a passing solution). Families: `greenfield`, `understanding`, +`modification`, `diagnosis`. `scenario.json` fields: + +| Field | Meaning | +| --- | --- | +| `id`, `family`, `language` | Identity; `language` is `workshop`, `opy`, or `ostw`, and a scenario may use a language only once its owner declares the needed capability supported | +| `entry` | Source file that Wright checks | +| `writable` | Files the agent may change; any other change is reported as an unsafe edit | +| `runtimeOnly` | Claims that only the Overwatch runtime can verify; reported as unverified, never as passed | +| `checks` | Deterministic checks, each with `id`, `kind`, and `layer` | + +Check kinds: `check` (`wright check` reports no errors), `lint` (at most `max` +findings with lint `code`), `symbols` (at least `min` symbols of `symbolKind` +via `wright serve`), `contains` / `absent` (source text, `text` may be a list +of alternatives, `min`/`max` occurrences), and `answer` (`answer.json` key equals +`expected`). `layer` names what a failure implicates: `agent` for a requirement +the produced work does not meet, or `workshop-rs` / `opy-rs` / `deltin-rs` / +`wright` for validity or analysis results owned by that layer. + +## Scenario validity + +`agent_bench.py validate` requires every scenario's reference to pass all checks +and its untouched seed to fail at least one. A reference that fails a check is a +product or engine gap named by that check's `layer`, not an agent failure. This +runs in the CI benchmark job; running agents does not. + +## Result + +`agent_bench.py run --agent-cmd CMD --agent-id LABEL` runs each +condition and writes `//-/result.json`, with +the workspace, `agent.log`, and Wright trace beside it. + +| Field | Meaning | +| --- | --- | +| `checks`, `passed`, `failedLayers` | Per-check outcome and the layers implicated by failures | +| `diagnostics` | Remaining `wright check` diagnostics with their owner origin | +| `unsafeEdits` | Files changed outside `writable` | +| `wrightUse` | Wright invocations by subcommand, failed invocations (correction rounds), and exits of 3 or 4 (unsupported or internal failures: candidate owner or environment gaps) | +| `unverifiedRuntimeClaims` | The scenario's `runtimeOnly` claims | +| `agent`, `environment` | Agent label, command, exit, duration; OS, Python, Wright version, timestamp | + +`wrightUse` is recorded per CLI invocation; requests inside one `wright serve` +session are not itemized. Comparing `baseline` and `wright` results for the same +scenario shows what Wright adds. Exit 3 or 4 entries and failed `layer` values +are the input for owner Issues. + +## Cadence + +The benchmark does not gate pull requests. Run it manually or on a schedule once +its cost and stability are understood.