| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615 |
- #!/usr/bin/env python3
- # /// script
- # requires-python = ">=3.9"
- # ///
- """Run eval cases through the configured platform adapter.
- A case is `input + rubric + optional state_prefix + optional files`. This
- runner does the runtime-specific part of an eval: it stages the skill under
- test and the case's fixture files into a clean working directory, builds the
- prompt the adapter understands, runs it, and records the transcript plus
- timing and token usage. Grading happens elsewhere; the grader subagent reads
- the transcript and artifacts this runner leaves behind.
- What this runner deliberately does NOT do:
- - No Docker, no PTY, no keychain staging, no dual-isolation strategy.
- - No hardcoded model. Everything runtime-specific comes from the adapter.
- Modes (--mode) decide which configs each case runs under:
- quality : one config, "skill" — the skill staged in the cwd.
- baseline : two configs per case — "skill" (skill staged) and "bare"
- (nothing staged), same input, so the bare-model floor is
- measured under identical conditions.
- variant : two configs — "skill" (--skill-path) and "variant"
- (--variant-path, the stripped or prior-version skill).
- Run layout: <run-dir>/<config>/<case-id>/ (plus /run-N/ when --runs > 1),
- so `aggregate_benchmark.py --baseline <run-dir>/bare --variant
- <run-dir>/skill` compares configs directly from the timing.json files.
- Skill staging: the skill directory is copied (symlink where possible) into
- <case-cwd>/<skill_dir>/<skill-name>/ before the adapter is invoked, where
- skill_dir comes from the adapter (default ".claude/skills"). Without this
- every config would measure the bare model.
- Fixtures: each path in a case's `files` list is staged into the case cwd at
- its own relative path. Sources resolve against --project-root, then the cases
- file's directory, then as absolute paths.
- Isolation: the subprocess env is built from scratch, never inherited. It
- holds PATH, a fresh empty HOME at <case>/.home, CLAUDE_CONFIG_DIR inside
- that HOME, the adapter's auth_env var ONLY if set non-empty in the host env
- (setting it to "" would break the runtime's own credential fallback), and any
- adapter `env_passthrough` keys present in the host env. Nothing else crosses.
- The adapter config file (JSON) — schema and discovery rules in
- references/platform-adapter.md, working example in
- assets/adapter-claude-code.json:
- invocation : argv template. "{prompt}" -> composed case prompt,
- "{cwd}" -> clean working directory.
- auth_env : env var name carrying auth (e.g. "ANTHROPIC_API_KEY").
- transcript : {"format": "stdout-jsonl"} or
- {"format": "file", "path": "transcript.jsonl"}.
- skill_dir : where the runtime discovers skills under the cwd.
- env_passthrough : optional list of extra host env vars to forward.
- If no adapter config is found, the runner degrades gracefully: it stages every
- case (clean cwd, skill, fixtures, prompt with state_prefix applied) and writes
- a manifest, but records each result as "skipped: no runtime adapter
- configured" instead of crashing. A human or a configured runtime can then
- complete the run.
- state_prefix handling: when a case carries a state_prefix, it is PREPENDED to
- the input to place the skill mid-workflow in one shot. The composed prompt is
- recorded so the grader sees exactly what ran.
- Usage:
- python3 run_evals.py \\
- --cases CASES.json \\
- --skill-path SKILL_DIR \\
- --output-dir DIR \\
- [--mode quality|baseline|variant] \\
- [--variant-path SKILL_DIR] \\
- [--project-root DIR] \\
- [--adapter ADAPTER.json] \\
- [--case-ids A1,B3] [--runs N] [--timeout SECS] [--workers N] [--quiet]
- CASES.json is either a list of cases or {"cases": [...]}. Each case:
- {"id": "...", "input": "...", "rubric": [...],
- "state_prefix": "..."?, "files": ["..."]?}
- """
- from __future__ import annotations
- import argparse
- import json
- import os
- import shutil
- import subprocess
- import sys
- import time
- from collections.abc import Mapping
- from concurrent.futures import ThreadPoolExecutor, as_completed
- from datetime import datetime, timezone
- from pathlib import Path
- # --- small self-contained helpers (no Docker/keychain imports) -------------
- def utc_now_iso() -> str:
- return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
- def new_run_id(label: str) -> str:
- return f"{datetime.now().strftime('%Y%m%d-%H%M%S')}-{label}"
- def write_json(path: Path, data: object) -> None:
- path.parent.mkdir(parents=True, exist_ok=True)
- path.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
- def read_json(path: Path) -> object:
- return json.loads(path.read_text(encoding="utf-8"))
- # --- adapter ----------------------------------------------------------------
- def find_adapter(explicit: Path | None, cases_file: Path) -> Path | None:
- """Locate the adapter config. Returns None when none is configured."""
- if explicit is not None:
- return explicit if explicit.is_file() else None
- env_path = os.environ.get("BMAD_EVAL_ADAPTER")
- if env_path and Path(env_path).is_file():
- return Path(env_path)
- for candidate in (
- cases_file.parent / "adapter.json",
- cases_file.parent / ".bmad-eval-adapter.json",
- ):
- if candidate.is_file():
- return candidate
- return None
- def load_adapter(path: Path) -> dict:
- cfg = read_json(path)
- if not isinstance(cfg, dict):
- raise ValueError(f"adapter config must be a JSON object: {path}")
- if "invocation" not in cfg or not isinstance(cfg["invocation"], list):
- raise ValueError("adapter config missing 'invocation' argv list")
- return cfg
- def build_argv(invocation: list, prompt: str, cwd: str) -> list[str]:
- argv: list[str] = []
- for tok in invocation:
- tok = str(tok)
- tok = (tok.replace("{prompt}", prompt)
- .replace("{query}", prompt)
- .replace("{cwd}", cwd))
- argv.append(tok)
- return argv
- def build_case_env(adapter: Mapping | None, home_dir: Path,
- host_env: Mapping[str, str]) -> dict[str, str]:
- """Build the subprocess environment from scratch — never from os.environ.
- Inheriting the host env would leak shell config, tokens, and runtime
- state into the clean room. The env holds exactly: PATH, a fresh HOME,
- CLAUDE_CONFIG_DIR inside it, the adapter's auth var ONLY when set
- non-empty in the host (an empty-string auth var breaks the runtime's own
- credential fallback), and any adapter env_passthrough keys present in
- the host env.
- """
- adapter = adapter or {}
- env = {
- "PATH": host_env.get("PATH", ""),
- "HOME": str(home_dir),
- "CLAUDE_CONFIG_DIR": str(home_dir / ".claude"),
- }
- auth_env = adapter.get("auth_env")
- if auth_env:
- val = host_env.get(str(auth_env))
- if val:
- env[str(auth_env)] = val
- for key in adapter.get("env_passthrough") or []:
- val = host_env.get(str(key))
- if val is not None:
- env[str(key)] = val
- return env
- # --- staging: skill under test + fixtures ------------------------------------
- def stage_skill(skill_path: Path, cwd: Path, skills_subdir: str) -> Path:
- """Place the skill where the runtime discovers skills inside the cwd.
- Symlink when possible (cheap, and the skill is read-only to the run);
- copy as the fallback.
- """
- dest_root = cwd / skills_subdir
- dest_root.mkdir(parents=True, exist_ok=True)
- dest = dest_root / skill_path.name
- if not dest.exists():
- try:
- os.symlink(skill_path, dest)
- except OSError:
- shutil.copytree(skill_path, dest, dirs_exist_ok=True)
- return dest
- def resolve_fixtures(files: list, project_root: Path,
- cases_dir: Path) -> list[tuple[Path, str]]:
- """Map each `files` entry to (source, dest-relative-path).
- The entry's own relative path is preserved inside the cwd, so a bare
- filename lands at the workspace root and a nested path keeps its
- directory structure — matching the path the case input references.
- """
- out: list[tuple[Path, str]] = []
- for entry in files or []:
- entry = str(entry)
- for candidate in (
- (project_root / entry).resolve(),
- (cases_dir / entry).resolve(),
- Path(entry).resolve(),
- ):
- if candidate.is_file():
- out.append((candidate, entry))
- break
- else:
- print(f"Warning: fixture not found: {entry}", file=sys.stderr)
- return out
- def stage_fixtures(fixtures: list[tuple[Path, str]], cwd: Path) -> None:
- for src, dest_rel in fixtures:
- dest = cwd / dest_rel
- dest.parent.mkdir(parents=True, exist_ok=True)
- shutil.copy2(src, dest)
- # --- case composition -------------------------------------------------------
- def compose_prompt(case: dict) -> str:
- """Apply state_prefix by prepending it to the input.
- The state_prefix is a bracketed prime that places the skill mid-workflow in
- one shot. Prepending keeps the input intact and visible to the grader.
- """
- input_text = str(case.get("input", ""))
- prefix = case.get("state_prefix")
- if prefix:
- return f"{str(prefix).rstrip()}\n\n{input_text}"
- return input_text
- # --- transcript + token accounting -----------------------------------------
- def read_transcript(transcript_cfg: dict, captured_stdout: bytes,
- cwd: Path) -> tuple[str, str]:
- """Return (transcript_text, source). Source names where it came from."""
- fmt = (transcript_cfg or {}).get("format", "stdout-jsonl")
- if fmt == "file":
- rel = (transcript_cfg or {}).get("path", "transcript.jsonl")
- f = cwd / rel
- if f.is_file():
- return f.read_text(encoding="utf-8", errors="replace"), f"file:{rel}"
- return "", f"file:{rel} (missing)"
- return captured_stdout.decode("utf-8", errors="replace"), "stdout"
- def account_transcript(transcript_text: str) -> dict:
- """Pull timing/token usage from a JSONL transcript when present.
- Reads usage out of the completion notification immediately, so tokens are
- captured at run time rather than recomputed later. Recognizes the common
- `result` event with a usage block and per-message usage blocks; unknown
- shapes degrade to zero counts without failing.
- """
- input_tokens = 0
- output_tokens = 0
- total_steps = 0
- tool_calls: dict[str, int] = {}
- found_usage = False
- for raw in transcript_text.splitlines():
- raw = raw.strip()
- if not raw:
- continue
- try:
- evt = json.loads(raw)
- except json.JSONDecodeError:
- continue
- if not isinstance(evt, dict):
- continue
- etype = evt.get("type")
- if etype == "assistant":
- total_steps += 1
- msg = evt.get("message", {})
- usage = msg.get("usage") if isinstance(msg, dict) else None
- if isinstance(usage, dict):
- found_usage = True
- input_tokens += int(usage.get("input_tokens", 0) or 0)
- output_tokens += int(usage.get("output_tokens", 0) or 0)
- for item in (msg.get("content", []) if isinstance(msg, dict) else []):
- if isinstance(item, dict) and item.get("type") == "tool_use":
- name = item.get("name", "?")
- tool_calls[name] = tool_calls.get(name, 0) + 1
- elif etype == "result":
- usage = evt.get("usage")
- if isinstance(usage, dict):
- found_usage = True
- # result usage is authoritative; prefer it over the running sum
- input_tokens = int(usage.get("input_tokens", input_tokens) or input_tokens)
- output_tokens = int(usage.get("output_tokens", output_tokens) or output_tokens)
- return {
- "input_tokens": input_tokens,
- "output_tokens": output_tokens,
- "total_tokens": input_tokens + output_tokens,
- "tokens_reported": found_usage,
- "total_steps": total_steps,
- "tool_calls": tool_calls,
- "total_tool_calls": sum(tool_calls.values()),
- }
- # --- per-case execution -----------------------------------------------------
- def run_case(case: dict, case_dir: Path, run_dir: Path,
- adapter: dict | None, timeout: int, config: str,
- skill_path: Path | None,
- fixtures: list[tuple[Path, str]]) -> dict:
- case_id = str(case.get("id", "unnamed"))
- cwd = case_dir / "cwd"
- cwd.mkdir(parents=True, exist_ok=True)
- stage_fixtures(fixtures, cwd)
- if skill_path is not None:
- skills_subdir = (adapter or {}).get("skill_dir", ".claude/skills")
- stage_skill(skill_path, cwd, skills_subdir)
- prompt = compose_prompt(case)
- (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
- write_json(case_dir / "case.json", case)
- if adapter is None:
- result = {
- "case_id": case_id,
- "config": config,
- "status": "skipped",
- "reason": "no runtime adapter configured",
- "prompt_chars": len(prompt),
- "cwd": str(cwd.relative_to(run_dir)),
- }
- write_json(case_dir / "timing.json", {
- "case_id": case_id, "config": config, "status": "skipped",
- "captured_at": utc_now_iso(),
- })
- return result
- transcript_path = case_dir / "transcript.jsonl"
- argv = build_argv(adapter["invocation"], prompt, str(cwd))
- home_dir = case_dir / ".home"
- (home_dir / ".claude").mkdir(parents=True, exist_ok=True)
- env = build_case_env(adapter, home_dir, os.environ)
- start = time.time()
- captured = b""
- return_code = 0
- error_tail = ""
- status = "ok"
- try:
- proc = subprocess.run(
- argv,
- stdout=subprocess.PIPE,
- stderr=subprocess.PIPE,
- cwd=str(cwd),
- env=env,
- timeout=timeout,
- )
- captured = proc.stdout or b""
- return_code = proc.returncode
- error_tail = (proc.stderr or b"").decode("utf-8", errors="replace")[-2000:]
- if return_code != 0:
- status = "error"
- except FileNotFoundError as e:
- # Adapter invocation command is not on PATH: degrade, do not crash.
- elapsed = time.time() - start
- write_json(case_dir / "timing.json", {
- "case_id": case_id, "config": config, "status": "adapter-missing",
- "elapsed_s": round(elapsed, 3), "captured_at": utc_now_iso(),
- })
- return {
- "case_id": case_id,
- "config": config,
- "status": "adapter-missing",
- "reason": f"invocation command not found: {e}",
- "cwd": str(cwd.relative_to(run_dir)),
- }
- except subprocess.TimeoutExpired as e:
- captured = e.stdout or b""
- return_code = -1
- status = "timeout"
- error_tail = f"TIMEOUT after {timeout}s"
- elapsed = time.time() - start
- transcript_text, source = read_transcript(
- adapter.get("transcript", {}), captured, cwd
- )
- transcript_path.write_text(transcript_text, encoding="utf-8")
- accounting = account_transcript(transcript_text)
- # Capture timing/tokens immediately to timing.json (run-time snapshot).
- timing = {
- "case_id": case_id,
- "config": config,
- "status": status,
- "elapsed_s": round(elapsed, 3),
- "return_code": return_code,
- "transcript_source": source,
- "input_tokens": accounting["input_tokens"],
- "output_tokens": accounting["output_tokens"],
- "total_tokens": accounting["total_tokens"],
- "tokens_reported": accounting["tokens_reported"],
- "total_steps": accounting["total_steps"],
- "total_tool_calls": accounting["total_tool_calls"],
- "captured_at": utc_now_iso(),
- }
- write_json(case_dir / "timing.json", timing)
- return {
- "case_id": case_id,
- "config": config,
- "status": status,
- "elapsed_s": round(elapsed, 3),
- "return_code": return_code,
- "transcript": str(transcript_path.relative_to(run_dir)),
- "cwd": str(cwd.relative_to(run_dir)),
- "tokens": accounting["total_tokens"],
- "tool_calls": accounting["tool_calls"],
- "error_tail": error_tail,
- }
- # --- main -------------------------------------------------------------------
- def load_cases(cases_file: Path) -> list[dict]:
- data = read_json(cases_file)
- if isinstance(data, dict) and "cases" in data:
- cases = data["cases"]
- elif isinstance(data, list):
- cases = data
- else:
- raise ValueError("cases file must be a list or {'cases': [...]}")
- if not isinstance(cases, list):
- raise ValueError("'cases' must be a list")
- return cases
- def main(argv: list[str] | None = None) -> int:
- p = argparse.ArgumentParser(
- description=__doc__,
- formatter_class=argparse.RawDescriptionHelpFormatter,
- )
- p.add_argument("--cases", required=True, type=Path)
- p.add_argument("--skill-path", required=True, type=Path,
- help="directory of the skill under test (contains SKILL.md)")
- p.add_argument("--output-dir", required=True, type=Path)
- p.add_argument("--mode", choices=("quality", "baseline", "variant"),
- default="quality")
- p.add_argument("--variant-path", type=Path, default=None,
- help="variant mode: the stripped or prior-version skill")
- p.add_argument("--project-root", type=Path, default=None,
- help="base for resolving fixture paths; defaults to the "
- "cases file's directory")
- p.add_argument("--adapter", type=Path, default=None,
- help="adapter config JSON; defaults to BMAD_EVAL_ADAPTER env "
- "or adapter.json beside the cases file")
- p.add_argument("--case-ids", default=None,
- help="comma-separated subset of case ids to run")
- p.add_argument("--runs", type=int, default=1,
- help="repeats per case per config for the variance benchmark")
- p.add_argument("--timeout", type=int, default=600)
- p.add_argument("--workers", type=int, default=4)
- p.add_argument("--label", default="evals", help="label for the run id")
- p.add_argument("--quiet", action="store_true")
- args = p.parse_args(argv)
- cases_file = args.cases.resolve()
- if not cases_file.is_file():
- print(f"cases file not found: {cases_file}", file=sys.stderr)
- return 2
- skill_path = args.skill_path.resolve()
- if not (skill_path / "SKILL.md").is_file():
- print(f"skill path has no SKILL.md: {skill_path}", file=sys.stderr)
- return 2
- if args.mode == "variant":
- if args.variant_path is None:
- print("--mode variant requires --variant-path", file=sys.stderr)
- return 2
- variant_path = args.variant_path.resolve()
- if not (variant_path / "SKILL.md").is_file():
- print(f"variant path has no SKILL.md: {variant_path}",
- file=sys.stderr)
- return 2
- else:
- variant_path = None
- project_root = (args.project_root.resolve() if args.project_root
- else cases_file.parent)
- # Each config is (name, skill-to-stage-or-None). Baseline runs every case
- # twice — skill staged and bare — so the floor is measured under
- # identical conditions.
- if args.mode == "baseline":
- configs: list[tuple[str, Path | None]] = [
- ("skill", skill_path), ("bare", None)]
- elif args.mode == "variant":
- configs = [("skill", skill_path), ("variant", variant_path)]
- else:
- configs = [("skill", skill_path)]
- cases = load_cases(cases_file)
- if args.case_ids:
- wanted = {x.strip() for x in args.case_ids.split(",") if x.strip()}
- cases = [c for c in cases if str(c.get("id")) in wanted]
- adapter_path = find_adapter(args.adapter, cases_file)
- adapter: dict | None = None
- adapter_note = "none"
- if adapter_path is not None:
- try:
- adapter = load_adapter(adapter_path)
- adapter_note = str(adapter_path)
- except Exception as e:
- print(f"adapter config invalid ({e}); degrading to skip-only",
- file=sys.stderr)
- adapter = None
- adapter_note = f"invalid: {e}"
- run_id = new_run_id(args.label)
- run_dir = (args.output_dir / run_id).resolve()
- run_dir.mkdir(parents=True, exist_ok=True)
- write_json(run_dir / "run.json", {
- "run_id": run_id,
- "cases_file": str(cases_file),
- "skill_path": str(skill_path),
- "variant_path": str(variant_path) if variant_path else None,
- "mode": args.mode,
- "configs": [name for name, _ in configs],
- "runs_per_case": args.runs,
- "adapter": adapter_note,
- "started_at": utc_now_iso(),
- "case_count": len(cases),
- })
- if adapter is None and not args.quiet:
- print("[run_evals] no runtime adapter configured; staging cases only "
- "(no crash). Configure an adapter to execute.", file=sys.stderr)
- results: list[dict] = []
- if not args.quiet:
- print(f"[run_evals] {len(cases)} cases x {len(configs)} configs x "
- f"{args.runs} runs, mode={args.mode}, run_dir={run_dir}",
- file=sys.stderr)
- jobs: list[tuple[str, dict, Path, Path | None]] = []
- for config_name, config_skill in configs:
- for c in cases:
- base = run_dir / config_name / str(c.get("id", "unnamed"))
- for i in range(max(1, args.runs)):
- case_dir = base / f"run-{i + 1}" if args.runs > 1 else base
- jobs.append((config_name, c, case_dir, config_skill))
- with ThreadPoolExecutor(max_workers=max(1, args.workers)) as pool:
- fut_to_case = {
- pool.submit(run_case, c, case_dir, run_dir, adapter,
- int(c.get("timeout", args.timeout)), config_name,
- config_skill,
- resolve_fixtures(c.get("files", []), project_root,
- cases_file.parent)): c
- for config_name, c, case_dir, config_skill in jobs
- }
- for fut in as_completed(fut_to_case):
- c = fut_to_case[fut]
- try:
- res = fut.result()
- except Exception as e:
- res = {"case_id": str(c.get("id")), "status": "exception",
- "reason": str(e)}
- results.append(res)
- if not args.quiet:
- print(f" [{res.get('status')}] {res.get('config', '?')}/"
- f"{res.get('case_id')} ({res.get('elapsed_s', 0)}s)",
- file=sys.stderr)
- summary = {
- "run_id": run_id,
- "completed_at": utc_now_iso(),
- "mode": args.mode,
- "total": len(jobs),
- "executed": sum(1 for r in results if r.get("status") == "ok"),
- "skipped": sum(1 for r in results if r.get("status") == "skipped"),
- "failures": sum(1 for r in results
- if r.get("status") in ("error", "timeout", "exception",
- "adapter-missing")),
- "run_dir": str(run_dir),
- "results": results,
- }
- write_json(run_dir / "execution-summary.json", summary)
- print(json.dumps(summary, indent=2))
- return 0
- if __name__ == "__main__":
- sys.exit(main())
|