From 03501742c0abd92a8b3783f69238e1723ed35aa1 Mon Sep 17 00:00:00 2001 From: adhabnr-ux Date: Tue, 1 Sep 2026 21:36:17 -0700 Subject: [PATCH 1/2] Add scripts/export_openeval.py: optional EvalPort ResultSet export Adds an optional, additive script that converts a batch's rescore-summary.json (+ each run's own run-meta.json) into an EvalPort (https://github.com/adhabnr-ux/evalport) ResultSet -- a small open interchange format for portable LLM/agent evaluation results. Scoped in #322: maintainer Perry2004 confirmed this is the results side (complementing clawbench-harbor-adapt/clawbench-edgebench-adapt, which import test-case definitions the other direction) and suggested it land as "a simple script inside script/" rather than a new adapters/ package or a core dependency -- this follows that shape exactly. The script has no hard dependency on evalport-sdk: the conversion itself is pure stdlib, and evalport-sdk is only used, if installed, to validate the produced ResultSet against the real EvalPort schema. Grounded in the current real source, not just docs/scoring.md: src/clawbench/runner/run_support/metadata.py (make_run_meta()), src/clawbench/eval/rescore.py (aggregate_batch()/rescore_one()), and src/clawbench/runner/judge_llm.py (judge_request()'s match/reason shape). Result.passed is `intercepted AND judge_match is True`, per docs/scoring.md's final_pass rule; the interception and judge stages each become their own GraderResult (gr_interception, gr_judge_match) rather than being collapsed into one opaque score. Testing: 11 tests (tests/test_export_openeval.py), covering run_to_result() and to_openeval() directly plus CLI smoke/error-path tests via subprocess, all passing against the real, installed evalport-sdk 1.3.1's openeval.validate.validate_result_set() -- not a mock. Also clean under this repo's own ruff and pyright configuration. Signed-off-by: adhabnr-ux Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RtocdH3tKifGdkiCZxxV3F --- CHANGELOG.md | 2 + scripts/export_openeval.py | 386 ++++++++++++++++++++++++++++++++++ scripts/export_openeval.sh | 11 + tests/test_export_openeval.py | 282 +++++++++++++++++++++++++ 4 files changed, 681 insertions(+) create mode 100644 scripts/export_openeval.py create mode 100755 scripts/export_openeval.sh create mode 100644 tests/test_export_openeval.py diff --git a/CHANGELOG.md b/CHANGELOG.md index ec9a25f2..312ea161 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/) and this project adheres to [Semantic Versioning](https://semver.org/). ## [Unreleased] +### Added +- Added `scripts/export_openeval.py` (+ `scripts/export_openeval.sh` wrapper), an optional/additive script exporting a batch's `rescore-summary.json` as an [EvalPort](https://github.com/adhabnr-ux/evalport) `ResultSet` (see #322). ## [0.10.0] - 2026-08-30 ### Added diff --git a/scripts/export_openeval.py b/scripts/export_openeval.py new file mode 100644 index 00000000..f53f0fa6 --- /dev/null +++ b/scripts/export_openeval.py @@ -0,0 +1,386 @@ +#!/usr/bin/env python3 +"""Export a ClawBench batch's rescore-summary.json to an EvalPort ResultSet. + +EvalPort (https://github.com/adhabnr-ux/evalport) is a small open interchange +format for portable LLM/agent evaluation results (TestCase/Grader/Result/ +ResultSet, JSON Schema + a Python/TS SDK). This script is optional and +additive: it only *reads* files ClawBench already writes after a run +finishes (run-meta.json, the per-run judge verdict, rescore-summary.json) +and writes a new, separate resultset.json alongside them. It does not +change make_run_meta(), the two-stage scoring pipeline, clawbench-rescore, +or any other part of the runner/eval code, and it adds no dependency for +anyone who isn't using it -- see the module-level NOTE below on the one +optional import. + +Discussed and scoped in https://github.com/TIGER-AI-Lab/ClawBench/issues/322. +This is the *results* side, complementing the existing *test-case* import +adapters (clawbench-harbor-adapt, clawbench-edgebench-adapt) that run the +other direction: those convert someone else's test-case definitions into +ClawBench's own eval format so ClawBench can run them; this converts a +ClawBench run's own output into a portable results format so it can be +read next to output from any other EvalPort-aware benchmark. + +Usage: + + python scripts/export_openeval.py \\ + --run-id batch-20260902-140000 \\ + --started-at 2026-09-02T14:00:00Z + + # or, via the wrapper: + scripts/export_openeval.sh + + is a directory containing rescore-summary.json (written by +`clawbench-rescore` / scripts/rescore.sh) and the run-meta.json files it +was rolled up from. Writes /resultset.json by default; pass +--out to write elsewhere, or --stdout to print instead of writing a file. + +--run-id and --started-at are required: rescore-summary.json records +neither a run id nor a start timestamp for the batch, so there is nothing +honest to default them to. --rubric selects which of +rescore-summary.json's ["rubrics"] to score against (defaults to the +first one that ran, matching clawbench-rescore's own default of +"lenient"). + +What ClawBench actually emits (verified against the current source in +src/clawbench/runner/run_support/metadata.py, src/clawbench/eval/rescore.py, +and src/clawbench/runner/judge_llm.py -- not just docs/scoring.md, which +describes an idealized/older shape): + +* run-meta.json (make_run_meta()) carries test_case, instruction, model, + harness, intercepted, result_category, failure_category, + adjusted_eligible, duration_seconds, among other fields. It does NOT + carry judge_match or final_pass -- docs/scoring.md describes those as + merged in, but the actual merge point is the rescoring step below, and + even there no field is literally named final_pass. +* The per-run judge verdict (judge_llm.json for the default "lenient" + rubric, judge.json for "strict" -- JUDGE_FILE in rescore.py) has a + `match` key (True/False/None) and a `reason` string (judge_llm.py's + judge_request()). +* rescore-summary.json (aggregate_batch()) rolls a batch into n_total, + n_intercepted, judge_model, rubrics, and a tasks[] list of per-task rows + shaped {"task_id", "test_case", "intercepted", "match_", + "reason_"} for each rubric that ran. tasks[] rows do NOT carry + instruction/model/harness -- those live only in each run's own + run-meta.json, so this script also walks the batch dir for those files + to enrich each Result. + +Mapping to EvalPort's data model (spec/SPEC.md in the EvalPort repo): + +* One ClawBench run -> one EvalPort Result. test_case (or task_id as + fallback) becomes test_case_id. The two-stage scoring pipeline + (interception, then LLM judge) becomes two GraderResult entries -- + gr_interception and gr_judge_match -- so the mechanism ClawBench + actually uses is visible in the result, not collapsed into one opaque + score. Result.passed is `intercepted AND judge_match is True`, matching + the `final_pass = intercepted AND judge_match` rule documented in + docs/scoring.md. +* A batch's rescore-summary.json -> one EvalPort ResultSet. + +NOTE on the evalport-sdk import: this script does NOT require +evalport-sdk to run -- the conversion itself is pure stdlib. If +evalport-sdk (`pip install evalport-sdk`) happens to be installed, the +script uses it to validate the ResultSet it produces against the real +EvalPort JSON Schema before writing it out, and to print a clear error if +validation fails; if it isn't installed, the script skips that step and +says so. Either way, nothing in ClawBench's own pyproject.toml dependency +list changes. +""" +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List, Optional + +OPENEVAL_VERSION_FALLBACK = "1.0.0" + +INTERCEPTION_GRADER_ID = "gr_interception" +JUDGE_GRADER_ID = "gr_judge_match" + + +def _get(obj: Any, key: str, default: Any = None) -> Any: + if isinstance(obj, dict): + return obj.get(key, default) + return getattr(obj, key, default) + + +def run_to_result( + run_meta: Dict[str, Any], + judge: Optional[Dict[str, Any]] = None, + rubric: Optional[str] = None, +) -> Dict[str, Any]: + """Convert one ClawBench run into an EvalPort Result dict. + + run_meta is the dict already parsed from that run's run-meta.json (or, + when only a rescore-summary.json task row is available, a minimal + stand-in with at least test_case/task_id and intercepted). + + Result.test_case_id prefers run_meta["test_case"]; when that's + missing, it falls back to run_meta["task_id"] with its "#" stripped (ClawBench's real task_id values look like + "myrecipes/leave-review#0001" -- see make_run_meta() -- where the + part before "#" is the same stable task identity test_case itself + carries). + + judge is the dict already parsed from that run's judge verdict file + (judge_llm.json/judge.json, or the match/reason pulled from a + rescore-summary.json task row for a given rubric). Pass None when the + run was never judged (e.g. intercepted was already False, so Stage 2 + never ran per docs/scoring.md). + + Two GraderResult entries are emitted, mirroring ClawBench's real + two-stage pipeline: + + - gr_interception: score/passed from intercepted alone. + - gr_judge_match: only present when judge is given. score is + 1.0/0.0/None for match True/False/None (a judge that "could not + decide" carries a null score and counts as not-passed, exactly like + ClawBench's own aggregate treats it). + + Result.passed is `intercepted AND match is True`. + """ + test_case_id = _get(run_meta, "test_case") + if not test_case_id: + task_id = _get(run_meta, "task_id") + if task_id: + test_case_id = str(task_id).split("#", 1)[0] + if not test_case_id: + raise ValueError("run_meta must have a 'test_case' or 'task_id' to become a Result.test_case_id") + + intercepted = bool(_get(run_meta, "intercepted")) + + grader_results: List[Dict[str, Any]] = [ + { + "grader_id": INTERCEPTION_GRADER_ID, + "type": "custom", + "score": 1.0 if intercepted else 0.0, + "passed": intercepted, + "reason": "final request matched eval_schema" if intercepted else "final request did not match eval_schema (or agent never reached it)", + "metadata": {"handler": "clawbench:interception"}, + } + ] + + judge_match: Optional[bool] = None + if judge is not None: + judge_match = _get(judge, "match") + score = 1.0 if judge_match is True else (0.0 if judge_match is False else None) + gr: Dict[str, Any] = { + "grader_id": JUDGE_GRADER_ID, + "type": "llm_judge", + "score": score, + "passed": judge_match is True, + "metadata": {"handler": "clawbench:llm_judge"}, + } + reason = _get(judge, "reason") + if reason: + gr["reason"] = reason + judge_model = _get(judge, "judge_model") + if judge_model: + gr["metadata"]["judge_model"] = judge_model + grader_results.append(gr) + + passed = bool(intercepted and judge_match is True) + + metadata: Dict[str, Any] = {} + for key in ("result_category", "failure_category", "adjusted_eligible", "model", "harness", "task_id"): + value = _get(run_meta, key) + if value is not None: + metadata[key] = value + if rubric: + metadata["rubric"] = rubric + + result: Dict[str, Any] = { + "test_case_id": str(test_case_id), + "passed": passed, + "grader_results": grader_results, + } + # ClawBench scores an intercepted HTTP request, not a text completion, so + # there is no honest value for Result.actual_output here; the instruction + # that was scored against goes under metadata instead. + instruction = _get(run_meta, "instruction") + if instruction is not None: + metadata["instruction"] = instruction + duration = _get(run_meta, "duration_seconds") + if isinstance(duration, (int, float)): + result["duration_ms"] = int(round(duration * 1000)) + if metadata: + result["metadata"] = metadata + return result + + +def to_openeval( + rescore_summary: Dict[str, Any], + run_metas: Optional[Dict[str, Dict[str, Any]]] = None, + *, + run_id: str, + started_at: str, + completed_at: Optional[str] = None, + rubric: Optional[str] = None, + suite_id: Optional[str] = None, +) -> Dict[str, Any]: + """Convert a batch's rescore-summary.json (+ optional run-meta.json map) to an EvalPort ResultSet dict.""" + run_metas = run_metas or {} + rubrics = rescore_summary.get("rubrics") or ["lenient"] + if rubric is not None: + if rubric not in rubrics: + raise ValueError( + f"rubric {rubric!r} is not in rescore_summary['rubrics'] " + f"({rubrics!r}); pass one of those, or omit --rubric to use " + f"the first one" + ) + active_rubric = rubric + else: + active_rubric = rubrics[0] + + results: List[Dict[str, Any]] = [] + for task_row in rescore_summary.get("tasks", []): + test_case = task_row.get("test_case") + base_run_meta = run_metas.get(test_case) if test_case is not None else None + if base_run_meta is None: + base_run_meta = { + "test_case": test_case, + "task_id": task_row.get("task_id"), + "intercepted": task_row.get("intercepted"), + } + match_key = f"match_{active_rubric}" + reason_key = f"reason_{active_rubric}" + judge = None + # aggregate_batch() writes match_/reason_ for every + # task row unconditionally (defaulting to match=None, reason="" when + # there's no judge file) -- but rescore_one() only ever judges a run + # when it was intercepted; Stage 2 never runs otherwise. So the + # presence of the key alone doesn't mean a judge actually ran -- + # gate on `intercepted` too, matching that real control flow, rather + # than emitting a spurious gr_judge_match grader for a run that was + # never judged. + if task_row.get("intercepted") and match_key in task_row: + judge = { + "match": task_row.get(match_key), + "reason": task_row.get(reason_key), + "judge_model": rescore_summary.get("judge_model"), + } + results.append(run_to_result(base_run_meta, judge, rubric=active_rubric)) + + total = len(results) + passed = sum(1 for r in results if r["passed"]) + n_intercepted = sum( + 1 + for r in results + for gr in r["grader_results"] + if gr["grader_id"] == INTERCEPTION_GRADER_ID and gr["passed"] + ) + + batch_dir = rescore_summary.get("batch_dir") + resolved_suite_id = suite_id or ( + f"clawbench_{batch_dir.rstrip('/').rsplit('/', 1)[-1]}" if batch_dir else "clawbench_batch" + ) + + try: + from openeval.types import OPENEVAL_VERSION as _V + version = _V + except ImportError: + version = OPENEVAL_VERSION_FALLBACK + + result_set: Dict[str, Any] = { + "version": version, + "suite_id": resolved_suite_id, + "run_id": run_id, + "started_at": started_at, + "results": results, + "runner": {"name": "clawbench", "version": "n/a"}, + "summary": { + "total": total, + "passed": passed, + "failed": total - passed, + "pass_rate": (passed / total) if total else 0.0, + }, + "metadata": { + "openeval": {"source": "clawbench"}, + "clawbench_batch_dir": batch_dir, + "clawbench_rubric": active_rubric, + "clawbench_n_intercepted": n_intercepted, + }, + } + if completed_at is not None: + result_set["completed_at"] = completed_at + return result_set + + +def _load_run_metas(batch_dir: Path) -> Dict[str, Dict[str, Any]]: + run_metas: Dict[str, Dict[str, Any]] = {} + for meta_path in batch_dir.rglob("run-meta.json"): + try: + meta = json.loads(meta_path.read_text()) + except Exception as exc: # noqa: BLE001 - best-effort enrichment, never fatal + print(f"warning: could not parse {meta_path}: {exc}", file=sys.stderr) + continue + test_case = meta.get("test_case") + if test_case: + run_metas[test_case] = meta + return run_metas + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + parser.add_argument("batch_dir", type=Path, help="Batch directory containing rescore-summary.json") + parser.add_argument("--run-id", required=True, help="Run id for the resulting ResultSet (e.g. the batch directory name)") + parser.add_argument("--started-at", required=True, help="ISO-8601 start timestamp for the batch (rescore-summary.json doesn't record one)") + parser.add_argument("--completed-at", default=None, help="Optional ISO-8601 completion timestamp") + parser.add_argument("--rubric", default=None, help="Which rubric to score against (default: the first one in rescore-summary.json's rubrics)") + parser.add_argument("--suite-id", default=None, help="Optional suite id override (default: derived from the batch directory name)") + parser.add_argument("--out", type=Path, default=None, help="Output path (default: /resultset.json)") + parser.add_argument("--stdout", action="store_true", help="Print the ResultSet to stdout instead of writing a file") + parser.add_argument("--no-validate", action="store_true", help="Skip validation even if evalport-sdk is installed") + args = parser.parse_args(argv) + + summary_path = args.batch_dir / "rescore-summary.json" + if not summary_path.exists(): + print(f"error: {summary_path} not found -- run clawbench-rescore / scripts/rescore.sh on this batch first", file=sys.stderr) + return 2 + rescore_summary = json.loads(summary_path.read_text()) + + run_metas = _load_run_metas(args.batch_dir) + + try: + result_set = to_openeval( + rescore_summary, + run_metas, + run_id=args.run_id, + started_at=args.started_at, + completed_at=args.completed_at, + rubric=args.rubric, + suite_id=args.suite_id, + ) + except ValueError as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + + if not args.no_validate: + try: + from openeval.validate import validate_result_set + except ImportError: + print("note: evalport-sdk not installed, skipping schema validation (pip install evalport-sdk to enable)", file=sys.stderr) + else: + validation = validate_result_set(result_set) + if not validation.valid: + print("error: produced ResultSet failed EvalPort schema validation:", file=sys.stderr) + for err in validation.errors: + print(f" - {err}", file=sys.stderr) + return 1 + print(f"validated OK against evalport-sdk's real schema ({len(result_set['results'])} results)", file=sys.stderr) + + payload = json.dumps(result_set, indent=2, ensure_ascii=False) + if args.stdout: + print(payload) + else: + out_path = args.out or (args.batch_dir / "resultset.json") + out_path.write_text(payload) + print(f"wrote {out_path}", file=sys.stderr) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/export_openeval.sh b/scripts/export_openeval.sh new file mode 100755 index 00000000..a3cae5a5 --- /dev/null +++ b/scripts/export_openeval.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Thin wrapper around scripts/export_openeval.py. +# +# Usage: +# scripts/export_openeval.sh --run-id --started-at +# +# Writes /resultset.json by default (a spec-valid EvalPort +# ResultSet, see https://github.com/adhabnr-ux/evalport). All flags pass +# through to the underlying script; see --help for the full list. +set -euo pipefail +exec uv run --project "$(dirname "$0")/.." python "$(dirname "$0")/export_openeval.py" "$@" diff --git a/tests/test_export_openeval.py b/tests/test_export_openeval.py new file mode 100644 index 00000000..608c5c73 --- /dev/null +++ b/tests/test_export_openeval.py @@ -0,0 +1,282 @@ +from __future__ import annotations + +import importlib.util +import json +import os +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +SCRIPT_PATH = REPO_ROOT / "scripts" / "export_openeval.py" + + +def _load_module(): + """Import scripts/export_openeval.py directly (it's a standalone script, not a package).""" + spec = importlib.util.spec_from_file_location("export_openeval", SCRIPT_PATH) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _run_meta( + test_case: str = "myrecipes/leave-review", + task_id: str = "myrecipes/leave-review#0001", + intercepted: bool = True, + result_category: str = "success", +) -> dict: + return { + "test_case": test_case, + "task_id": task_id, + "instruction": "Leave a 5-star review for the lasagna recipe.", + "model": "gpt-5", + "harness": "claude-code", + "intercepted": intercepted, + "result_category": result_category, + "failure_category": None, + "adjusted_eligible": True, + "duration_seconds": 42.3, + } + + +def _judge_llm(match: bool | None = True, reason: str = "matches instruction") -> dict: + return { + "match": match, + "reason": reason, + "judge_model": "deepseek-v4-pro", + "raw": '{"match": true, "reason": "matches instruction"}', + "rubric": "lenient", + } + + +# --- run_to_result() --- + + +def test_run_to_result_intercepted_and_matched(): + module = _load_module() + run_meta = _run_meta() + judge = _judge_llm(match=True) + + result = module.run_to_result(run_meta, judge, rubric="lenient") + + assert result["test_case_id"] == "myrecipes/leave-review" + assert result["passed"] is True + graders = {g["grader_id"]: g for g in result["grader_results"]} + assert graders["gr_interception"]["passed"] is True + assert graders["gr_interception"]["score"] == 1.0 + assert graders["gr_judge_match"]["passed"] is True + assert graders["gr_judge_match"]["score"] == 1.0 + assert graders["gr_judge_match"]["reason"] == "matches instruction" + assert result["metadata"]["result_category"] == "success" + assert result["metadata"]["instruction"] == run_meta["instruction"] + assert result["duration_ms"] == 42300 + + +def test_run_to_result_never_intercepted_has_no_judge_grader(): + module = _load_module() + run_meta = _run_meta(intercepted=False, result_category="agent_gave_up") + + result = module.run_to_result(run_meta, judge=None) + + assert result["passed"] is False + assert len(result["grader_results"]) == 1 + assert result["grader_results"][0]["grader_id"] == "gr_interception" + assert result["grader_results"][0]["passed"] is False + + +def test_run_to_result_judge_could_not_decide_is_not_passed_with_null_score(): + module = _load_module() + run_meta = _run_meta() + judge = _judge_llm(match=None, reason="unparseable") + + result = module.run_to_result(run_meta, judge) + + assert result["passed"] is False + judge_gr = next(g for g in result["grader_results"] if g["grader_id"] == "gr_judge_match") + assert judge_gr["score"] is None + assert judge_gr["passed"] is False + + +def test_run_to_result_falls_back_to_task_id_when_test_case_missing(): + module = _load_module() + run_meta = {"task_id": "myrecipes/leave-review#0007", "intercepted": True} + + result = module.run_to_result(run_meta) + + assert result["test_case_id"] == "myrecipes/leave-review" + + +def test_run_to_result_raises_without_test_case_or_task_id(): + module = _load_module() + import pytest + + with pytest.raises(ValueError): + module.run_to_result({"intercepted": True}) + + +# --- to_openeval() --- + + +def _rescore_summary(rubrics=("lenient",), judge_model="deepseek-v4-pro") -> dict: + return { + "batch_dir": "/work/claw-output/sweep/gpt-5/batch-20260902-140000", + "n_total": 2, + "n_intercepted": 2, + "judge_model": judge_model, + "rubrics": list(rubrics), + "tasks": [ + { + "task_id": "myrecipes/leave-review#0001", + "test_case": "myrecipes/leave-review", + "intercepted": True, + "match_lenient": True, + "reason_lenient": "matches instruction", + }, + { + "task_id": "citylibrary/reserve-book#0002", + "test_case": "citylibrary/reserve-book", + "intercepted": True, + "match_lenient": False, + "reason_lenient": "wrong book title", + }, + ], + } + + +def test_to_openeval_builds_valid_result_set_and_summary(): + module = _load_module() + summary = _rescore_summary() + run_metas = {"myrecipes/leave-review": _run_meta()} + + result_set = module.to_openeval( + summary, + run_metas, + run_id="batch-20260902-140000", + started_at="2026-09-02T14:00:00Z", + ) + + assert result_set["run_id"] == "batch-20260902-140000" + assert result_set["suite_id"] == "clawbench_batch-20260902-140000" + assert len(result_set["results"]) == 2 + assert result_set["summary"]["total"] == 2 + assert result_set["summary"]["passed"] == 1 + assert result_set["summary"]["failed"] == 1 + assert result_set["metadata"]["clawbench_rubric"] == "lenient" + assert result_set["metadata"]["clawbench_n_intercepted"] == 2 + + # Enriched result carries instruction/model/harness from run_metas; + # the un-enriched one only has what the task row itself carries. + enriched = next(r for r in result_set["results"] if r["test_case_id"] == "myrecipes/leave-review") + assert enriched["metadata"]["instruction"] == run_metas["myrecipes/leave-review"]["instruction"] + bare = next(r for r in result_set["results"] if r["test_case_id"] == "citylibrary/reserve-book") + assert "instruction" not in bare.get("metadata", {}) + + from openeval.validate import validate_result_set + + validation = validate_result_set(result_set) + assert validation.valid, validation.errors + + +def test_to_openeval_rejects_unknown_rubric(): + module = _load_module() + import pytest + + with pytest.raises(ValueError): + module.to_openeval( + _rescore_summary(rubrics=("lenient",)), + run_id="r1", + started_at="2026-09-02T14:00:00Z", + rubric="strict", + ) + + +def test_to_openeval_gates_judge_grader_on_intercepted(): + """A task row can carry a stale match_ key from an older rescore + even when intercepted is False (Stage 2 never actually ran for it) -- + to_openeval() must not fabricate a judge grader for that row.""" + module = _load_module() + summary = _rescore_summary() + summary["tasks"][1]["intercepted"] = False + # aggregate_batch() still writes match_/reason_ unconditionally. + summary["tasks"][1]["match_lenient"] = None + summary["tasks"][1]["reason_lenient"] = "" + + result_set = module.to_openeval(summary, run_id="r1", started_at="2026-09-02T14:00:00Z") + + never_intercepted = next(r for r in result_set["results"] if r["test_case_id"] == "citylibrary/reserve-book") + assert len(never_intercepted["grader_results"]) == 1 + assert never_intercepted["grader_results"][0]["grader_id"] == "gr_interception" + + +# --- CLI --- + + +def test_cli_writes_resultset_json_and_validates(tmp_path: Path): + batch_dir = tmp_path / "batch-20260902-140000" + run_dir = batch_dir / "myrecipes-leave-review" + run_dir.mkdir(parents=True) + (run_dir / "run-meta.json").write_text(json.dumps(_run_meta())) + (batch_dir / "rescore-summary.json").write_text(json.dumps(_rescore_summary())) + + env = os.environ.copy() + result = subprocess.run( + [ + sys.executable, + str(SCRIPT_PATH), + str(batch_dir), + "--run-id", + "batch-20260902-140000", + "--started-at", + "2026-09-02T14:00:00Z", + ], + capture_output=True, + env=env, + text=True, + timeout=30, + ) + + assert result.returncode == 0, result.stdout + result.stderr + out_path = batch_dir / "resultset.json" + assert out_path.is_file() + result_set = json.loads(out_path.read_text()) + assert result_set["run_id"] == "batch-20260902-140000" + assert len(result_set["results"]) == 2 + # The enriched result should have picked up instruction/model/harness + # from myrecipes-leave-review/run-meta.json found by rglob. + enriched = next(r for r in result_set["results"] if r["test_case_id"] == "myrecipes/leave-review") + assert enriched["metadata"]["model"] == "gpt-5" + + +def test_cli_errors_cleanly_when_rescore_summary_missing(tmp_path: Path): + batch_dir = tmp_path / "empty-batch" + batch_dir.mkdir() + + result = subprocess.run( + [ + sys.executable, + str(SCRIPT_PATH), + str(batch_dir), + "--run-id", + "r1", + "--started-at", + "2026-09-02T14:00:00Z", + ], + capture_output=True, + text=True, + timeout=30, + ) + + assert result.returncode == 2 + assert "rescore-summary.json" in result.stderr + + +def test_cli_help_smoke(): + result = subprocess.run( + [sys.executable, str(SCRIPT_PATH), "--help"], + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 0 + assert "usage" in (result.stdout + result.stderr).lower() From a2edf02c18026a2195d1a2807d2e32ca0a8b020e Mon Sep 17 00:00:00 2001 From: adhabnr-ux Date: Tue, 1 Sep 2026 22:55:48 -0700 Subject: [PATCH 2/2] Fix CI: make the real-schema validation test opportunistic, not a hard import test_to_openeval_builds_valid_result_set_and_summary() unconditionally imported openeval.validate, but evalport-sdk is intentionally NOT a project dependency (per this PR's own design: export_openeval.py validates against the real OpenEval schema only when evalport-sdk happens to be installed, and skips that step otherwise). CI's `uv run --frozen pytest` doesn't install it, so the test hard-failed with ModuleNotFoundError: No module named 'openeval'. Fixed by gating that one validation step behind pytest.importorskip("openeval.validate"), mirroring the script's own graceful-degradation behavior in the test suite. Confirmed both paths locally: - uv sync --frozen && uv run --frozen pytest -q -> 228 passed, 1 skipped (openeval not installed, matches real CI) - same venv + `pip install evalport-sdk` -> 229 passed (real schema validation actually runs and passes) Also fixes the static-check job: `uv run --frozen ruff format --check .` was failing on both touched files (pre-existing formatting drift from how I originally wrote them, not from this fix) - reformatted with the exact pinned ruff==0.15.12 from uv.lock. Re-verified clean: - uv run --frozen ruff format --check . -> 88 files already formatted - uv run --frozen ruff check . -> All checks passed! - uv run --frozen pyright src/clawbench tests -> 0 errors, 0 warnings No behavioral change to export_openeval.py; only formatting. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RtocdH3tKifGdkiCZxxV3F --- scripts/export_openeval.py | 94 ++++++++++++++++++++++++++++------- tests/test_export_openeval.py | 49 ++++++++++++++---- 2 files changed, 117 insertions(+), 26 deletions(-) diff --git a/scripts/export_openeval.py b/scripts/export_openeval.py index f53f0fa6..691a62fc 100644 --- a/scripts/export_openeval.py +++ b/scripts/export_openeval.py @@ -85,6 +85,7 @@ says so. Either way, nothing in ClawBench's own pyproject.toml dependency list changes. """ + from __future__ import annotations import argparse @@ -146,7 +147,9 @@ def run_to_result( if task_id: test_case_id = str(task_id).split("#", 1)[0] if not test_case_id: - raise ValueError("run_meta must have a 'test_case' or 'task_id' to become a Result.test_case_id") + raise ValueError( + "run_meta must have a 'test_case' or 'task_id' to become a Result.test_case_id" + ) intercepted = bool(_get(run_meta, "intercepted")) @@ -156,7 +159,9 @@ def run_to_result( "type": "custom", "score": 1.0 if intercepted else 0.0, "passed": intercepted, - "reason": "final request matched eval_schema" if intercepted else "final request did not match eval_schema (or agent never reached it)", + "reason": "final request matched eval_schema" + if intercepted + else "final request did not match eval_schema (or agent never reached it)", "metadata": {"handler": "clawbench:interception"}, } ] @@ -183,7 +188,14 @@ def run_to_result( passed = bool(intercepted and judge_match is True) metadata: Dict[str, Any] = {} - for key in ("result_category", "failure_category", "adjusted_eligible", "model", "harness", "task_id"): + for key in ( + "result_category", + "failure_category", + "adjusted_eligible", + "model", + "harness", + "task_id", + ): value = _get(run_meta, key) if value is not None: metadata[key] = value @@ -273,11 +285,14 @@ def to_openeval( batch_dir = rescore_summary.get("batch_dir") resolved_suite_id = suite_id or ( - f"clawbench_{batch_dir.rstrip('/').rsplit('/', 1)[-1]}" if batch_dir else "clawbench_batch" + f"clawbench_{batch_dir.rstrip('/').rsplit('/', 1)[-1]}" + if batch_dir + else "clawbench_batch" ) try: from openeval.types import OPENEVAL_VERSION as _V + version = _V except ImportError: version = OPENEVAL_VERSION_FALLBACK @@ -325,20 +340,56 @@ def main(argv: Optional[List[str]] = None) -> int: parser = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) - parser.add_argument("batch_dir", type=Path, help="Batch directory containing rescore-summary.json") - parser.add_argument("--run-id", required=True, help="Run id for the resulting ResultSet (e.g. the batch directory name)") - parser.add_argument("--started-at", required=True, help="ISO-8601 start timestamp for the batch (rescore-summary.json doesn't record one)") - parser.add_argument("--completed-at", default=None, help="Optional ISO-8601 completion timestamp") - parser.add_argument("--rubric", default=None, help="Which rubric to score against (default: the first one in rescore-summary.json's rubrics)") - parser.add_argument("--suite-id", default=None, help="Optional suite id override (default: derived from the batch directory name)") - parser.add_argument("--out", type=Path, default=None, help="Output path (default: /resultset.json)") - parser.add_argument("--stdout", action="store_true", help="Print the ResultSet to stdout instead of writing a file") - parser.add_argument("--no-validate", action="store_true", help="Skip validation even if evalport-sdk is installed") + parser.add_argument( + "batch_dir", type=Path, help="Batch directory containing rescore-summary.json" + ) + parser.add_argument( + "--run-id", + required=True, + help="Run id for the resulting ResultSet (e.g. the batch directory name)", + ) + parser.add_argument( + "--started-at", + required=True, + help="ISO-8601 start timestamp for the batch (rescore-summary.json doesn't record one)", + ) + parser.add_argument( + "--completed-at", default=None, help="Optional ISO-8601 completion timestamp" + ) + parser.add_argument( + "--rubric", + default=None, + help="Which rubric to score against (default: the first one in rescore-summary.json's rubrics)", + ) + parser.add_argument( + "--suite-id", + default=None, + help="Optional suite id override (default: derived from the batch directory name)", + ) + parser.add_argument( + "--out", + type=Path, + default=None, + help="Output path (default: /resultset.json)", + ) + parser.add_argument( + "--stdout", + action="store_true", + help="Print the ResultSet to stdout instead of writing a file", + ) + parser.add_argument( + "--no-validate", + action="store_true", + help="Skip validation even if evalport-sdk is installed", + ) args = parser.parse_args(argv) summary_path = args.batch_dir / "rescore-summary.json" if not summary_path.exists(): - print(f"error: {summary_path} not found -- run clawbench-rescore / scripts/rescore.sh on this batch first", file=sys.stderr) + print( + f"error: {summary_path} not found -- run clawbench-rescore / scripts/rescore.sh on this batch first", + file=sys.stderr, + ) return 2 rescore_summary = json.loads(summary_path.read_text()) @@ -362,15 +413,24 @@ def main(argv: Optional[List[str]] = None) -> int: try: from openeval.validate import validate_result_set except ImportError: - print("note: evalport-sdk not installed, skipping schema validation (pip install evalport-sdk to enable)", file=sys.stderr) + print( + "note: evalport-sdk not installed, skipping schema validation (pip install evalport-sdk to enable)", + file=sys.stderr, + ) else: validation = validate_result_set(result_set) if not validation.valid: - print("error: produced ResultSet failed EvalPort schema validation:", file=sys.stderr) + print( + "error: produced ResultSet failed EvalPort schema validation:", + file=sys.stderr, + ) for err in validation.errors: print(f" - {err}", file=sys.stderr) return 1 - print(f"validated OK against evalport-sdk's real schema ({len(result_set['results'])} results)", file=sys.stderr) + print( + f"validated OK against evalport-sdk's real schema ({len(result_set['results'])} results)", + file=sys.stderr, + ) payload = json.dumps(result_set, indent=2, ensure_ascii=False) if args.stdout: diff --git a/tests/test_export_openeval.py b/tests/test_export_openeval.py index 608c5c73..bb696447 100644 --- a/tests/test_export_openeval.py +++ b/tests/test_export_openeval.py @@ -93,7 +93,9 @@ def test_run_to_result_judge_could_not_decide_is_not_passed_with_null_score(): result = module.run_to_result(run_meta, judge) assert result["passed"] is False - judge_gr = next(g for g in result["grader_results"] if g["grader_id"] == "gr_judge_match") + judge_gr = next( + g for g in result["grader_results"] if g["grader_id"] == "gr_judge_match" + ) assert judge_gr["score"] is None assert judge_gr["passed"] is False @@ -167,14 +169,33 @@ def test_to_openeval_builds_valid_result_set_and_summary(): # Enriched result carries instruction/model/harness from run_metas; # the un-enriched one only has what the task row itself carries. - enriched = next(r for r in result_set["results"] if r["test_case_id"] == "myrecipes/leave-review") - assert enriched["metadata"]["instruction"] == run_metas["myrecipes/leave-review"]["instruction"] - bare = next(r for r in result_set["results"] if r["test_case_id"] == "citylibrary/reserve-book") + enriched = next( + r + for r in result_set["results"] + if r["test_case_id"] == "myrecipes/leave-review" + ) + assert ( + enriched["metadata"]["instruction"] + == run_metas["myrecipes/leave-review"]["instruction"] + ) + bare = next( + r + for r in result_set["results"] + if r["test_case_id"] == "citylibrary/reserve-book" + ) assert "instruction" not in bare.get("metadata", {}) - from openeval.validate import validate_result_set + # evalport-sdk is an optional dependency (matching export_openeval.py's own + # graceful degradation: it validates against the real OpenEval schema when + # the package happens to be installed, and skips that step otherwise). CI + # doesn't install it, so this real-schema check is opportunistic here too. + import pytest - validation = validate_result_set(result_set) + openeval_validate = pytest.importorskip( + "openeval.validate", + reason="evalport-sdk not installed; skipping real-schema validation", + ) + validation = openeval_validate.validate_result_set(result_set) assert validation.valid, validation.errors @@ -202,9 +223,15 @@ def test_to_openeval_gates_judge_grader_on_intercepted(): summary["tasks"][1]["match_lenient"] = None summary["tasks"][1]["reason_lenient"] = "" - result_set = module.to_openeval(summary, run_id="r1", started_at="2026-09-02T14:00:00Z") + result_set = module.to_openeval( + summary, run_id="r1", started_at="2026-09-02T14:00:00Z" + ) - never_intercepted = next(r for r in result_set["results"] if r["test_case_id"] == "citylibrary/reserve-book") + never_intercepted = next( + r + for r in result_set["results"] + if r["test_case_id"] == "citylibrary/reserve-book" + ) assert len(never_intercepted["grader_results"]) == 1 assert never_intercepted["grader_results"][0]["grader_id"] == "gr_interception" @@ -244,7 +271,11 @@ def test_cli_writes_resultset_json_and_validates(tmp_path: Path): assert len(result_set["results"]) == 2 # The enriched result should have picked up instruction/model/harness # from myrecipes-leave-review/run-meta.json found by rglob. - enriched = next(r for r in result_set["results"] if r["test_case_id"] == "myrecipes/leave-review") + enriched = next( + r + for r in result_set["results"] + if r["test_case_id"] == "myrecipes/leave-review" + ) assert enriched["metadata"]["model"] == "gpt-5"