Spaces:
Running
Running
Download source/study/programbench_adapter.py from burtenshaw/beam-pi-programbench: direct link, hf CLI and curl.
- Browser
- Download file 33.1 kB
-
https://huggingface.co/spaces/burtenshaw/beam-pi-programbench/resolve/main/source/study/programbench_adapter.py
- Command line
-
hf download hf://spaces/burtenshaw/beam-pi-programbench/source/study/programbench_adapter.py
-
curl -L -o programbench_adapter.py https://huggingface.co/spaces/burtenshaw/beam-pi-programbench/resolve/main/source/study/programbench_adapter.py
33.1 kB
| #!/usr/bin/env python3 | |
| """Controller-only ProgramBench adapter. Never mount this checkout into a solver.""" | |
| from __future__ import annotations | |
| import argparse | |
| from datetime import datetime, timezone | |
| import hashlib | |
| import json | |
| import os | |
| from pathlib import Path, PurePosixPath | |
| import re | |
| import subprocess | |
| import sys | |
| import tarfile | |
| import tempfile | |
| import time | |
| from typing import Any | |
| import uuid | |
| BENCHMARK_COMMIT = "27f02157c785f8da3647aa6dbbe6b9137f99f10e" | |
| DATASET_COMMIT = "de0ddfb637590c7ecb54fa0b5301f6dc7dfbcee5" | |
| DEFAULT_MANIFEST = Path(__file__).resolve().parents[1] / "docs/programbench-five-task-manifest.json" | |
| CANONICAL_GIT_DIR = "/workspace/.pi-study/shared.git" | |
| TASK_PROMPT = """Recreate the reference program as an independent implementation. | |
| The reference executable is /reference/executable. Its bundled documentation is | |
| under /reference. Observe the reference only through its normal user interface: | |
| flags, input files, stdin, stdout, and exit status. Do not decompile, disassemble, | |
| trace, instrument, wrap, copy, or reuse the reference executable. Do not obtain | |
| the original source or an existing implementation. Internet access is disabled. | |
| The reference directory is read-only; your current directory is your own checkout. | |
| Write source code and an executable ./compile.sh which, on a fresh checkout in | |
| /workspace, produces ./executable without network access. Do not commit binaries | |
| or build artifacts. Commit and push your implementation to the shared main branch; | |
| only that canonical branch is submitted. Coordinate Git integration with peers. | |
| Use /reference/executable to compare behavior with your own implementation. | |
| """ | |
| def run(argv: list[str], *, timeout: int = 120, **kwargs: Any) -> subprocess.CompletedProcess: | |
| result = subprocess.run(argv, capture_output=True, text=True, timeout=timeout, **kwargs) | |
| if result.returncode: | |
| raise RuntimeError(f"{argv[0]} exited {result.returncode}: {result.stderr.strip()[-2000:]}") | |
| return result | |
| def sha256(path: Path) -> str: | |
| digest = hashlib.sha256() | |
| with path.open("rb") as stream: | |
| for chunk in iter(lambda: stream.read(1024 * 1024), b""): | |
| digest.update(chunk) | |
| return digest.hexdigest() | |
| def write_json(path: Path, value: Any) -> None: | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| temporary = path.with_suffix(path.suffix + ".tmp") | |
| temporary.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") | |
| temporary.replace(path) | |
| def load_manifest(path: Path) -> dict: | |
| manifest = json.loads(path.read_text()) | |
| if manifest["benchmark"]["commit"] != BENCHMARK_COMMIT: | |
| raise ValueError("Unexpected ProgramBench revision") | |
| if manifest["test_dataset"]["revision"] != DATASET_COMMIT: | |
| raise ValueError("Unexpected test dataset revision") | |
| tasks = manifest["tasks"] | |
| if len(tasks) != 5 or len({task["instance_id"] for task in tasks}) != 5: | |
| raise ValueError("Expected exactly five distinct pinned tasks") | |
| for task in tasks: | |
| image = task["task_cleanroom_image"] | |
| if image["platform"] != "linux/amd64" or not re.fullmatch(r"sha256:[a-f0-9]{64}", image["digest"]): | |
| raise ValueError("Expected a pinned Linux/amd64 image") | |
| if image["immutable_reference"] != image["repository"] + "@" + image["digest"]: | |
| raise ValueError("Image reference disagrees with its digest") | |
| return manifest | |
| def selected_task(manifest: dict, instance_id: str) -> dict: | |
| return next(task for task in manifest["tasks"] if task["instance_id"] == instance_id) | |
| def check_source(root: Path, manifest: dict) -> None: | |
| actual = run(["git", "-C", str(root), "rev-parse", "HEAD"]).stdout.strip() | |
| if actual != BENCHMARK_COMMIT: | |
| raise ValueError(f"ProgramBench checkout is {actual}, expected {BENCHMARK_COMMIT}") | |
| run(["git", "-C", str(root), "diff", "--exit-code", "HEAD", "--", "src/programbench", "pyproject.toml"]) | |
| for task in manifest["tasks"]: | |
| folder = root / "src/programbench/data/tasks" / task["instance_id"] | |
| for filename, key in [("task.yaml", "task_metadata"), ("tests.json", "public_test_manifest")]: | |
| if sha256(folder / filename) != task[key]["sha256"]: | |
| raise ValueError(f"Pinned metadata hash mismatch: {task['instance_id']}/{filename}") | |
| def check_local_image(docker: str, task: dict) -> dict: | |
| image = task["task_cleanroom_image"]["immutable_reference"] | |
| info = json.loads(run([docker, "image", "inspect", image]).stdout)[0] | |
| if info.get("Architecture") != "amd64" or info.get("Os") != "linux": | |
| raise ValueError("The local image is not Linux/amd64") | |
| return info | |
| def resource_record(args: argparse.Namespace) -> dict: | |
| mode = getattr(args, "resource_mode", "container") | |
| if mode == "slurm": | |
| if not os.environ.get("SLURM_JOB_ID"): | |
| raise RuntimeError("Slurm resource mode requires an active Slurm allocation") | |
| return {"resource_mode": "slurm", "container_limits_enforced": False, | |
| "slurm_job_id": os.environ["SLURM_JOB_ID"], | |
| "slurm_cpus_per_task": os.environ.get("SLURM_CPUS_PER_TASK"), | |
| "slurm_mem_per_node_mb": os.environ.get("SLURM_MEM_PER_NODE"), | |
| "pytest_workers": args.cpus} | |
| return {"resource_mode": "container", "container_limits_enforced": True, | |
| "cpus": args.cpus, "memory": args.memory} | |
| def resource_arguments(args: argparse.Namespace) -> list[str]: | |
| if resource_record(args)["resource_mode"] == "slurm": | |
| return [] | |
| return ["--cpus", str(args.cpus), "--memory", args.memory, "--memory-swap", args.memory] | |
| def prepare(args: argparse.Namespace, manifest: dict) -> dict: | |
| task = selected_task(manifest, args.instance_id) | |
| check_local_image(args.docker, task) | |
| output = args.output_dir.resolve() | |
| output.mkdir(parents=True, exist_ok=True) | |
| if (output / "prepared.json").exists(): | |
| raise FileExistsError("prepared.json already exists; choose a fresh episode directory") | |
| name = "pb-" + re.sub(r"[^a-z0-9_.-]", "-", args.episode_id.lower())[:48] | |
| image = task["task_cleanroom_image"]["immutable_reference"] | |
| command = [args.docker, "run", "--pull", "never", "--detach", "--init", "--name", name, | |
| "--label", "beam-pi-programbench=true", "--network", "none", "--user", "agent", | |
| "--cap-drop", "SYS_PTRACE", "--cap-drop", "NET_RAW", "--security-opt", "no-new-privileges", | |
| *resource_arguments(args), | |
| "--workdir", "/workspace", image, "sleep", str(args.ttl_seconds)] | |
| container = run(command, timeout=300).stdout.strip() | |
| try: | |
| # No host volumes, evaluator assets, provider keys, or host environment are injected. | |
| setup = """set -eu | |
| test -x /workspace/executable | |
| test ! -e /reference | |
| mkdir /reference | |
| find /workspace -mindepth 1 -maxdepth 1 -exec mv -t /reference -- {} + | |
| chown -R root:root /reference | |
| chmod -R a-w /reference | |
| mkdir -p /workspace/solution | |
| chown agent:agent /workspace /workspace/solution | |
| """ | |
| run([args.docker, "exec", "--user", "root", container, "bash", "-c", setup]) | |
| seed = """set -eu | |
| git init -q -b main | |
| git config user.name programbench-agent | |
| git config user.email agent@programbench.local | |
| printf '%s\\n' '/executable' '/build/' '/target/' '__pycache__/' > .gitignore | |
| git add .gitignore | |
| git -c commit.gpgsign=false commit -qm 'initial independent implementation' | |
| """ | |
| run([args.docker, "exec", "--user", "agent", "--workdir", "/workspace/solution", container, | |
| "bash", "-c", seed]) | |
| run([args.docker, "exec", "--user", "agent", container, "bash", "-c", | |
| "set -eu; for tool in bash git timeout setsid; do command -v \"$tool\" >/dev/null; done; " | |
| "timeout --help | grep -q -- --kill-after"]) | |
| info = json.loads(run([args.docker, "inspect", container]).stdout)[0] | |
| if info["HostConfig"]["NetworkMode"] != "none" or info["HostConfig"].get("Binds"): | |
| raise RuntimeError("Solver container isolation check failed") | |
| prompt = output / "task_prompt.txt" | |
| prompt.write_text(TASK_PROMPT) | |
| result = {"status": "prepared", "instance_id": args.instance_id, "container_id": container, | |
| "container_name": name, "docker_executable": args.docker, "user": "agent", | |
| "seed_dir": "/workspace/solution", "work_root": "/workspace", | |
| "canonical_git_dir": CANONICAL_GIT_DIR, "canonical_ref": "main", | |
| "reference_executable": "/reference/executable", "task_prompt_path": str(prompt), | |
| "task_prompt": TASK_PROMPT, "image": image, "network": "none", | |
| **resource_record(args)} | |
| write_json(output / "prepared.json", result) | |
| return result | |
| except BaseException: | |
| subprocess.run([args.docker, "rm", "-f", container], capture_output=True, timeout=60) | |
| raise | |
| def validate_archive(path: Path) -> None: | |
| """Reject host-style path escapes before the official in-container extraction.""" | |
| with tarfile.open(path, "r:gz") as archive: | |
| for member in archive: | |
| item = PurePosixPath(member.name) | |
| if item.is_absolute() or ".." in item.parts or member.isdev() or member.isfifo(): | |
| raise ValueError(f"Unsafe submission archive entry: {member.name}") | |
| if member.issym() or member.islnk(): | |
| target = PurePosixPath(member.linkname) | |
| parts = list(item.parent.parts) if member.issym() else [] | |
| if target.is_absolute(): | |
| raise ValueError("Submission links must stay within the archive") | |
| for part in target.parts: | |
| if part == "..": | |
| if not parts: | |
| raise ValueError("Submission link escapes the archive") | |
| parts.pop() | |
| elif part != ".": | |
| parts.append(part) | |
| def snapshot(args: argparse.Namespace, manifest: dict) -> dict: | |
| selected_task(manifest, args.instance_id) | |
| if args.ref != "main": | |
| raise ValueError("Only canonical main may be submitted") | |
| base = [args.docker, "exec", "--user", "agent", "--workdir", "/workspace", args.container, | |
| "git", "--git-dir", args.git_dir] | |
| commit = run(base + ["rev-parse", "--verify", "refs/heads/main^{commit}"]).stdout.strip() | |
| if not re.fullmatch(r"[a-f0-9]{40,64}", commit): | |
| raise ValueError("Invalid canonical commit") | |
| directory = args.output_dir.resolve() / args.instance_id | |
| directory.mkdir(parents=True, exist_ok=True) | |
| archive = directory / "submission.tar.gz" | |
| if archive.exists(): | |
| raise FileExistsError("Snapshot already exists; use a new output directory") | |
| temporary = archive.with_suffix(".tmp") | |
| try: | |
| with temporary.open("xb") as stream: | |
| result = subprocess.run(base + ["archive", "--format=tar.gz", commit], stdout=stream, | |
| stderr=subprocess.PIPE, timeout=300) | |
| if result.returncode: | |
| raise RuntimeError(result.stderr.decode(errors="replace")[-2000:]) | |
| validate_archive(temporary) | |
| temporary.replace(archive) | |
| finally: | |
| temporary.unlink(missing_ok=True) | |
| result = {"status": "snapshotted", "instance_id": args.instance_id, "canonical_commit": commit, | |
| "submission_archive": str(archive), "sha256": sha256(archive), | |
| "created_at_utc": datetime.now(timezone.utc).isoformat()} | |
| write_json(directory / "snapshot.json", result) | |
| return result | |
| def evaluator_api(args: argparse.Namespace, manifest: dict) -> dict: | |
| root = args.programbench_root.resolve() | |
| check_source(root, manifest) | |
| os.environ["PROGRAMBENCH_DOCKER_EXECUTABLE"] = args.docker | |
| os.environ["PROGRAMBENCH_HF_REVISION"] = DATASET_COMMIT | |
| os.environ["PROGRAMBENCH_HF_REPO"] = manifest["test_dataset"]["repository"] | |
| # Never silently use an unversioned external blob directory or authenticated HF state. | |
| os.environ.pop("PROGRAMBENCH_BLOB_DIR", None) | |
| os.environ["HF_HUB_DISABLE_IMPLICIT_TOKEN"] = "1" | |
| sys.path.insert(0, str(root / "src")) | |
| import programbench | |
| from programbench.eval.eval import Evaluator | |
| from programbench.submission import score_from_tests, test_results_map | |
| from programbench.utils.blob_store import get_blob_dir | |
| from programbench.utils.load_data import (get_active_branches, get_ignored_branches, | |
| get_ignored_tests, load_all_instances) | |
| if not Path(programbench.__file__).resolve().is_relative_to(root / "src"): | |
| raise RuntimeError("Imported ProgramBench outside the pinned checkout") | |
| return locals() | |
| def verify_wheelhouse(path: Path | None) -> str | None: | |
| if path is None: | |
| return None | |
| manifest = json.loads((path / "manifest.json").read_text()) | |
| filenames = {item["filename"] for item in manifest["wheels"]} | |
| if filenames != {item.name for item in path.glob("*.whl")}: | |
| raise ValueError("Wheelhouse contains missing or unrecorded wheels") | |
| for item in manifest["wheels"]: | |
| if Path(item["filename"]).name != item["filename"] or sha256(path / item["filename"]) != item["sha256"]: | |
| raise ValueError("Evaluator wheel hash mismatch") | |
| for filename, key in [("requirements.lock", "requirements_sha256"), ("constraints.txt", "constraints_sha256")]: | |
| if sha256(path / filename) != manifest[key]: | |
| raise ValueError("Evaluator dependency lock hash mismatch") | |
| return sha256(path / "manifest.json") | |
| def copy_evaluator_wheelhouse(environment: Any, path: Path) -> None: | |
| """Host UID values may exceed a rootless namespace; use container-root ownership.""" | |
| manifest = json.loads((path / "manifest.json").read_text()) | |
| names = [item["filename"] for item in manifest["wheels"]] + ["manifest.json", "constraints.txt", "requirements.lock"] | |
| with tempfile.NamedTemporaryFile(suffix=".tar") as temporary: | |
| with tarfile.open(fileobj=temporary, mode="w") as archive: | |
| for name in names: | |
| source = path / name | |
| info = archive.gettarinfo(str(source), arcname=name) | |
| if not info.isfile(): | |
| raise ValueError("Wheelhouse entries must be regular files") | |
| info.uid = info.gid = 0 | |
| info.uname = info.gname = "root" | |
| info.mode, info.mtime = 0o644, 0 | |
| with source.open("rb") as stream: | |
| archive.addfile(info, stream) | |
| temporary.flush() | |
| environment.copy_in_tar(Path(temporary.name), "/opt/programbench-evaluator-wheels") | |
| def copy_controller_artifact(environment: Any, source: Path, destination: str) -> None: | |
| """Copy pinned evaluator artifacts to explicit paths with rootless-safe ownership. | |
| File destinations name the new file; directory destinations receive contents. | |
| This covers the pinned evaluator, not every Docker cp destination convention. | |
| """ | |
| directory = source.is_dir() | |
| target = PurePosixPath(destination) | |
| with tempfile.NamedTemporaryFile(suffix=".tar") as temporary: | |
| def normalize(info): | |
| info.uid = info.gid = 0 | |
| info.uname = info.gname = "root" | |
| return info | |
| with tarfile.open(fileobj=temporary, mode="w") as archive: | |
| archive.add(source, arcname="." if directory else target.name, filter=normalize) | |
| temporary.flush() | |
| environment.copy_in_tar(Path(temporary.name), str(target if directory else target.parent)) | |
| def expected_mask(instance: dict, api: dict) -> list[str]: | |
| ignored = api["get_ignored_tests"](instance) | |
| return sorted({f"{branch}/{name}" for branch in api["get_active_branches"](instance) | |
| for name in instance["branches"][branch]["tests"] if f"{branch}/{name}" not in ignored}) | |
| def classify_score(summary: dict) -> dict: | |
| """Distinguish a failed submitted build from an unusable evaluator observation.""" | |
| candidate_errors = {"compile_failed", "clear_stale_executable_failed", | |
| "copy_executable_failed", "hash_executable_failed"} | |
| candidate_failed = (not summary["reference"] and summary["error_code"] in candidate_errors | |
| and summary["passed"] == 0 and summary["test_count"] > 0 | |
| and summary["test_count"] == summary["expected_test_count"] | |
| and not any(summary[key] for key in ("branch_error_count", "system_error_count", "warning_count", | |
| "missing_test_count", "unexpected_test_count"))) | |
| valid = summary["complete"] or candidate_failed | |
| return {"valid": valid, "analysis_score": summary["score"] if valid else None, | |
| "scoring_status": "submission_failed" if candidate_failed else "scored" if valid else "infrastructure_failed"} | |
| def evaluate(args: argparse.Namespace, task: dict, instance: dict, api: dict, | |
| directory: Path, *, reference: bool, submission: Path | None = None) -> dict: | |
| check_local_image(args.docker, task) | |
| directory.mkdir(parents=True, exist_ok=True) | |
| eval_path = directory / f"{task['instance_id']}.eval.json" | |
| if eval_path.exists(): | |
| raise FileExistsError(f"Refusing to overwrite evaluation: {eval_path}") | |
| branches = api["get_active_branches"](instance) | |
| ignored = api["get_ignored_tests"](instance) | |
| blob_dir = api["get_blob_dir"](task["instance_id"]) | |
| if blob_dir is None: | |
| raise RuntimeError("Pinned test archives unavailable") | |
| for branch in branches: | |
| if not (blob_dir / "tests" / f"{branch}.tar.gz").is_file(): | |
| raise FileNotFoundError("A required pinned test archive is missing") | |
| digest = task["task_cleanroom_image"]["immutable_reference"] | |
| Evaluator = api["Evaluator"] | |
| wheelhouse = getattr(args, "evaluator_wheelhouse", None) | |
| dependency_digest = verify_wheelhouse(wheelhouse) | |
| class PinnedEvaluator(Evaluator): | |
| def _new_env(self, image: str, *, serial_pytest: bool = False): | |
| from programbench.container import ContainerEnvironment | |
| class EvaluatorEnvironment(ContainerEnvironment): | |
| def copy_in(self, local_path: Path, container_path: str) -> None: | |
| # Docker cp and host tar ownership can both exceed rootless UID maps. | |
| copy_controller_artifact(self, local_path, container_path) | |
| # Evaluator.run's first image is name:tag; subsequent images are local build commits. | |
| initial = image == f"{self.image_name}:{self.image_tag}" | |
| if initial: | |
| image = digest | |
| env = {"PYTEST_ADDOPTS": "--max-worker-restart=4"} | |
| if wheelhouse is not None: | |
| env.update(PIP_NO_INDEX="1", PIP_FIND_LINKS="/opt/programbench-evaluator-wheels", | |
| PIP_CONSTRAINT="/opt/programbench-evaluator-wheels/constraints.txt", | |
| PIP_DISABLE_PIP_VERSION_CHECK="1", PIP_BREAK_SYSTEM_PACKAGES="1") | |
| if serial_pytest: | |
| env["PYTEST_XDIST_AUTO_NUM_WORKERS"] = "1" | |
| isolation = ["--pull", "never", "--network", "none", "--cap-drop", "SYS_PTRACE", | |
| "--cap-drop", "NET_RAW", "--security-opt", "no-new-privileges"] | |
| if resource_record(args)["resource_mode"] == "slurm": | |
| class SlurmEnvironment(EvaluatorEnvironment): | |
| # Upstream's constructor always injects --cpus. Retain its exec/copy/commit | |
| # implementation but omit unsupported daemon resource flags explicitly. | |
| def __init__(self): | |
| self.cwd, self.executable, self.default_timeout = "/workspace", args.docker, 600 | |
| self.cpus = args.cpus | |
| self._name = "programbench-" + uuid.uuid4().hex[:12] | |
| environment = {"PYTEST_XDIST_AUTO_NUM_WORKERS": str(args.cpus), **env} | |
| env_flags = [part for key, value in environment.items() for part in ("-e", f"{key}={value}")] | |
| command = [args.docker, "run", "-d", "--init", "--name", self._name, | |
| "-w", self.cwd, *env_flags, *isolation, image, "sleep", "2h"] | |
| self.container_id = run(command, timeout=300).stdout.strip() | |
| environment = SlurmEnvironment() | |
| else: | |
| environment = EvaluatorEnvironment(image=image, cwd="/workspace", executable=args.docker, | |
| timeout=600, cpus=args.cpus, env=env, | |
| run_args=[*isolation, "--memory", args.memory, "--memory-swap", args.memory]) | |
| if initial and wheelhouse is not None: | |
| try: | |
| copy_evaluator_wheelhouse(environment, wheelhouse) | |
| self._run_step("python3 -m pip install --no-index --find-links=/opt/programbench-evaluator-wheels " | |
| "--require-hashes --no-deps -r /opt/programbench-evaluator-wheels/requirements.lock", | |
| env=environment, log_buf=self.result.log, step_name="install_offline_evaluator_dependencies", timeout=180) | |
| self._run_step("python3 -m pytest --help | grep -q -- --timeout", | |
| env=environment, log_buf=self.result.log, step_name="verify_timeout_plugin", timeout=30) | |
| except BaseException: | |
| environment.cleanup() | |
| raise | |
| return environment | |
| def _compile_executable(self, env, log_buf): | |
| if not reference: | |
| return super()._compile_executable(env, log_buf) | |
| # Calibration keeps the image's reference binary; all branch execution is upstream. | |
| self._run_step(f"cp -L ./executable {self._stashed_executable}", env=env, log_buf=log_buf, | |
| step_name="stash_reference", timeout=300) | |
| result = self._run_step(f"sha256sum {self._stashed_executable}", env=env, log_buf=log_buf, | |
| step_name="hash_executable", timeout=300) | |
| self.result.executable_hash = result["output"].split()[0] | |
| install = self._run_step("pip3 install -q --disable-pip-version-check pytest-rerunfailures==16.4", | |
| env=env, log_buf=log_buf, step_name="install_rerunfailures", accept_failure=True, timeout=120) | |
| self._has_rerunfailures = install["returncode"] == 0 | |
| started = time.monotonic() | |
| evaluator = PinnedEvaluator(image_name=instance["image_name"], image_tag="task_cleanroom_v6", | |
| solution_branch="reference" if reference else "submission", submission_archive=submission, | |
| blob_dir=blob_dir, tests_branches=branches, | |
| tests_by_branch={branch: instance["branches"][branch]["tests"] for branch in branches}, | |
| ignored_tests=ignored, ignored_branches=api["get_ignored_branches"](instance), | |
| remove_hashes=instance.get("eval_clean_hashes", []), instance_id=task["instance_id"], | |
| docker_cpus=args.cpus, branch_workers=1, branch_retries=1) | |
| result = evaluator.run() | |
| eval_path.write_text(result.model_dump_json(indent=2)) | |
| tests = api["test_results_map"](eval_path, instance) | |
| mask = expected_mask(instance, api) | |
| write_json(directory / "official-mask.json", mask) | |
| filtered = result.for_branches(branches).without_ignored(ignored) | |
| summary = {"instance_id": task["instance_id"], "reference": reference, | |
| "score": api["score_from_tests"](tests), "passed": sum(tests.values()), "test_count": len(tests), | |
| "expected_test_count": len(mask), "missing_test_count": len(set(mask) - tests.keys()), | |
| "unexpected_test_count": len(tests.keys() - set(mask)), | |
| "not_run_count": sum(test.status == "not_run" for test in filtered), | |
| "error_code": result.error_code, | |
| "branch_error_count": len(filtered.test_branch_errors), "system_error_count": filtered.n_system_errors, | |
| "warning_count": len(result.warnings), "executable_hash": result.executable_hash, | |
| "wall_seconds": time.monotonic() - started, "eval_json": str(eval_path), | |
| "official_mask_sha256": sha256(directory / "official-mask.json"), | |
| "image": digest, "network": "none", "evaluator_dependencies_sha256": dependency_digest, | |
| **resource_record(args)} | |
| summary["complete"] = bool(tests) and not any(summary[key] for key in ( | |
| "error_code", "branch_error_count", "system_error_count", "warning_count", | |
| "missing_test_count", "unexpected_test_count", "not_run_count")) | |
| summary.update(classify_score(summary)) | |
| write_json(directory / "summary.json", summary) | |
| return summary | |
| def calibrate(args: argparse.Namespace, manifest: dict) -> dict: | |
| api = evaluator_api(args, manifest) | |
| instances = {item["instance_id"]: item for item in api["load_all_instances"]()} | |
| root = args.output_dir.resolve() | |
| root.mkdir(parents=True, exist_ok=True) | |
| if (root / "calibration.json").exists(): | |
| raise FileExistsError("Calibration already exists; choose a fresh directory") | |
| report = {"status": "running", "benchmark_commit": BENCHMARK_COMMIT, | |
| "dataset_commit": DATASET_COMMIT, "manifest_sha256": sha256(args.manifest), | |
| "evaluator_dependencies_sha256": verify_wheelhouse(args.evaluator_wheelhouse), | |
| "minimum_reference_score": args.min_reference_score, "repetitions": args.repetitions, | |
| "mask_policy": "Official active branches and ignored-test exclusions; no exclusions derived from reference failures.", | |
| "tasks": []} | |
| for task in manifest["tasks"]: | |
| repetitions = [] | |
| for repetition in range(args.repetitions): | |
| directory = root / task["instance_id"] / f"reference-{repetition + 1}" | |
| try: | |
| summary = evaluate(args, task, instances[task["instance_id"]], api, directory, reference=True) | |
| except Exception as exc: | |
| summary = {"instance_id": task["instance_id"], "complete": False, | |
| "error_code": type(exc).__name__, "error_details": str(exc), "score": None} | |
| write_json(directory / "summary.json", summary) | |
| repetitions.append(summary) | |
| passed = all(item["complete"] and item["score"] >= args.min_reference_score for item in repetitions) | |
| masks = {item.get("official_mask_sha256") for item in repetitions} | |
| report["tasks"].append({"instance_id": task["instance_id"], "passed": passed and len(masks) == 1, | |
| "repetitions": repetitions}) | |
| write_json(root / "calibration.json", report) | |
| report["status"] = "passed" if all(task["passed"] for task in report["tasks"]) else "failed" | |
| write_json(root / "calibration.json", report) | |
| return report | |
| def check_calibration(path: Path, manifest_path: Path, manifest: dict) -> dict: | |
| report = json.loads(path.read_text()) | |
| wanted = {task["instance_id"] for task in manifest["tasks"]} | |
| if report.get("status") != "passed" or report.get("manifest_sha256") != sha256(manifest_path): | |
| raise ValueError("Successful calibration of this exact manifest is required") | |
| if {task["instance_id"] for task in report["tasks"]} != wanted or not all(task["passed"] for task in report["tasks"]): | |
| raise ValueError("All five selected tasks must pass calibration; no silent task replacement") | |
| return report | |
| def grade(args: argparse.Namespace, manifest: dict) -> dict: | |
| calibration = check_calibration(args.calibration, args.manifest, manifest) | |
| if calibration.get("evaluator_dependencies_sha256") != verify_wheelhouse(args.evaluator_wheelhouse): | |
| raise ValueError("Evaluator dependency cache differs from calibration") | |
| validate_archive(args.submission) | |
| api = evaluator_api(args, manifest) | |
| instance = next(item for item in api["load_all_instances"]() if item["instance_id"] == args.instance_id) | |
| task = selected_task(manifest, args.instance_id) | |
| summary = evaluate(args, task, instance, api, args.output_dir.resolve() / args.instance_id, | |
| reference=False, submission=args.submission.resolve()) | |
| reference = next(item for item in calibration["tasks"] if item["instance_id"] == args.instance_id) | |
| if summary["official_mask_sha256"] != reference["repetitions"][0]["official_mask_sha256"]: | |
| raise ValueError("Evaluation mask differs from calibration") | |
| summary["submission_sha256"] = sha256(args.submission) | |
| write_json(args.output_dir.resolve() / args.instance_id / "summary.json", summary) | |
| return summary | |
| def probe(args: argparse.Namespace, manifest: dict) -> dict: | |
| """A setup diagnostic, never a substitute for whole-task calibration.""" | |
| api = evaluator_api(args, manifest) | |
| instance = next(item for item in api["load_all_instances"]() if item["instance_id"] == args.instance_id) | |
| branch = api["get_active_branches"](instance)[0] | |
| scoped = {**instance, "branches": {branch: instance["branches"][branch]}} | |
| task = selected_task(manifest, args.instance_id) | |
| result = evaluate(args, task, scoped, api, args.output_dir.resolve(), reference=True) | |
| result["diagnostic_scope"] = "first_active_branch_only_not_full_calibration" | |
| result["status"] = "passed" if result["complete"] and result["score"] >= 0.9 else "failed" | |
| write_json(args.output_dir.resolve() / "probe.json", result) | |
| return result | |
| def parser() -> argparse.ArgumentParser: | |
| result = argparse.ArgumentParser(description=__doc__) | |
| result.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST) | |
| result.add_argument("--programbench-root", type=Path) | |
| result.add_argument("--docker", default="docker") | |
| result.add_argument("--cpus", type=int, default=8) | |
| result.add_argument("--memory", default="16g") | |
| result.add_argument("--resource-mode", choices=["container", "slurm"], default="container", | |
| help="Use Slurm allocation limits when rootless Docker has no cgroups") | |
| result.add_argument("--evaluator-wheelhouse", type=Path, | |
| default=Path(os.environ["PROGRAMBENCH_EVALUATOR_WHEELHOUSE"]) if os.environ.get("PROGRAMBENCH_EVALUATOR_WHEELHOUSE") else None) | |
| commands = result.add_subparsers(dest="command", required=True) | |
| prepare_parser = commands.add_parser("prepare") | |
| prepare_parser.add_argument("--instance-id", required=True) | |
| prepare_parser.add_argument("--episode-id", required=True) | |
| prepare_parser.add_argument("--output-dir", type=Path, required=True) | |
| prepare_parser.add_argument("--ttl-seconds", type=int, default=10800) | |
| snapshot_parser = commands.add_parser("snapshot") | |
| snapshot_parser.add_argument("--instance-id", required=True) | |
| snapshot_parser.add_argument("--container", required=True) | |
| snapshot_parser.add_argument("--git-dir", default=CANONICAL_GIT_DIR) | |
| snapshot_parser.add_argument("--ref", default="main") | |
| snapshot_parser.add_argument("--output-dir", type=Path, required=True) | |
| calibration_parser = commands.add_parser("calibrate") | |
| calibration_parser.add_argument("--output-dir", type=Path, required=True) | |
| calibration_parser.add_argument("--repetitions", type=int, default=2) | |
| calibration_parser.add_argument("--min-reference-score", type=float, default=0.9) | |
| grade_parser = commands.add_parser("grade") | |
| grade_parser.add_argument("--instance-id", required=True) | |
| grade_parser.add_argument("--submission", type=Path, required=True) | |
| grade_parser.add_argument("--calibration", type=Path, required=True) | |
| grade_parser.add_argument("--output-dir", type=Path, required=True) | |
| probe_parser = commands.add_parser("probe") | |
| probe_parser.add_argument("--instance-id", required=True) | |
| probe_parser.add_argument("--output-dir", type=Path, required=True) | |
| return result | |
| def main() -> int: | |
| args = parser().parse_args() | |
| try: | |
| if args.cpus < 1: | |
| raise ValueError("cpus must be positive") | |
| if args.command in {"grade", "calibrate", "probe"} and args.programbench_root is None: | |
| raise ValueError("--programbench-root is required for evaluator commands") | |
| if args.command in {"grade", "calibrate"} and args.evaluator_wheelhouse is None: | |
| raise ValueError("--evaluator-wheelhouse is required for offline evaluation") | |
| if args.command == "calibrate" and (args.repetitions < 1 or not 0 <= args.min_reference_score <= 1): | |
| raise ValueError("Invalid calibration repetition count or threshold") | |
| manifest = load_manifest(args.manifest) | |
| output = {"prepare": prepare, "snapshot": snapshot, "calibrate": calibrate, "grade": grade, "probe": probe}[args.command](args, manifest) | |
| print(json.dumps(output, sort_keys=True)) | |
| return 1 if output.get("status") == "failed" else 0 | |
| except Exception as exc: | |
| print(json.dumps({"status": "error", "error_code": type(exc).__name__, "error_details": str(exc)})) | |
| return 1 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |