Spaces:
Running
Running
File size: 14,515 Bytes
5741b22 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 | """Offline checks: no Docker daemon, test downloads, or model calls."""
import argparse
import io
import json
import os
from pathlib import Path
import subprocess
import tarfile
from unittest.mock import patch
import pytest
from study import programbench_adapter as adapter
@pytest.fixture
def manifest():
return adapter.load_manifest(adapter.DEFAULT_MANIFEST)
def args(tmp_path):
return argparse.Namespace(docker="docker", cpus=2, memory="4g", output_dir=tmp_path,
manifest=adapter.DEFAULT_MANIFEST, programbench_root=Path(os.environ.get(
"PROGRAMBENCH_TEST_ROOT", "/tmp/programbench-adapter-source")),
repetitions=2, min_reference_score=0.9, evaluator_wheelhouse=None)
def official_api():
pytest.importorskip("programbench")
from programbench.eval.eval import EvaluationResult, TestResult
from programbench.submission import score_from_tests, test_results_map
from programbench.utils.load_data import get_active_branches, get_ignored_branches, get_ignored_tests
return locals()
def test_manifest_rejects_mutable_image(tmp_path, manifest):
manifest["tasks"][0]["task_cleanroom_image"]["immutable_reference"] = "image:latest"
path = tmp_path / "manifest.json"
path.write_text(json.dumps(manifest))
with pytest.raises(ValueError, match="disagrees"):
adapter.load_manifest(path)
def test_slurm_mode_omits_unsupported_flags_and_records_allocation(tmp_path):
request = args(tmp_path)
request.resource_mode = "slurm"
with patch.dict(os.environ, {"SLURM_JOB_ID": "123", "SLURM_CPUS_PER_TASK": "16", "SLURM_MEM_PER_NODE": "30720"}):
assert adapter.resource_arguments(request) == []
record = adapter.resource_record(request)
assert record["container_limits_enforced"] is False
assert record["slurm_cpus_per_task"] == "16"
assert record["slurm_mem_per_node_mb"] == "30720"
def test_slurm_mode_rejects_execution_outside_allocation(tmp_path):
request = args(tmp_path)
request.resource_mode = "slurm"
with patch.dict(os.environ, {}, clear=True), pytest.raises(RuntimeError, match="active Slurm"):
adapter.resource_arguments(request)
@pytest.mark.parametrize("error_code,valid", [("compile_failed", True), ("copy_executable_failed", True),
("hash_executable_failed", True), ("wipe_workspace_failed", False), ("seed_git_failed", False)])
def test_candidate_build_failure_is_zero_but_infrastructure_failure_is_unscored(error_code, valid):
summary = {"reference": False, "complete": False, "error_code": error_code,
"passed": 0, "test_count": 10, "expected_test_count": 10, "score": 0.0,
"branch_error_count": 0, "system_error_count": 0, "warning_count": 0,
"missing_test_count": 0, "unexpected_test_count": 0, "not_run_count": 10}
result = adapter.classify_score(summary)
assert result["valid"] is valid
assert result["analysis_score"] == (0.0 if valid else None)
assert result["scoring_status"] == ("submission_failed" if valid else "infrastructure_failed")
def test_reference_failure_never_counts_as_valid_candidate_zero():
summary = {"reference": True, "complete": False, "error_code": "compile_failed",
"score": 0.0}
assert adapter.classify_score(summary)["valid"] is False
def test_wheelhouse_refuses_tampered_packages(tmp_path):
wheel = tmp_path / "pytest_timeout-2.4.0-py3-none-any.whl"
wheel.write_bytes(b"pinned package bytes")
(tmp_path / "requirements.lock").write_text("pinned requirements")
(tmp_path / "constraints.txt").write_text("pinned constraints")
manifest = {"wheels": [{"filename": wheel.name, "sha256": adapter.sha256(wheel)}],
"requirements_sha256": adapter.sha256(tmp_path / "requirements.lock"),
"constraints_sha256": adapter.sha256(tmp_path / "constraints.txt")}
adapter.write_json(tmp_path / "manifest.json", manifest)
assert adapter.verify_wheelhouse(tmp_path)
wheel.write_bytes(b"different package")
with pytest.raises(ValueError, match="hash mismatch"):
adapter.verify_wheelhouse(tmp_path)
def test_wheelhouse_copy_normalizes_host_uid_and_excludes_unrelated_files(tmp_path):
name = "plugin-1.0-py3-none-any.whl"
(tmp_path / name).write_bytes(b"wheel")
for filename in ["constraints.txt", "requirements.lock", "unrelated"]:
(tmp_path / filename).write_text("data")
adapter.write_json(tmp_path / "manifest.json", {"wheels": [{"filename": name}]})
class Environment:
def copy_in_tar(self, path, destination):
assert destination == "/opt/programbench-evaluator-wheels"
with tarfile.open(path) as archive:
assert set(archive.getnames()) == {name, "manifest.json", "constraints.txt", "requirements.lock"}
for member in archive:
assert member.uid == member.gid == 0
assert member.isfile()
assert member.mode == 0o644
adapter.copy_evaluator_wheelhouse(Environment(), tmp_path)
def test_generated_upstream_plugin_copy_normalizes_uid_and_preserves_bytes(tmp_path):
source = tmp_path / "generated.py"
source.write_bytes(b"# pinned upstream plugin\n")
source.chmod(0o600)
class Environment:
def copy_in_tar(self, path, destination):
assert destination == "/opt/plugins"
with tarfile.open(path) as archive:
item = archive.getmember("programbench_pytest_timeout.py")
assert item.uid == item.gid == 0
assert item.mode == 0o600
assert archive.extractfile(item).read() == source.read_bytes()
adapter.copy_controller_artifact(Environment(), source, "/opt/plugins/programbench_pytest_timeout.py")
def test_controller_directory_copy_preserves_contents_modes_and_nested_symlinks(tmp_path):
source = tmp_path / "source"
source.mkdir(mode=0o750)
executable = source / "compile.sh"
executable.write_bytes(b"#!/bin/sh\nexit 0\n")
executable.chmod(0o755)
nested = source / "nested"
nested.mkdir()
(nested / "compile").symlink_to("../compile.sh")
class Environment:
def copy_in_tar(self, path, destination):
assert destination == "/workspace/solution"
with tarfile.open(path) as archive:
assert set(archive.getnames()) == {".", "./compile.sh", "./nested", "./nested/compile"}
assert archive.getmember(".").mode == 0o750
item = archive.getmember("./compile.sh")
assert item.mode == 0o755
assert archive.extractfile(item).read() == executable.read_bytes()
link = archive.getmember("./nested/compile")
assert link.issym() and link.linkname == "../compile.sh"
assert all(member.uid == member.gid == 0 for member in archive)
adapter.copy_controller_artifact(Environment(), source, "/workspace/solution")
def test_pinned_source_metadata_matches(tmp_path, manifest):
root = args(tmp_path).programbench_root
if not root.is_dir():
pytest.skip("Set PROGRAMBENCH_TEST_ROOT to the pinned official checkout")
adapter.check_source(root, manifest)
@pytest.mark.parametrize("name,link", [("../escape", None), ("/absolute", None),
("link", "/reference/executable"), ("nested/link", "../../escape")])
def test_rejects_submission_path_escapes(tmp_path, name, link):
path = tmp_path / "archive.tar.gz"
with tarfile.open(path, "w:gz") as archive:
item = tarfile.TarInfo(name)
if link:
item.type, item.linkname = tarfile.SYMTYPE, link
archive.addfile(item)
with pytest.raises(ValueError):
adapter.validate_archive(path)
def test_allows_normal_source_and_internal_symlink(tmp_path):
path = tmp_path / "archive.tar.gz"
with tarfile.open(path, "w:gz") as archive:
item = tarfile.TarInfo("src/main.py")
item.size = 10
archive.addfile(item, io.BytesIO(b"print(123)"))
item = tarfile.TarInfo("nested/code")
item.type, item.linkname = tarfile.SYMTYPE, "../src/main.py"
archive.addfile(item)
adapter.validate_archive(path)
def test_snapshot_archives_main_and_freezes_revision(tmp_path, manifest):
repo = tmp_path / "repo"
repo.mkdir()
adapter.run(["git", "init", "-q", "-b", "main", str(repo)])
adapter.run(["git", "-C", str(repo), "config", "user.name", "tester"])
adapter.run(["git", "-C", str(repo), "config", "user.email", "tester@example.invalid"])
(repo / "source.txt").write_text("canonical\n")
adapter.run(["git", "-C", str(repo), "add", "."])
adapter.run(["git", "-C", str(repo), "-c", "commit.gpgsign=false", "commit", "-qm", "main"])
adapter.run(["git", "-C", str(repo), "checkout", "-qb", "helper"])
(repo / "source.txt").write_text("uncommitted helper\n")
request = args(tmp_path / "snapshot")
request.instance_id = manifest["tasks"][0]["instance_id"]
request.container, request.git_dir, request.ref = "fake", str(repo / ".git"), "main"
real_run = subprocess.run
def fake_docker(command, **kwargs):
assert command[:2] == ["docker", "exec"]
return real_run(command[command.index("git"):], **kwargs)
with patch.object(adapter.subprocess, "run", side_effect=fake_docker):
result = adapter.snapshot(request, manifest)
with tarfile.open(result["submission_archive"]) as archive:
assert archive.extractfile("source.txt").read() == b"canonical\n"
assert result["sha256"] == adapter.sha256(Path(result["submission_archive"]))
assert not (Path(result["submission_archive"]).parent / "submission.tar.tmp").exists()
def test_calibration_failure_keeps_all_five_tasks(tmp_path, manifest):
request = args(tmp_path)
instances = [{"instance_id": item["instance_id"]} for item in manifest["tasks"]]
calls = []
def fake_evaluate(_args, task, *unused, **kwargs):
calls.append(task["instance_id"])
if len(calls) == 1:
raise RuntimeError("reference infrastructure failure")
return {"complete": True, "score": 1.0, "official_mask_sha256": "fixed"}
with patch.object(adapter, "evaluator_api", return_value={"load_all_instances": lambda: instances}), \
patch.object(adapter, "evaluate", side_effect=fake_evaluate):
result = adapter.calibrate(request, manifest)
assert result["status"] == "failed"
assert len(calls) == 10
assert len(result["tasks"]) == 5
assert result["tasks"][0]["passed"] is False
assert result["tasks"][0]["repetitions"][0]["error_code"] == "RuntimeError"
assert (tmp_path / "calibration.json").exists()
@pytest.mark.parametrize("change", ["failed", "missing-task", "wrong-manifest"])
def test_grade_gate_requires_complete_exact_manifest_calibration(tmp_path, manifest, change):
report = {"status": "passed", "manifest_sha256": adapter.sha256(adapter.DEFAULT_MANIFEST),
"tasks": [{"instance_id": item["instance_id"], "passed": True} for item in manifest["tasks"]]}
if change == "failed":
report["tasks"][0]["passed"] = False
elif change == "missing-task":
report["tasks"].pop()
else:
report["manifest_sha256"] = "different"
path = tmp_path / "calibration.json"
path.write_text(json.dumps(report))
with pytest.raises(ValueError):
adapter.check_calibration(path, adapter.DEFAULT_MANIFEST, manifest)
@pytest.mark.parametrize("not_run", [False, True])
def test_official_masks_fraction_and_incomplete_result(tmp_path, manifest, not_run):
api = official_api()
task = manifest["tasks"][0]
instance = {"image_name": "unused", "branches": {
"active": {"tests": ["passes", "fails", "excluded"], "ignored_tests": [{"name": "excluded"}]},
"ignored": {"tests": ["excluded_branch"], "ignored": True}}}
branch_dir = tmp_path / "blobs/tests"
branch_dir.mkdir(parents=True)
(branch_dir / "active.tar.gz").touch()
results = [api["TestResult"](name=name, branch=branch, status=status, extra={}) for name, branch, status in [
("passes", "active", "passed"), ("fails", "active", "not_run" if not_run else "failure"),
("excluded", "active", "passed"), ("excluded_branch", "ignored", "passed")]]
class FakeEvaluator:
def __init__(self, **kwargs):
self.kwargs = kwargs
def run(self):
assert self.kwargs["tests_branches"] == ["active"]
return api["EvaluationResult"](test_results=results, test_branches=["active", "ignored"])
api["Evaluator"] = FakeEvaluator
api["get_blob_dir"] = lambda iid: branch_dir.parent
with patch.object(adapter, "check_local_image"):
result = adapter.evaluate(args(tmp_path), task, instance, api, tmp_path / "result", reference=True)
assert result["score"] == 0.5
assert result["passed"] == 1
assert result["test_count"] == 2
assert result["expected_test_count"] == 2
assert result["complete"] is (not not_run)
assert json.loads((tmp_path / "result/official-mask.json").read_text()) == ["active/fails", "active/passes"]
def test_prepare_enforces_isolation_and_layout(tmp_path, manifest):
request = args(tmp_path)
request.instance_id = manifest["tasks"][0]["instance_id"]
request.episode_id, request.ttl_seconds = "one-single", 7200
calls = []
def fake_run(command, **kwargs):
calls.append(command)
value = ""
if command[1] == "run":
value = "container123\n"
if command[1] == "inspect":
value = json.dumps([{"HostConfig": {"NetworkMode": "none", "Binds": None}}])
return subprocess.CompletedProcess(command, 0, value, "")
with patch.object(adapter, "check_local_image"), patch.object(adapter, "run", side_effect=fake_run):
result = adapter.prepare(request, manifest)
start = calls[0]
assert start[start.index("--network") + 1] == "none"
assert start[start.index("--user") + 1] == "agent"
assert start[start.index("--pull") + 1] == "never"
assert "--volume" not in start and "--mount" not in start and "--env" not in start
assert result["seed_dir"] == "/workspace/solution"
assert result["canonical_git_dir"] == adapter.CANONICAL_GIT_DIR
assert "chown -R root:root /reference" in calls[1][-1]
assert "chmod -R a-w /reference" in calls[1][-1]
assert "original source" in Path(result["task_prompt_path"]).read_text()
|