"""Fail-closed outcome-independent audit of frozen E11/E12 manifests.""" from __future__ import annotations from hashlib import sha256 import json from pathlib import Path from agent_harness.specs import load_task_split, load_tasks SALT = "study4-ancillary-20260719" def digest(path: Path) -> str: return sha256(path.read_bytes()).hexdigest() def run(root: Path) -> dict[str, object]: for experiment_id in ("E11", "E12"): raw = root / "results" / "raw" / experiment_id if raw.exists() and any(raw.rglob("*")): raise RuntimeError(f"{experiment_id} already contains outcome data") split = load_task_split(root / "tasks/splits/study4_fresh.txt") tasks = load_tasks(root) selected: dict[str, str] = {} for repository_id in ("R001", "R002", "R003"): candidates = [task_id for task_id in split if f"_{repository_id}_" in task_id] selected[repository_id] = min( candidates, key=lambda task_id: sha256(f"{SALT}:{task_id}".encode()).hexdigest(), ) expected_tasks = set(selected.values()) e11_path = root / "configs/reliability/E11_repeat_cells.json" e12_path = root / "configs/context/E12_context_cells.json" e11 = json.loads(e11_path.read_text()) e12 = json.loads(e12_path.read_text()) if {row["task_id"] for row in e11["cells"]} != expected_tasks: raise RuntimeError("E11 tasks do not match the outcome-independent selector") if {row["task_id"] for row in e12["cells"]} != expected_tasks: raise RuntimeError("E12 tasks do not match the outcome-independent selector") if len(e11["cells"]) != 6 or e11["seeds"] != [0, 1, 2]: raise RuntimeError("E11 must freeze six groups and three seeds") if len(e12["cells"]) != 12 or e12["seeds"] != [0]: raise RuntimeError("E12 must freeze twelve context cells") if {row["model_id"] for row in e11["cells"]} != {"M002", "M003", "M004"}: raise RuntimeError("E11 model balance drifted") for field, values in ( ("model_id", ("M002", "M003", "M004")), ("harness_id", ("H000", "H007")), ): counts = {value: sum(row[field] == value for row in e11["cells"]) for value in values} if len(set(counts.values())) != 1: raise RuntimeError(f"E11 {field} is not balanced: {counts}") for task_id in expected_tasks: if tasks[task_id].validation_status != "end_to_end_ready": raise RuntimeError(f"selected task is not executable: {task_id}") report = { "schema_version": 1, "outcome_independent": True, "selection_salt": SALT, "selected_tasks": selected, "planned_responses": {"E11": 18, "E12": 12}, "manifest_sha256": {"E11": digest(e11_path), "E12": digest(e12_path)}, } output = root / "docs/STUDY4_ANCILLARY_AUDIT.json" output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") return report if __name__ == "__main__": root = Path(__file__).resolve().parents[1] print(json.dumps(run(root), indent=2, sort_keys=True))