agent-harness / scripts /audit_study3_design.py
cuber12's picture
Publish agent harness research code and paper artifacts
d61821a verified
Raw
History Blame Contribute Delete
8.95 kB
"""Outcome-blind deterministic audit for the prospective E09 design freeze."""
from __future__ import annotations
from collections import Counter, defaultdict
from hashlib import sha256
import json
from pathlib import Path
import subprocess
from typing import Any
from agent_harness.specs import (
load_edit_interfaces,
load_experiments,
load_models,
load_repositories,
load_task_split,
load_tasks,
validate_configuration_tree,
)
def digest(path: Path) -> str:
return sha256(path.read_bytes()).hexdigest()
def canonical_digest(value: Any) -> str:
return sha256(
json.dumps(value, sort_keys=True, separators=(",", ":")).encode()
).hexdigest()
def run(root: Path) -> dict[str, Any]:
errors, warnings = validate_configuration_tree(root)
if errors or warnings:
raise RuntimeError(f"configuration failed: errors={errors}, warnings={warnings}")
raw = root / "results" / "raw" / "E09"
if raw.exists() and any(raw.rglob("*")):
raise RuntimeError("E09 outcome directory already exists; design is no longer outcome-blind")
experiment = load_experiments(root)["E09"]
interfaces = load_edit_interfaces(root)
models = load_models(root)
repositories = load_repositories(root)
tasks = load_tasks(root)
split = load_task_split(root / "tasks" / "splits" / "study3_protocol.txt")
study2_split = load_task_split(
root / "tasks" / "splits" / "study2_confirmatory.txt"
)
if split != study2_split or len(split) != 60:
raise RuntimeError("E09 split must exactly reuse the 60 ordered E08 tasks")
if experiment.cells_per_task() != 9:
raise RuntimeError("E09 must define nine model-interface cells per task")
url_to_repository = {
item.repository_url: item.repository_id for item in repositories.values()
}
repository_counts: Counter[str] = Counter()
language_counts: Counter[str] = Counter()
input_files: list[Path] = [
root / "configs" / "experiments" / "E09_protocol_interface.toml",
root / "tasks" / "splits" / "study3_protocol.txt",
root / "docs" / "STUDY3_PREREGISTRATION.md",
root / "docs" / "STUDY3_IMPLEMENTATION.md",
root / "docs" / "PROTOCOL_AMENDMENTS.md",
root / "scripts" / "analyze_study3.py",
root / "scripts" / "audit_study3_design.py",
root / "scripts" / "preflight_study3.py",
root / "src" / "agent_harness" / "cli.py",
root / "src" / "agent_harness" / "lm_studio.py",
root / "src" / "agent_harness" / "protocol_experiment.py",
root / "src" / "agent_harness" / "specs.py",
root / "src" / "agent_harness" / "study2_experiment.py",
root / "tests" / "test_protocol_experiment.py",
root / "tests" / "test_specs.py",
root / "tests" / "test_study3_analysis.py",
]
input_files.extend(
sorted((root / "configs" / "repositories").glob("R*.toml"))
)
task_records: list[dict[str, Any]] = []
for task_id in split:
task = tasks[task_id]
if task.validation_status != "end_to_end_ready":
raise RuntimeError(f"{task_id} is not end-to-end ready")
repository_id = url_to_repository.get(task.repository_url)
if repository_id is None:
raise RuntimeError(f"{task_id} repository is outside the registry")
repository_counts[repository_id] += 1
language_counts[task.language] += 1
manifest = root / "tasks" / "manifests" / f"{task_id}.toml"
source_patch = root / "tasks" / task.gold_patch
test_patch = root / "tasks" / task.test_patch
validation = root / "tasks" / "validation" / "study2" / f"{task_id}.json"
if not validation.is_file():
validation = root / "tasks" / "validation" / f"{task_id}.json"
required = (manifest, source_patch, test_patch, validation)
if not all(path.is_file() for path in required):
raise RuntimeError(f"{task_id} is missing a frozen task artifact")
input_files.extend(required)
task_records.append(
{
"task_id": task_id,
"config_hash": task.config_hash,
"repository_id": repository_id,
"language": task.language,
"base_commit": task.base_commit,
"gold_commit": task.gold_commit,
"gold_files": list(task.gold_files),
"artifact_sha256": {
path.relative_to(root).as_posix(): digest(path) for path in required
},
}
)
if repository_counts != Counter({"R001": 20, "R002": 20, "R003": 20}):
raise RuntimeError(f"repository balance drift: {repository_counts}")
for interface_id in experiment.edit_interface_ids:
input_files.append(
root
/ "configs"
/ "edit_interfaces"
/ {
"P001": "P001_unified_diff.toml",
"P002": "P002_exact_replace.toml",
"P003": "P003_whole_file.toml",
}[interface_id]
)
for model_id in experiment.model_ids:
matching = sorted((root / "configs" / "models").glob(f"{model_id.lower()}_*.toml"))
if len(matching) != 1:
raise RuntimeError(f"no unique model config for {model_id}: {matching}")
input_files.append(matching[0])
model_positions: Counter[tuple[str, int]] = Counter()
interface_positions: Counter[tuple[str, str, int]] = Counter()
identities: set[tuple[str, str, str]] = set()
model_ids = list(experiment.model_ids)
interface_ids = list(experiment.edit_interface_ids)
for task_index, task_id in enumerate(split):
model_offset = task_index % len(model_ids)
model_order = model_ids[model_offset:] + model_ids[:model_offset]
for model_position, model_id in enumerate(model_order):
model_positions[(model_id, model_position)] += 1
canonical_model_index = model_ids.index(model_id)
offset = (task_index + canonical_model_index) % len(interface_ids)
interface_order = interface_ids[offset:] + interface_ids[:offset]
for interface_position, interface_id in enumerate(interface_order):
interface_positions[(model_id, interface_id, interface_position)] += 1
identities.add((task_id, model_id, interface_id))
if len(identities) != 540:
raise RuntimeError(f"planned E09 identities are not unique: {len(identities)}")
if set(model_positions.values()) != {20} or set(interface_positions.values()) != {20}:
raise RuntimeError("counterbalancing does not place every factor level 20 times per position")
unique_inputs = sorted(set(input_files))
input_hashes = {
path.relative_to(root).as_posix(): digest(path) for path in unique_inputs
}
revision = subprocess.run(
["git", "rev-parse", "HEAD"], cwd=root, check=True,
capture_output=True, text=True,
).stdout.strip()
report = {
"schema_version": 1,
"study": "E09 protocol-normalized edit-interface compatibility",
"outcome_blind": True,
"prior_revision": revision,
"raw_e09_absent": True,
"tasks": len(split),
"models": list(experiment.model_ids),
"interfaces": list(experiment.edit_interface_ids),
"planned_cells": len(identities),
"repository_counts": dict(sorted(repository_counts.items())),
"language_counts": dict(sorted(language_counts.items())),
"model_position_counts": {
f"{model_id}@{position}": count
for (model_id, position), count in sorted(model_positions.items())
},
"interface_position_counts": {
f"{model_id}/{interface_id}@{position}": count
for (model_id, interface_id, position), count in sorted(interface_positions.items())
},
"model_config_hashes": {
item: models[item].config_hash for item in experiment.model_ids
},
"interface_config_hashes": {
item: interfaces[item].config_hash for item in experiment.edit_interface_ids
},
"task_records": task_records,
"input_sha256": input_hashes,
}
report["design_sha256"] = canonical_digest(report)
return report
def main() -> None:
root = Path(__file__).resolve().parents[1]
report = run(root)
output = root / "docs" / "STUDY3_DESIGN_AUDIT.json"
output.write_text(
json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
print(
json.dumps(
{
"output": str(output),
"planned_cells": report["planned_cells"],
"design_sha256": report["design_sha256"],
},
indent=2,
sort_keys=True,
)
)
if __name__ == "__main__":
main()