agent-harness / scripts /preflight_study4.py
cuber12's picture
Publish agent harness research code and paper artifacts
d61821a verified
Raw
History Blame Contribute Delete
7.37 kB
"""Outcome-blind LM Studio, repository, and executor preflight for E10."""
from __future__ import annotations
from dataclasses import asdict
import importlib.metadata
import json
from pathlib import Path
import platform
import shutil
import sys
import time
from typing import Any
from preflight_study3 import PACKAGES, _command, _executor_conformance, _tool_probe
from agent_harness.lm_studio import LMStudioClient
from agent_harness.lm_studio_embeddings import LMStudioEmbeddingClient
from agent_harness.lm_studio_management import LMStudioResidencyManager, LMStudioServer
from agent_harness.pilot import research_code_revision
from agent_harness.protocol_experiment import protocol_tool_definitions
from agent_harness.specs import (
load_edit_interfaces,
load_embeddings,
load_experiments,
load_harnesses,
load_models,
load_repositories,
load_task_split,
load_tasks,
validate_configuration_tree,
)
from agent_harness.study2_experiment import tokenizer_for
def run(root: Path) -> dict[str, Any]:
revision = research_code_revision(root)
errors, warnings = validate_configuration_tree(root)
if errors or warnings:
raise RuntimeError(f"configuration failed: errors={errors}, warnings={warnings}")
raw = root / "results" / "raw" / "E10"
if raw.exists() and any(raw.rglob("*")):
raise RuntimeError("E10 raw directory exists before outcome-blind preflight")
audit = json.loads((root / "docs" / "STUDY4_DESIGN_AUDIT.json").read_text())
if audit.get("planned_cells") != 180 or not audit.get("outcome_blind"):
raise RuntimeError("Study 4 design audit is not a valid 180-cell freeze")
experiment = load_experiments(root)["E10"]
split = load_task_split(root / "tasks" / "splits" / "study4_fresh.txt")
if len(split) * len(experiment.model_ids) * len(experiment.harness_ids) != 180:
raise RuntimeError("E10 execution plan is not exactly 180 gated cells")
disk = shutil.disk_usage(root)
if disk.free < 12 * 1024**3:
raise RuntimeError(f"less than 12 GiB free before Study 4: {disk.free} bytes")
models = load_models(root)
embedding = load_embeddings(root)[experiment.embedding_id]
repositories = load_repositories(root)
tasks = load_tasks(root)
harnesses = load_harnesses(root)
interfaces = load_edit_interfaces(root)
gate = json.loads((root / "configs/gates/E09_model_interface_gate.json").read_text())["selected"]
first_task = tasks[split[0]]
tool_signatures = {}
for model_id, interface_id in gate.items():
tool_signatures[model_id] = {}
for harness_id in experiment.harness_ids:
tool_signatures[model_id][harness_id] = [
item["function"]["name"]
for item in protocol_tool_definitions(
interfaces[interface_id], first_task, harnesses[harness_id]
)
]
if any(
values["H000"] != values["H007"] for values in tool_signatures.values()
):
raise RuntimeError("H000/H007 tool signatures are not identical")
server = LMStudioServer(port=1234)
server_state = server.ensure_running()
residency = LMStudioResidencyManager(
models["M002"].base_url, models["M002"].api_token_env, timeout_seconds=1_800
)
report: dict[str, Any] = {
"schema_version": 1, "study": "Study 4 / E10", "outcome_blind": True,
"research_code_revision": revision, "design_sha256": audit["design_sha256"],
"planned_cells": 180, "started_unix": time.time(), "platform": platform.platform(),
"python": sys.version, "torch_used": False,
"disk": {"total": disk.total, "used": disk.used, "free": disk.free},
"lms_version": _command([str(server.cli_path), "--version"]),
"server_start": server_state,
"dependencies": {package: importlib.metadata.version(package) for package in PACKAGES},
"executor_conformance": _executor_conformance(root),
"tool_signatures": tool_signatures,
"repositories": {}, "embedding": {}, "models": {},
}
try:
residency.unload_all()
for repository_id, repository in sorted(repositories.items()):
observed = _command(["git", "rev-parse", "HEAD"], cwd=root / repository.local_path)
if observed["returncode"] or observed["stdout"].strip() != repository.pinned_head:
raise RuntimeError(f"repository head mismatch: {repository_id}: {observed}")
report["repositories"][repository_id] = {
"spec": asdict(repository), "observed_head": observed["stdout"].strip()
}
embedding_transition = residency.ensure_exclusive(
embedding.model_key, embedding.loaded_context_length
)
embedding_client = LMStudioEmbeddingClient(embedding, timeout_seconds=1_800)
report["embedding"] = {
"spec": asdict(embedding), "config_hash": embedding.config_hash,
"transition": embedding_transition.to_dict(),
"resolved": embedding_client.resolve(), "probe": embedding_client.probe().to_dict(),
"unload": residency.unload_all().to_dict(),
}
for model_id in experiment.model_ids:
model = models[model_id]
transition = residency.ensure_exclusive(model.expected_inference_key, model.context_length)
client = LMStudioClient(model, timeout_seconds=1_800)
discovery, resolved = client.resolve()
tokenizer = tokenizer_for(model)
report["models"][model_id] = {
"spec": asdict(model), "config_hash": model.config_hash,
"gate_interface": gate[model_id], "transition": transition.to_dict(),
"resolved": resolved.to_dict(), "discovery_errors": discovery.endpoint_errors,
"tokenizer_path": str(tokenizer.path), "tokenizer_sha256": tokenizer.sha256,
"tool_probe": _tool_probe(client, resolved.inference_key),
"unload": residency.unload_all().to_dict(),
}
report["passed"] = True
return report
finally:
cleanup_errors = []
try:
report["final_unload"] = residency.unload_all().to_dict()
except Exception as exc:
cleanup_errors.append(f"unload_all: {exc}")
try:
status = server.status()
report["server_stop"] = server.stop() if status["running"] else {"action": "already_stopped", "status": status}
except Exception as exc:
cleanup_errors.append(f"server_stop: {exc}")
report["cleanup_errors"] = cleanup_errors
report["finished_unix"] = time.time()
def main() -> None:
root = Path(__file__).resolve().parents[1]
output = root / "results" / "reports" / "study4_preflight.json"
output.parent.mkdir(parents=True, exist_ok=True)
report: dict[str, Any] = {}
try:
report = run(root)
except Exception as exc:
report = {**report, "passed": False, "error": repr(exc)}
output.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n")
raise
output.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n")
print(json.dumps({"passed": True, "report": str(output)}, indent=2))
if __name__ == "__main__":
main()