| """Outcome-blind LM Studio, repository, and executor preflight for E10.""" |
|
|
| from __future__ import annotations |
|
|
| from dataclasses import asdict |
| import importlib.metadata |
| import json |
| from pathlib import Path |
| import platform |
| import shutil |
| import sys |
| import time |
| from typing import Any |
|
|
| from preflight_study3 import PACKAGES, _command, _executor_conformance, _tool_probe |
| from agent_harness.lm_studio import LMStudioClient |
| from agent_harness.lm_studio_embeddings import LMStudioEmbeddingClient |
| from agent_harness.lm_studio_management import LMStudioResidencyManager, LMStudioServer |
| from agent_harness.pilot import research_code_revision |
| from agent_harness.protocol_experiment import protocol_tool_definitions |
| from agent_harness.specs import ( |
| load_edit_interfaces, |
| load_embeddings, |
| load_experiments, |
| load_harnesses, |
| load_models, |
| load_repositories, |
| load_task_split, |
| load_tasks, |
| validate_configuration_tree, |
| ) |
| from agent_harness.study2_experiment import tokenizer_for |
|
|
|
|
| def run(root: Path) -> dict[str, Any]: |
| revision = research_code_revision(root) |
| errors, warnings = validate_configuration_tree(root) |
| if errors or warnings: |
| raise RuntimeError(f"configuration failed: errors={errors}, warnings={warnings}") |
| raw = root / "results" / "raw" / "E10" |
| if raw.exists() and any(raw.rglob("*")): |
| raise RuntimeError("E10 raw directory exists before outcome-blind preflight") |
| audit = json.loads((root / "docs" / "STUDY4_DESIGN_AUDIT.json").read_text()) |
| if audit.get("planned_cells") != 180 or not audit.get("outcome_blind"): |
| raise RuntimeError("Study 4 design audit is not a valid 180-cell freeze") |
| experiment = load_experiments(root)["E10"] |
| split = load_task_split(root / "tasks" / "splits" / "study4_fresh.txt") |
| if len(split) * len(experiment.model_ids) * len(experiment.harness_ids) != 180: |
| raise RuntimeError("E10 execution plan is not exactly 180 gated cells") |
|
|
| disk = shutil.disk_usage(root) |
| if disk.free < 12 * 1024**3: |
| raise RuntimeError(f"less than 12 GiB free before Study 4: {disk.free} bytes") |
| models = load_models(root) |
| embedding = load_embeddings(root)[experiment.embedding_id] |
| repositories = load_repositories(root) |
| tasks = load_tasks(root) |
| harnesses = load_harnesses(root) |
| interfaces = load_edit_interfaces(root) |
| gate = json.loads((root / "configs/gates/E09_model_interface_gate.json").read_text())["selected"] |
| first_task = tasks[split[0]] |
| tool_signatures = {} |
| for model_id, interface_id in gate.items(): |
| tool_signatures[model_id] = {} |
| for harness_id in experiment.harness_ids: |
| tool_signatures[model_id][harness_id] = [ |
| item["function"]["name"] |
| for item in protocol_tool_definitions( |
| interfaces[interface_id], first_task, harnesses[harness_id] |
| ) |
| ] |
| if any( |
| values["H000"] != values["H007"] for values in tool_signatures.values() |
| ): |
| raise RuntimeError("H000/H007 tool signatures are not identical") |
|
|
| server = LMStudioServer(port=1234) |
| server_state = server.ensure_running() |
| residency = LMStudioResidencyManager( |
| models["M002"].base_url, models["M002"].api_token_env, timeout_seconds=1_800 |
| ) |
| report: dict[str, Any] = { |
| "schema_version": 1, "study": "Study 4 / E10", "outcome_blind": True, |
| "research_code_revision": revision, "design_sha256": audit["design_sha256"], |
| "planned_cells": 180, "started_unix": time.time(), "platform": platform.platform(), |
| "python": sys.version, "torch_used": False, |
| "disk": {"total": disk.total, "used": disk.used, "free": disk.free}, |
| "lms_version": _command([str(server.cli_path), "--version"]), |
| "server_start": server_state, |
| "dependencies": {package: importlib.metadata.version(package) for package in PACKAGES}, |
| "executor_conformance": _executor_conformance(root), |
| "tool_signatures": tool_signatures, |
| "repositories": {}, "embedding": {}, "models": {}, |
| } |
| try: |
| residency.unload_all() |
| for repository_id, repository in sorted(repositories.items()): |
| observed = _command(["git", "rev-parse", "HEAD"], cwd=root / repository.local_path) |
| if observed["returncode"] or observed["stdout"].strip() != repository.pinned_head: |
| raise RuntimeError(f"repository head mismatch: {repository_id}: {observed}") |
| report["repositories"][repository_id] = { |
| "spec": asdict(repository), "observed_head": observed["stdout"].strip() |
| } |
|
|
| embedding_transition = residency.ensure_exclusive( |
| embedding.model_key, embedding.loaded_context_length |
| ) |
| embedding_client = LMStudioEmbeddingClient(embedding, timeout_seconds=1_800) |
| report["embedding"] = { |
| "spec": asdict(embedding), "config_hash": embedding.config_hash, |
| "transition": embedding_transition.to_dict(), |
| "resolved": embedding_client.resolve(), "probe": embedding_client.probe().to_dict(), |
| "unload": residency.unload_all().to_dict(), |
| } |
| for model_id in experiment.model_ids: |
| model = models[model_id] |
| transition = residency.ensure_exclusive(model.expected_inference_key, model.context_length) |
| client = LMStudioClient(model, timeout_seconds=1_800) |
| discovery, resolved = client.resolve() |
| tokenizer = tokenizer_for(model) |
| report["models"][model_id] = { |
| "spec": asdict(model), "config_hash": model.config_hash, |
| "gate_interface": gate[model_id], "transition": transition.to_dict(), |
| "resolved": resolved.to_dict(), "discovery_errors": discovery.endpoint_errors, |
| "tokenizer_path": str(tokenizer.path), "tokenizer_sha256": tokenizer.sha256, |
| "tool_probe": _tool_probe(client, resolved.inference_key), |
| "unload": residency.unload_all().to_dict(), |
| } |
| report["passed"] = True |
| return report |
| finally: |
| cleanup_errors = [] |
| try: |
| report["final_unload"] = residency.unload_all().to_dict() |
| except Exception as exc: |
| cleanup_errors.append(f"unload_all: {exc}") |
| try: |
| status = server.status() |
| report["server_stop"] = server.stop() if status["running"] else {"action": "already_stopped", "status": status} |
| except Exception as exc: |
| cleanup_errors.append(f"server_stop: {exc}") |
| report["cleanup_errors"] = cleanup_errors |
| report["finished_unix"] = time.time() |
|
|
|
|
| def main() -> None: |
| root = Path(__file__).resolve().parents[1] |
| output = root / "results" / "reports" / "study4_preflight.json" |
| output.parent.mkdir(parents=True, exist_ok=True) |
| report: dict[str, Any] = {} |
| try: |
| report = run(root) |
| except Exception as exc: |
| report = {**report, "passed": False, "error": repr(exc)} |
| output.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n") |
| raise |
| output.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n") |
| print(json.dumps({"passed": True, "report": str(output)}, indent=2)) |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|