tracinginsights's picture
download
raw
8.55 kB
#!/usr/bin/env python3
"""Run the Rp.py optimization campaign in isolated output directories.
The extraction scripts skip an existing ``*_tel.json`` file. This harness
therefore gives every run its own directory under a timestamped run directory,
creates the R.py-derived reference first, and compares every candidate against
that reference byte-for-byte.
Example:
.venv/bin/python experiments/benchmark.py --output-root /tmp/rp-benchmark
The benchmark is intentionally opt-in: it can take several minutes and needs
the warm FastF1 ``cache/`` plus the dependencies in ``requirements.txt``.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
from pathlib import Path
import shutil
import subprocess
import sys
import tempfile
import time
HERE = Path(__file__).resolve().parent
REPO = HERE.parent
DEFAULT_VARIANTS = (
"run_seq.py",
"run_par.py",
"exp5_saturate.py",
"exp6_numpy_drvhead.py",
"exp7_numpy_merge.py",
"Rf.py",
)
EXPLORATORY_VARIANTS = (
"exp2_merge_once.py",
"exp3_workers.py",
"exp4_fastslice.py",
)
def _json_files(root: Path) -> dict[str, Path]:
return {
str(path.relative_to(root)): path
for path in root.rglob("*.json")
if path.is_file()
}
def _tree_digest(files: dict[str, Path]) -> str:
digest = hashlib.sha256()
for name in sorted(files):
digest.update(name.encode("utf-8"))
digest.update(b"\0")
digest.update(hashlib.sha256(files[name].read_bytes()).digest())
return digest.hexdigest()
def _compare(reference: Path, candidate: Path) -> tuple[bool, dict[str, int | str]]:
ref_files = _json_files(reference)
candidate_files = _json_files(candidate)
names = sorted(set(ref_files) | set(candidate_files))
differing = 0
missing = 0
for name in names:
left = ref_files.get(name)
right = candidate_files.get(name)
if left is None or right is None:
missing += 1
elif left.read_bytes() != right.read_bytes():
differing += 1
return (
missing == 0 and differing == 0,
{
"json_files": len(candidate_files),
"telemetry_files": sum(name.endswith("_tel.json") for name in candidate_files),
"missing_json_files": missing,
"differing_json_files": differing,
"tree_sha256": _tree_digest(candidate_files),
},
)
def _assert_generated_artifacts_current() -> None:
"""Ensure checked-in candidates still derive from current source scripts."""
generated = (
"run_seq.py",
"run_par.py",
"exp2_merge_once.py",
"exp3_workers.py",
"exp4_fastslice.py",
"exp5_saturate.py",
"exp6_numpy_drvhead.py",
"exp7_numpy_merge.py",
"Rf.py",
)
with tempfile.TemporaryDirectory(prefix="rp-builder-check-") as temp:
root = Path(temp)
shutil.copy2(REPO / "R.py", root / "R.py")
shutil.copy2(REPO / "Rp.py", root / "Rp.py")
shutil.copytree(HERE, root / "experiments")
for command in ("builder.py", "make_exp6.py", "make_exp7.py"):
subprocess.run(
[sys.executable, f"experiments/{command}"],
cwd=root,
check=True,
stdout=subprocess.DEVNULL,
)
for name in generated:
if name == "Rf.py":
generated_path = root / name
actual_path = REPO / name
else:
generated_path = root / "experiments" / name
actual_path = HERE / name
expected = generated_path.read_bytes()
actual = actual_path.read_bytes()
if expected != actual:
raise SystemExit(
f"{name} is stale; regenerate the experiments before benchmarking"
)
def _run_variant(script: Path, output_dir: Path, timeout: int, workers: str | None) -> float:
env = os.environ.copy()
env["EXP_ROOT"] = str(output_dir)
env.setdefault("no_proxy", "*")
if workers:
env["TEL_MAX_WORKERS"] = workers
started = time.perf_counter()
subprocess.run(
[sys.executable, str(script)],
cwd=REPO,
env=env,
check=True,
timeout=timeout,
)
return time.perf_counter() - started
def _parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--output-root",
type=Path,
required=True,
help="Parent directory for this run's fresh per-variant output trees.",
)
parser.add_argument(
"--variant",
dest="variants",
action="append",
help="Experiment filename; repeat to select a subset (default: accepted candidates).",
)
parser.add_argument(
"--include-exploratory",
action="store_true",
help="Also run exp2/exp3/exp4, which are intentionally non-identical.",
)
parser.add_argument(
"--timeout",
type=int,
default=1800,
help="Per-variant timeout in seconds (default: 1800).",
)
parser.add_argument(
"--workers",
help="Optional TEL_MAX_WORKERS override applied to every variant.",
)
return parser.parse_args()
def main() -> int:
args = _parse_args()
if args.variants:
variants = tuple(args.variants)
else:
variants = DEFAULT_VARIANTS + (
EXPLORATORY_VARIANTS if args.include_exploratory else ()
)
scripts = [HERE / name for name in variants]
missing = [str(path) for path in scripts if not path.is_file()]
if missing:
raise SystemExit("Unknown variant file(s): " + ", ".join(missing))
_assert_generated_artifacts_current()
output_root = args.output_root.expanduser().resolve()
output_root.mkdir(parents=True, exist_ok=True)
run_root = output_root / time.strftime("run-%Y%m%d-%H%M%S")
suffix = 0
while run_root.exists():
suffix += 1
run_root = output_root / f"{time.strftime('%Y%m%d-%H%M%S')}-{suffix}"
run_root.mkdir()
results: list[dict[str, object]] = []
reference_dir = run_root / "run_seq.py"
print(f"Running reference: {reference_dir}", flush=True)
reference_seconds = _run_variant(
HERE / "run_seq.py", reference_dir, args.timeout, args.workers
)
reference_files = _json_files(reference_dir)
if not reference_files:
raise SystemExit(
"R.py reference produced no JSON files; refusing a vacuous identity pass"
)
reference_digest = _tree_digest(reference_files)
results.append(
{
"file": "run_seq.py",
"role": "R.py reference",
"seconds": round(reference_seconds, 3),
"byte_identical": True,
"json_files": len(reference_files),
"telemetry_files": sum(
name.endswith("_tel.json") for name in reference_files
),
"tree_sha256": reference_digest,
"output_dir": str(reference_dir),
}
)
for name in variants:
if name == "run_seq.py":
continue
output_dir = run_root / Path(name).stem
print(f"Running {name}: {output_dir}", flush=True)
seconds = _run_variant(HERE / name, output_dir, args.timeout, args.workers)
identical, details = _compare(reference_dir, output_dir)
row: dict[str, object] = {
"file": name,
"role": "candidate",
"seconds": round(seconds, 3),
"byte_identical": identical,
"output_dir": str(output_dir),
}
row.update(details)
results.append(row)
print(
f" {seconds:.3f}s, {details['json_files']} JSON / "
f"{details['telemetry_files']} telemetry, byte-identical={identical}",
flush=True,
)
report = {
"title": "Rp.py optimization campaign benchmark",
"reference": "run_seq.py (adapter generated from R.py)",
"output_root": str(run_root),
"python": sys.executable,
"workers": args.workers,
"results": results,
}
report_path = run_root / "benchmark.json"
report_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
print(f"Wrote {report_path}")
return 0 if all(bool(row["byte_identical"]) for row in results) else 1
if __name__ == "__main__":
raise SystemExit(main())

Xet Storage Details

Size:
8.55 kB
·
Xet hash:
2bc2e4e5052be61ccd0e8780e0e2ebe410465b565da59e0c5f7cbf87558a45ad

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.