"""Build a self-contained evaluation package that runs anywhere. The problem this solves: benchmark numbers from different machines are only comparable if the machines ran the same code over the same images with the same models. Describing that in a README and hoping is not enough -- someone will have a different dataset slice, or a stale export, and the comparison quietly stops meaning anything. So the package carries everything: the three model formats, a fixed set of frames as real files, the calibration and priors those frames need, the depth and benchmark code, and three shell scripts that each print one table. python scripts/make_evalpack.py --frames 100 --tar Ship the tarball, run setup.sh once, then run_all.sh. Notes on what goes in: * Images are **copied**, not symlinked. The whole repo uses symlinks into the dataset roots to avoid duplicating 8 GB, which is right at home and useless in something meant to be sent somewhere else. * Frames are taken in sorted order, not sampled randomly, so two builds of the same size contain the same frames and two machines can be compared directly. * Only the calibrations those frames actually use are copied -- otherwise the pack carries fifteen Argoverse 2 calibrations that nothing references. """ from __future__ import annotations import argparse import shutil import subprocess from pathlib import Path import pandas as pd from src.common import paths, schema CODE = ["src/__init__.py", "src/common", "src/depth", "src/bench"] REQUIREMENTS = """\ # Pinned to the versions the comparison was validated against. Ultralytics in # particular changes training and export defaults between minor releases. ultralytics<=8.3.40 onnxruntime>=1.17 ncnn numpy>=1.24 pandas>=2.0 pyarrow>=14.0 pyyaml>=6.0 """ SETUP = """\ #!/usr/bin/env bash # Create a virtual environment and install the pinned dependencies. # Run once per machine. Safe to re-run. # # Everything is logged to setup.log and echoed to the terminal, and pip is NOT # run quietly. On a Pi this pulls about 2 GB onto an SD card over several # minutes, and silence for that long is indistinguishable from a hang. set -euo pipefail cd "$(dirname "$0")" LOG=setup.log : > "$LOG" exec > >(tee -a "$LOG") 2>&1 START=$(date +%s) step() { echo; echo "[$(date +%H:%M:%S) +$(($(date +%s) - START))s] $*"; } PYTHON=${PYTHON:-python3} step "checking prerequisites" echo " python : $("$PYTHON" -V 2>&1) at $(command -v "$PYTHON")" echo " arch : $(uname -m)" df -h . | awk 'NR==2 {print " disk : " $4 " free"}' if command -v uv >/dev/null 2>&1; then step "creating .venv with uv" [ -d .venv ] || uv venv .venv step "installing dependencies with uv" VIRTUAL_ENV="$PWD/.venv" uv pip install -r requirements.txt else echo " uv : not installed (pip will be used; uv is far faster)" # Only pip's path needs this. uv builds a venv without ensurepip, so the # check belongs here rather than up front. The classic Raspberry Pi OS # failure is python3 present but python3-venv absent, which makes # `python3 -m venv` sit for a while and then fail confusingly. if ! "$PYTHON" -c "import ensurepip" 2>/dev/null; then echo echo " !! $PYTHON cannot import ensurepip, so it cannot build a venv." echo " On Raspberry Pi OS or Debian:" echo " sudo apt update && sudo apt install -y python3-venv python3-pip" echo echo " Or install uv, which does not need it:" echo " curl -LsSf https://astral.sh/uv/install.sh | sh" exit 1 fi echo " ensurepip: OK" if [ -d .venv ]; then step ".venv already exists, reusing it" else step "creating .venv with $PYTHON" echo " bootstraps pip into the new environment; on an SD card this can" echo " take a few minutes with no output. Watch it from another shell:" echo " watch -n2 du -sh $PWD/.venv" "$PYTHON" -m venv .venv fi step "upgrading pip" ./.venv/bin/python -m pip install --upgrade pip step "installing dependencies (~2 GB, several minutes on a Pi)" # Deliberately not --quiet: progress here is the difference between # "working" and "hung". ./.venv/bin/python -m pip install --progress-bar on -r requirements.txt fi step "verifying" ./.venv/bin/python - <<'PY' import platform print(f" python {platform.python_version()} {platform.machine()}") missing = [] for name in ("torch", "ultralytics", "onnxruntime", "numpy", "cv2", "ncnn"): try: mod = __import__(name) print(f" {name:12s} {getattr(mod, '__version__', 'installed')}") except ImportError: print(f" {name:12s} NOT INSTALLED") missing.append(name) required = [m for m in missing if m != "ncnn"] if required: print() print(f" missing and required: {', '.join(required)}") elif "ncnn" in missing: print() print(" ncnn is absent. PyTorch and ONNX still run; see PI_RUNBOOK.md step 5.") PY step "done in $(($(date +%s) - START))s, logged to $LOG" echo echo "next: ./run_onnx.sh (the format known to work)" echo " ./run_all.sh (all three, then the comparison)" """ RUNNER = """\ #!/usr/bin/env bash # Benchmark the {label} model and print the timing table. # # THREADS controls the CPU budget. Pin it to the same value on every machine # being compared -- an unpinned run measures core count as much as format. # A Pi 5 has 4 cores, so 4 is the default. set -euo pipefail cd "$(dirname "$0")" THREADS=${{THREADS:-4}} FRAMES=${{FRAMES:-{frames}}} export UNIFIED_ROOT=data exec ./.venv/bin/python -m src.bench.eval_{tag} \\ --weights "{weights}" \\ --unified-root data \\ --limit "$FRAMES" \\ --threads "$THREADS" \\ --out-dir results \\ "$@" """ RUN_ALL = """\ #!/usr/bin/env bash # Run all three formats and print the comparison. # # NCNN is allowed to fail without taking the run down: on x86 a pnnx/ncnn # version mismatch makes it segfault, and the other two numbers are still worth # having. On the Pi it should succeed, and that is the number that matters. set -uo pipefail cd "$(dirname "$0")" ./run_pt.sh "$@" || echo "!! PyTorch run failed" ./run_onnx.sh "$@" || echo "!! ONNX run failed" ./run_ncnn.sh "$@" || echo "!! NCNN run failed (expected on x86 -- see README)" echo export UNIFIED_ROOT=data ./.venv/bin/python -m src.bench.compare --bench-dir results """ EXPORT_NCNN = """\ #!/usr/bin/env bash # Re-export the NCNN model on THIS machine. # # Needed because an NCNN export is not portable in practice. The pnnx tool that # writes it and the ncnn runtime that reads it must agree on the weight layout, # and a pack built elsewhere usually does not match the ncnn wheel installed # here -- it loads, builds the graph, accepts the input, then segfaults in the # forward pass. Exporting locally makes both sides the same version. # # Takes a few minutes: it downloads a pnnx binary for this architecture on the # first run. set -euo pipefail cd "$(dirname "$0")" WEIGHTS=${1:-models/yolo11n.pt} echo "exporting $WEIGHTS to NCNN" ./.venv/bin/python - "$WEIGHTS" <<'PY' import sys, shutil from pathlib import Path from ultralytics import YOLO source = Path(sys.argv[1]) out = YOLO(str(source)).export(format="ncnn", imgsz=640) out = Path(out) # Ultralytics writes _ncnn_model next to the weights. Put it where # run_ncnn.sh looks, replacing whatever shipped in the pack. target = Path("models") / f"{source.stem}_ncnn_model" if target.resolve() != out.resolve(): if target.exists(): shutil.rmtree(target) shutil.move(str(out), str(target)) print(f"\\n{target}") for f in sorted(target.iterdir()): print(f" {f.name} {f.stat().st_size / 1e6:.1f} MB") PY echo echo "now: ./run_ncnn.sh --weights models/$(basename "${WEIGHTS%.*}")_ncnn_model" """ README = """\ # cone-distance evaluation package Self-contained. Runs the same detector, over the same frames, with the same depth estimator, on any machine -- so the numbers from two machines can be put next to each other. ## Use ```bash ./setup.sh # once per machine: creates .venv, installs pinned deps TORCH_CPU=1 ./setup.sh # same, without the 2.5 GB CUDA build of torch ./run_all.sh # all three formats, then the comparison table ``` Individually: ```bash ./run_pt.sh ./run_onnx.sh ./run_ncnn.sh THREADS=4 FRAMES=100 ./run_pt.sh # defaults shown ./run_pt.sh --device 0 # extra flags pass through ``` Results land in `results/` as `_timing.json` and `_detections.csv`. ## What it measures Six stages per frame, reported as p50 and p95: | stage | what | | --- | --- | | `read` | JPEG off disk into a numpy array | | `preprocess` | letterbox and normalise | | `inference` | the forward pass -- the only stage the format changes | | `postprocess` | NMS and rescaling boxes | | `depth` | box to distance, by ray-plane intersection | | `total` | wall clock for the frame | Five warm-up frames are discarded before measuring. ### Read the `depth` row first It is identical numpy over the same boxes whatever the backend, so it has to come out the same in all three columns. When it does not, the run measured CPU contention rather than the export format, and the other rows cannot be trusted either. `compare.py` prints a warning when it drifts more than 25%. This is not hypothetical: onnxruntime defaults to using every core and spin-waits between inferences, which starved the depth stage and made it read 1.5 ms under PyTorch and 16.2 ms under ONNX. The benchmark pins thread count and process CPU affinity to stop that, which is also why `THREADS` should be the same on every machine you compare. ## Contents ``` models/ the three export formats data/ frames, manifest, calibration, class priors src/ depth estimation and benchmark code results/ written by the runners ``` ## Depth estimation Distance comes from geometry, not from a depth sensor or a learned model: a pixel is back-projected to a ray and intersected with the road plane, using focal length, principal point and camera height from `data/calib/`. See `src/depth/__init__.py`. Ground-contact classes (cone, barrier, barrel) use the ground plane, with the class size prior as a cross-check. Stop signs are pole-mounted, so their box bottom is not a ground contact and they use the size prior alone. ## If NCNN fails It typically segfaults in the forward pass: a version mismatch between the pnnx that wrote the export and the ncnn reading it. The runner detects this in a child process and explains rather than dying silently. The fix is to re-export on the machine that will run it: ```bash ./export_ncnn.sh models/yolo11n.pt ./run_ncnn.sh ``` This is the normal thing to do on a Pi, and it is why `export_ncnn.sh` ships with the pack. ## Caveat on the accuracy numbers The bundled weights are stock COCO YOLO11n, which has no cone, barrel or barrier class. It detects cars and people on these frames, so the distances are exercised and timed but are not measurements of anything. Swap in fine-tuned weights and the same scripts become a real evaluation. """ def copy_code(destination: Path) -> None: for item in CODE: source = Path(item) target = destination / item target.parent.mkdir(parents=True, exist_ok=True) if source.is_dir(): shutil.copytree(source, target, dirs_exist_ok=True, ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) else: shutil.copy(source, target) def copy_models(destination: Path, models: list[str]) -> list[str]: (destination / "models").mkdir(parents=True, exist_ok=True) copied = [] for model in models: source = Path(model) if not source.exists(): print(f" !! {source} missing, skipping") continue target = destination / "models" / source.name if source.is_dir(): shutil.copytree(source, target, dirs_exist_ok=True) else: shutil.copy(source, target) copied.append(source.name) print(f" models/{source.name}") return copied def copy_data(root: Path, destination: Path, dataset: str, split: str, frames: int) -> int: """Copy a fixed slice of frames plus the metadata they need.""" manifest = schema.read_manifest(root / "manifest.parquet") subset = manifest[(manifest["source"] == dataset) & (manifest["split"] == split)] # Sorted, not sampled: two builds of the same size must contain the same # frames or the machines are not comparable. keep = sorted(subset["image_path"].unique())[:frames] subset = subset[subset["image_path"].isin(keep)].copy() data = destination / "data" (data / "calib").mkdir(parents=True, exist_ok=True) for image_path in keep: source = paths.resolve_image(root, image_path).resolve() target = data / "images" / image_path target.parent.mkdir(parents=True, exist_ok=True) shutil.copy(source, target) schema.write_manifest(subset, data / "manifest.parquet") shutil.copy(root / "class_priors.yaml", data / "class_priors.yaml") for sensor_id in sorted(subset["sensor_id"].dropna().unique()): calibration = root / "calib" / f"{sensor_id}.yaml" if calibration.exists(): shutil.copy(calibration, data / "calib" / calibration.name) print(f" data/ {len(keep)} frames, {len(subset)} manifest rows, " f"{len(list((data / 'calib').glob('*.yaml')))} calibrations") return len(keep) def write_scripts(destination: Path, frames: int, models: list[str]) -> None: (destination / "requirements.txt").write_text(REQUIREMENTS) (destination / "README.md").write_text(README) (destination / "setup.sh").write_text(SETUP) weights = {"pt": "models/yolo11n.pt", "onnx": "models/yolo11n.onnx", "ncnn": "models/yolo11n_ncnn_model"} labels = {"pt": "PyTorch", "onnx": "ONNX", "ncnn": "NCNN"} for tag, path in weights.items(): (destination / f"run_{tag}.sh").write_text( RUNNER.format(tag=tag, weights=path, label=labels[tag], frames=frames)) (destination / "run_all.sh").write_text(RUN_ALL) (destination / "export_ncnn.sh").write_text(EXPORT_NCNN) for script in destination.glob("*.sh"): script.chmod(0o755) (destination / "results").mkdir(exist_ok=True) def main() -> None: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--out", type=Path, default=Path("evalpack")) parser.add_argument("--unified-root", type=Path, default=None) parser.add_argument("--dataset", default="nuscenes") parser.add_argument("--split", default="val") parser.add_argument("--frames", type=int, default=100) parser.add_argument("--models", nargs="+", default=["yolo11n.pt", "yolo11n.onnx", "yolo11n_ncnn_model"]) parser.add_argument("--tar", action="store_true", help="also write .tar.gz") args = parser.parse_args() root = paths.unified_root(args.unified_root) if args.out.exists(): shutil.rmtree(args.out) args.out.mkdir(parents=True) print(f"building {args.out}/") copy_code(args.out) copy_models(args.out, args.models) copy_data(root, args.out, args.dataset, args.split, args.frames) write_scripts(args.out, args.frames, args.models) size = sum(f.stat().st_size for f in args.out.rglob("*") if f.is_file()) print(f"\n{args.out}/ {size / 1e6:.0f} MB") if args.tar: archive = f"{args.out}.tar.gz" # Excluded rather than assumed absent: the pack is usually tarred after # someone has already run setup.sh in it, and a 2 GB venv built for the # wrong architecture is worse than useless on the far end. subprocess.run(["tar", "-czf", archive, "--exclude=.venv", "--exclude=results", "--exclude=__pycache__", "-C", str(args.out.parent), args.out.name], check=True) print(f"{archive} {Path(archive).stat().st_size / 1e6:.0f} MB") if __name__ == "__main__": main()