cone-distance / scripts /make_evalpack.py
Aryan Sethi
Claude Opus 5 (1M context)
Make setup.sh say what it is doing
250b018
Raw History Blame Contribute Delete
16.4 kB
"""Build a self-contained evaluation package that runs anywhere.
The problem this solves: benchmark numbers from different machines are only
comparable if the machines ran the same code over the same images with the same
models. Describing that in a README and hoping is not enough -- someone will
have a different dataset slice, or a stale export, and the comparison quietly
stops meaning anything.
So the package carries everything: the three model formats, a fixed set of
frames as real files, the calibration and priors those frames need, the depth
and benchmark code, and three shell scripts that each print one table.
python scripts/make_evalpack.py --frames 100 --tar
Ship the tarball, run setup.sh once, then run_all.sh.
Notes on what goes in:
* Images are **copied**, not symlinked. The whole repo uses symlinks into the
dataset roots to avoid duplicating 8 GB, which is right at home and useless
in something meant to be sent somewhere else.
* Frames are taken in sorted order, not sampled randomly, so two builds of the
same size contain the same frames and two machines can be compared directly.
* Only the calibrations those frames actually use are copied -- otherwise the
pack carries fifteen Argoverse 2 calibrations that nothing references.
"""
from __future__ import annotations
import argparse
import shutil
import subprocess
from pathlib import Path
import pandas as pd
from src.common import paths, schema
CODE = ["src/__init__.py", "src/common", "src/depth", "src/bench"]
REQUIREMENTS = """\
# Pinned to the versions the comparison was validated against. Ultralytics in
# particular changes training and export defaults between minor releases.
ultralytics<=8.3.40
onnxruntime>=1.17
ncnn
numpy>=1.24
pandas>=2.0
pyarrow>=14.0
pyyaml>=6.0
"""
SETUP = """\
#!/usr/bin/env bash
# Create a virtual environment and install the pinned dependencies.
# Run once per machine. Safe to re-run.
#
# Everything is logged to setup.log and echoed to the terminal, and pip is NOT
# run quietly. On a Pi this pulls about 2 GB onto an SD card over several
# minutes, and silence for that long is indistinguishable from a hang.
set -euo pipefail
cd "$(dirname "$0")"
LOG=setup.log
: > "$LOG"
exec > >(tee -a "$LOG") 2>&1
START=$(date +%s)
step() { echo; echo "[$(date +%H:%M:%S) +$(($(date +%s) - START))s] $*"; }
PYTHON=${PYTHON:-python3}
step "checking prerequisites"
echo " python : $("$PYTHON" -V 2>&1) at $(command -v "$PYTHON")"
echo " arch : $(uname -m)"
df -h . | awk 'NR==2 {print " disk : " $4 " free"}'
if command -v uv >/dev/null 2>&1; then
step "creating .venv with uv"
[ -d .venv ] || uv venv .venv
step "installing dependencies with uv"
VIRTUAL_ENV="$PWD/.venv" uv pip install -r requirements.txt
else
echo " uv : not installed (pip will be used; uv is far faster)"
# Only pip's path needs this. uv builds a venv without ensurepip, so the
# check belongs here rather than up front. The classic Raspberry Pi OS
# failure is python3 present but python3-venv absent, which makes
# `python3 -m venv` sit for a while and then fail confusingly.
if ! "$PYTHON" -c "import ensurepip" 2>/dev/null; then
echo
echo " !! $PYTHON cannot import ensurepip, so it cannot build a venv."
echo " On Raspberry Pi OS or Debian:"
echo " sudo apt update && sudo apt install -y python3-venv python3-pip"
echo
echo " Or install uv, which does not need it:"
echo " curl -LsSf https://astral.sh/uv/install.sh | sh"
exit 1
fi
echo " ensurepip: OK"
if [ -d .venv ]; then
step ".venv already exists, reusing it"
else
step "creating .venv with $PYTHON"
echo " bootstraps pip into the new environment; on an SD card this can"
echo " take a few minutes with no output. Watch it from another shell:"
echo " watch -n2 du -sh $PWD/.venv"
"$PYTHON" -m venv .venv
fi
step "upgrading pip"
./.venv/bin/python -m pip install --upgrade pip
step "installing dependencies (~2 GB, several minutes on a Pi)"
# Deliberately not --quiet: progress here is the difference between
# "working" and "hung".
./.venv/bin/python -m pip install --progress-bar on -r requirements.txt
fi
step "verifying"
./.venv/bin/python - <<'PY'
import platform
print(f" python {platform.python_version()} {platform.machine()}")
missing = []
for name in ("torch", "ultralytics", "onnxruntime", "numpy", "cv2", "ncnn"):
try:
mod = __import__(name)
print(f" {name:12s} {getattr(mod, '__version__', 'installed')}")
except ImportError:
print(f" {name:12s} NOT INSTALLED")
missing.append(name)
required = [m for m in missing if m != "ncnn"]
if required:
print()
print(f" missing and required: {', '.join(required)}")
elif "ncnn" in missing:
print()
print(" ncnn is absent. PyTorch and ONNX still run; see PI_RUNBOOK.md step 5.")
PY
step "done in $(($(date +%s) - START))s, logged to $LOG"
echo
echo "next: ./run_onnx.sh (the format known to work)"
echo " ./run_all.sh (all three, then the comparison)"
"""
RUNNER = """\
#!/usr/bin/env bash
# Benchmark the {label} model and print the timing table.
#
# THREADS controls the CPU budget. Pin it to the same value on every machine
# being compared -- an unpinned run measures core count as much as format.
# A Pi 5 has 4 cores, so 4 is the default.
set -euo pipefail
cd "$(dirname "$0")"
THREADS=${{THREADS:-4}}
FRAMES=${{FRAMES:-{frames}}}
export UNIFIED_ROOT=data
exec ./.venv/bin/python -m src.bench.eval_{tag} \\
--weights "{weights}" \\
--unified-root data \\
--limit "$FRAMES" \\
--threads "$THREADS" \\
--out-dir results \\
"$@"
"""
RUN_ALL = """\
#!/usr/bin/env bash
# Run all three formats and print the comparison.
#
# NCNN is allowed to fail without taking the run down: on x86 a pnnx/ncnn
# version mismatch makes it segfault, and the other two numbers are still worth
# having. On the Pi it should succeed, and that is the number that matters.
set -uo pipefail
cd "$(dirname "$0")"
./run_pt.sh "$@" || echo "!! PyTorch run failed"
./run_onnx.sh "$@" || echo "!! ONNX run failed"
./run_ncnn.sh "$@" || echo "!! NCNN run failed (expected on x86 -- see README)"
echo
export UNIFIED_ROOT=data
./.venv/bin/python -m src.bench.compare --bench-dir results
"""
EXPORT_NCNN = """\
#!/usr/bin/env bash
# Re-export the NCNN model on THIS machine.
#
# Needed because an NCNN export is not portable in practice. The pnnx tool that
# writes it and the ncnn runtime that reads it must agree on the weight layout,
# and a pack built elsewhere usually does not match the ncnn wheel installed
# here -- it loads, builds the graph, accepts the input, then segfaults in the
# forward pass. Exporting locally makes both sides the same version.
#
# Takes a few minutes: it downloads a pnnx binary for this architecture on the
# first run.
set -euo pipefail
cd "$(dirname "$0")"
WEIGHTS=${1:-models/yolo11n.pt}
echo "exporting $WEIGHTS to NCNN"
./.venv/bin/python - "$WEIGHTS" <<'PY'
import sys, shutil
from pathlib import Path
from ultralytics import YOLO
source = Path(sys.argv[1])
out = YOLO(str(source)).export(format="ncnn", imgsz=640)
out = Path(out)
# Ultralytics writes <stem>_ncnn_model next to the weights. Put it where
# run_ncnn.sh looks, replacing whatever shipped in the pack.
target = Path("models") / f"{source.stem}_ncnn_model"
if target.resolve() != out.resolve():
if target.exists():
shutil.rmtree(target)
shutil.move(str(out), str(target))
print(f"\\n{target}")
for f in sorted(target.iterdir()):
print(f" {f.name} {f.stat().st_size / 1e6:.1f} MB")
PY
echo
echo "now: ./run_ncnn.sh --weights models/$(basename "${WEIGHTS%.*}")_ncnn_model"
"""
README = """\
# cone-distance evaluation package
Self-contained. Runs the same detector, over the same frames, with the same
depth estimator, on any machine -- so the numbers from two machines can be put
next to each other.
## Use
```bash
./setup.sh # once per machine: creates .venv, installs pinned deps
TORCH_CPU=1 ./setup.sh # same, without the 2.5 GB CUDA build of torch
./run_all.sh # all three formats, then the comparison table
```
Individually:
```bash
./run_pt.sh
./run_onnx.sh
./run_ncnn.sh
THREADS=4 FRAMES=100 ./run_pt.sh # defaults shown
./run_pt.sh --device 0 # extra flags pass through
```
Results land in `results/` as `<tag>_timing.json` and `<tag>_detections.csv`.
## What it measures
Six stages per frame, reported as p50 and p95:
| stage | what |
| --- | --- |
| `read` | JPEG off disk into a numpy array |
| `preprocess` | letterbox and normalise |
| `inference` | the forward pass -- the only stage the format changes |
| `postprocess` | NMS and rescaling boxes |
| `depth` | box to distance, by ray-plane intersection |
| `total` | wall clock for the frame |
Five warm-up frames are discarded before measuring.
### Read the `depth` row first
It is identical numpy over the same boxes whatever the backend, so it has to
come out the same in all three columns. When it does not, the run measured CPU
contention rather than the export format, and the other rows cannot be trusted
either. `compare.py` prints a warning when it drifts more than 25%.
This is not hypothetical: onnxruntime defaults to using every core and
spin-waits between inferences, which starved the depth stage and made it read
1.5 ms under PyTorch and 16.2 ms under ONNX. The benchmark pins thread count and
process CPU affinity to stop that, which is also why `THREADS` should be the
same on every machine you compare.
## Contents
```
models/ the three export formats
data/ frames, manifest, calibration, class priors
src/ depth estimation and benchmark code
results/ written by the runners
```
## Depth estimation
Distance comes from geometry, not from a depth sensor or a learned model: a
pixel is back-projected to a ray and intersected with the road plane, using
focal length, principal point and camera height from `data/calib/`. See
`src/depth/__init__.py`.
Ground-contact classes (cone, barrier, barrel) use the ground plane, with the
class size prior as a cross-check. Stop signs are pole-mounted, so their box
bottom is not a ground contact and they use the size prior alone.
## If NCNN fails
It typically segfaults in the forward pass: a version mismatch between the pnnx
that wrote the export and the ncnn reading it. The runner detects this in a
child process and explains rather than dying silently.
The fix is to re-export on the machine that will run it:
```bash
./export_ncnn.sh models/yolo11n.pt
./run_ncnn.sh
```
This is the normal thing to do on a Pi, and it is why `export_ncnn.sh` ships
with the pack.
## Caveat on the accuracy numbers
The bundled weights are stock COCO YOLO11n, which has no cone, barrel or
barrier class. It detects cars and people on these frames, so the distances are
exercised and timed but are not measurements of anything. Swap in fine-tuned
weights and the same scripts become a real evaluation.
"""
def copy_code(destination: Path) -> None:
for item in CODE:
source = Path(item)
target = destination / item
target.parent.mkdir(parents=True, exist_ok=True)
if source.is_dir():
shutil.copytree(source, target, dirs_exist_ok=True,
ignore=shutil.ignore_patterns("__pycache__", "*.pyc"))
else:
shutil.copy(source, target)
def copy_models(destination: Path, models: list[str]) -> list[str]:
(destination / "models").mkdir(parents=True, exist_ok=True)
copied = []
for model in models:
source = Path(model)
if not source.exists():
print(f" !! {source} missing, skipping")
continue
target = destination / "models" / source.name
if source.is_dir():
shutil.copytree(source, target, dirs_exist_ok=True)
else:
shutil.copy(source, target)
copied.append(source.name)
print(f" models/{source.name}")
return copied
def copy_data(root: Path, destination: Path, dataset: str, split: str,
frames: int) -> int:
"""Copy a fixed slice of frames plus the metadata they need."""
manifest = schema.read_manifest(root / "manifest.parquet")
subset = manifest[(manifest["source"] == dataset) & (manifest["split"] == split)]
# Sorted, not sampled: two builds of the same size must contain the same
# frames or the machines are not comparable.
keep = sorted(subset["image_path"].unique())[:frames]
subset = subset[subset["image_path"].isin(keep)].copy()
data = destination / "data"
(data / "calib").mkdir(parents=True, exist_ok=True)
for image_path in keep:
source = paths.resolve_image(root, image_path).resolve()
target = data / "images" / image_path
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copy(source, target)
schema.write_manifest(subset, data / "manifest.parquet")
shutil.copy(root / "class_priors.yaml", data / "class_priors.yaml")
for sensor_id in sorted(subset["sensor_id"].dropna().unique()):
calibration = root / "calib" / f"{sensor_id}.yaml"
if calibration.exists():
shutil.copy(calibration, data / "calib" / calibration.name)
print(f" data/ {len(keep)} frames, {len(subset)} manifest rows, "
f"{len(list((data / 'calib').glob('*.yaml')))} calibrations")
return len(keep)
def write_scripts(destination: Path, frames: int, models: list[str]) -> None:
(destination / "requirements.txt").write_text(REQUIREMENTS)
(destination / "README.md").write_text(README)
(destination / "setup.sh").write_text(SETUP)
weights = {"pt": "models/yolo11n.pt", "onnx": "models/yolo11n.onnx",
"ncnn": "models/yolo11n_ncnn_model"}
labels = {"pt": "PyTorch", "onnx": "ONNX", "ncnn": "NCNN"}
for tag, path in weights.items():
(destination / f"run_{tag}.sh").write_text(
RUNNER.format(tag=tag, weights=path, label=labels[tag], frames=frames))
(destination / "run_all.sh").write_text(RUN_ALL)
(destination / "export_ncnn.sh").write_text(EXPORT_NCNN)
for script in destination.glob("*.sh"):
script.chmod(0o755)
(destination / "results").mkdir(exist_ok=True)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--out", type=Path, default=Path("evalpack"))
parser.add_argument("--unified-root", type=Path, default=None)
parser.add_argument("--dataset", default="nuscenes")
parser.add_argument("--split", default="val")
parser.add_argument("--frames", type=int, default=100)
parser.add_argument("--models", nargs="+",
default=["yolo11n.pt", "yolo11n.onnx", "yolo11n_ncnn_model"])
parser.add_argument("--tar", action="store_true", help="also write <out>.tar.gz")
args = parser.parse_args()
root = paths.unified_root(args.unified_root)
if args.out.exists():
shutil.rmtree(args.out)
args.out.mkdir(parents=True)
print(f"building {args.out}/")
copy_code(args.out)
copy_models(args.out, args.models)
copy_data(root, args.out, args.dataset, args.split, args.frames)
write_scripts(args.out, args.frames, args.models)
size = sum(f.stat().st_size for f in args.out.rglob("*") if f.is_file())
print(f"\n{args.out}/ {size / 1e6:.0f} MB")
if args.tar:
archive = f"{args.out}.tar.gz"
# Excluded rather than assumed absent: the pack is usually tarred after
# someone has already run setup.sh in it, and a 2 GB venv built for the
# wrong architecture is worse than useless on the far end.
subprocess.run(["tar", "-czf", archive,
"--exclude=.venv", "--exclude=results",
"--exclude=__pycache__",
"-C", str(args.out.parent), args.out.name], check=True)
print(f"{archive} {Path(archive).stat().st_size / 1e6:.0f} MB")
if __name__ == "__main__":
main()