File size: 16,351 Bytes
36f5b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
250b018
 
 
 
36f5b69
 
 
250b018
 
 
 
 
 
 
36f5b69
 
250b018
 
 
 
36f5b69
 
250b018
36f5b69
250b018
 
36f5b69
250b018
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36f5b69
 
250b018
 
 
 
 
 
 
 
36f5b69
 
250b018
36f5b69
 
 
250b018
 
36f5b69
 
250b018
36f5b69
 
250b018
 
 
 
 
 
 
 
36f5b69
250b018
 
36f5b69
250b018
 
36f5b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1301e84
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36f5b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1301e84
 
 
 
 
 
 
 
 
 
 
 
 
36f5b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1301e84
36f5b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
"""Build a self-contained evaluation package that runs anywhere.

The problem this solves: benchmark numbers from different machines are only
comparable if the machines ran the same code over the same images with the same
models. Describing that in a README and hoping is not enough -- someone will
have a different dataset slice, or a stale export, and the comparison quietly
stops meaning anything.

So the package carries everything: the three model formats, a fixed set of
frames as real files, the calibration and priors those frames need, the depth
and benchmark code, and three shell scripts that each print one table.

    python scripts/make_evalpack.py --frames 100 --tar

Ship the tarball, run setup.sh once, then run_all.sh.

Notes on what goes in:

* Images are **copied**, not symlinked. The whole repo uses symlinks into the
  dataset roots to avoid duplicating 8 GB, which is right at home and useless
  in something meant to be sent somewhere else.
* Frames are taken in sorted order, not sampled randomly, so two builds of the
  same size contain the same frames and two machines can be compared directly.
* Only the calibrations those frames actually use are copied -- otherwise the
  pack carries fifteen Argoverse 2 calibrations that nothing references.
"""

from __future__ import annotations

import argparse
import shutil
import subprocess
from pathlib import Path

import pandas as pd

from src.common import paths, schema

CODE = ["src/__init__.py", "src/common", "src/depth", "src/bench"]

REQUIREMENTS = """\
# Pinned to the versions the comparison was validated against. Ultralytics in
# particular changes training and export defaults between minor releases.
ultralytics<=8.3.40
onnxruntime>=1.17
ncnn
numpy>=1.24
pandas>=2.0
pyarrow>=14.0
pyyaml>=6.0
"""

SETUP = """\
#!/usr/bin/env bash
# Create a virtual environment and install the pinned dependencies.
# Run once per machine. Safe to re-run.
#
# Everything is logged to setup.log and echoed to the terminal, and pip is NOT
# run quietly. On a Pi this pulls about 2 GB onto an SD card over several
# minutes, and silence for that long is indistinguishable from a hang.
set -euo pipefail
cd "$(dirname "$0")"

LOG=setup.log
: > "$LOG"
exec > >(tee -a "$LOG") 2>&1

START=$(date +%s)
step() { echo; echo "[$(date +%H:%M:%S)  +$(($(date +%s) - START))s] $*"; }

PYTHON=${PYTHON:-python3}

step "checking prerequisites"
echo "  python : $("$PYTHON" -V 2>&1) at $(command -v "$PYTHON")"
echo "  arch   : $(uname -m)"
df -h . | awk 'NR==2 {print "  disk   : " $4 " free"}'

if command -v uv >/dev/null 2>&1; then
    step "creating .venv with uv"
    [ -d .venv ] || uv venv .venv
    step "installing dependencies with uv"
    VIRTUAL_ENV="$PWD/.venv" uv pip install -r requirements.txt
else
    echo "  uv     : not installed (pip will be used; uv is far faster)"

    # Only pip's path needs this. uv builds a venv without ensurepip, so the
    # check belongs here rather than up front. The classic Raspberry Pi OS
    # failure is python3 present but python3-venv absent, which makes
    # `python3 -m venv` sit for a while and then fail confusingly.
    if ! "$PYTHON" -c "import ensurepip" 2>/dev/null; then
        echo
        echo "  !! $PYTHON cannot import ensurepip, so it cannot build a venv."
        echo "     On Raspberry Pi OS or Debian:"
        echo "         sudo apt update && sudo apt install -y python3-venv python3-pip"
        echo
        echo "     Or install uv, which does not need it:"
        echo "         curl -LsSf https://astral.sh/uv/install.sh | sh"
        exit 1
    fi
    echo "  ensurepip: OK"

    if [ -d .venv ]; then
        step ".venv already exists, reusing it"
    else
        step "creating .venv with $PYTHON"
        echo "  bootstraps pip into the new environment; on an SD card this can"
        echo "  take a few minutes with no output. Watch it from another shell:"
        echo "      watch -n2 du -sh $PWD/.venv"
        "$PYTHON" -m venv .venv
    fi

    step "upgrading pip"
    ./.venv/bin/python -m pip install --upgrade pip

    step "installing dependencies (~2 GB, several minutes on a Pi)"
    # Deliberately not --quiet: progress here is the difference between
    # "working" and "hung".
    ./.venv/bin/python -m pip install --progress-bar on -r requirements.txt
fi

step "verifying"
./.venv/bin/python - <<'PY'
import platform
print(f"  python   {platform.python_version()}  {platform.machine()}")
missing = []
for name in ("torch", "ultralytics", "onnxruntime", "numpy", "cv2", "ncnn"):
    try:
        mod = __import__(name)
        print(f"  {name:12s} {getattr(mod, '__version__', 'installed')}")
    except ImportError:
        print(f"  {name:12s} NOT INSTALLED")
        missing.append(name)
required = [m for m in missing if m != "ncnn"]
if required:
    print()
    print(f"  missing and required: {', '.join(required)}")
elif "ncnn" in missing:
    print()
    print("  ncnn is absent. PyTorch and ONNX still run; see PI_RUNBOOK.md step 5.")
PY

step "done in $(($(date +%s) - START))s, logged to $LOG"
echo
echo "next: ./run_onnx.sh    (the format known to work)"
echo "      ./run_all.sh     (all three, then the comparison)"
"""

RUNNER = """\
#!/usr/bin/env bash
# Benchmark the {label} model and print the timing table.
#
# THREADS controls the CPU budget. Pin it to the same value on every machine
# being compared -- an unpinned run measures core count as much as format.
# A Pi 5 has 4 cores, so 4 is the default.
set -euo pipefail
cd "$(dirname "$0")"

THREADS=${{THREADS:-4}}
FRAMES=${{FRAMES:-{frames}}}

export UNIFIED_ROOT=data
exec ./.venv/bin/python -m src.bench.eval_{tag} \\
    --weights "{weights}" \\
    --unified-root data \\
    --limit "$FRAMES" \\
    --threads "$THREADS" \\
    --out-dir results \\
    "$@"
"""

RUN_ALL = """\
#!/usr/bin/env bash
# Run all three formats and print the comparison.
#
# NCNN is allowed to fail without taking the run down: on x86 a pnnx/ncnn
# version mismatch makes it segfault, and the other two numbers are still worth
# having. On the Pi it should succeed, and that is the number that matters.
set -uo pipefail
cd "$(dirname "$0")"

./run_pt.sh   "$@" || echo "!! PyTorch run failed"
./run_onnx.sh "$@" || echo "!! ONNX run failed"
./run_ncnn.sh "$@" || echo "!! NCNN run failed (expected on x86 -- see README)"

echo
export UNIFIED_ROOT=data
./.venv/bin/python -m src.bench.compare --bench-dir results
"""

EXPORT_NCNN = """\
#!/usr/bin/env bash
# Re-export the NCNN model on THIS machine.
#
# Needed because an NCNN export is not portable in practice. The pnnx tool that
# writes it and the ncnn runtime that reads it must agree on the weight layout,
# and a pack built elsewhere usually does not match the ncnn wheel installed
# here -- it loads, builds the graph, accepts the input, then segfaults in the
# forward pass. Exporting locally makes both sides the same version.
#
# Takes a few minutes: it downloads a pnnx binary for this architecture on the
# first run.
set -euo pipefail
cd "$(dirname "$0")"

WEIGHTS=${1:-models/yolo11n.pt}
echo "exporting $WEIGHTS to NCNN"

./.venv/bin/python - "$WEIGHTS" <<'PY'
import sys, shutil
from pathlib import Path
from ultralytics import YOLO

source = Path(sys.argv[1])
out = YOLO(str(source)).export(format="ncnn", imgsz=640)
out = Path(out)

# Ultralytics writes <stem>_ncnn_model next to the weights. Put it where
# run_ncnn.sh looks, replacing whatever shipped in the pack.
target = Path("models") / f"{source.stem}_ncnn_model"
if target.resolve() != out.resolve():
    if target.exists():
        shutil.rmtree(target)
    shutil.move(str(out), str(target))
print(f"\\n{target}")
for f in sorted(target.iterdir()):
    print(f"  {f.name}  {f.stat().st_size / 1e6:.1f} MB")
PY

echo
echo "now: ./run_ncnn.sh --weights models/$(basename "${WEIGHTS%.*}")_ncnn_model"
"""

README = """\
# cone-distance evaluation package

Self-contained. Runs the same detector, over the same frames, with the same
depth estimator, on any machine -- so the numbers from two machines can be put
next to each other.

## Use

```bash
./setup.sh                  # once per machine: creates .venv, installs pinned deps
TORCH_CPU=1 ./setup.sh      # same, without the 2.5 GB CUDA build of torch
./run_all.sh        # all three formats, then the comparison table
```

Individually:

```bash
./run_pt.sh
./run_onnx.sh
./run_ncnn.sh
THREADS=4 FRAMES=100 ./run_pt.sh        # defaults shown
./run_pt.sh --device 0                  # extra flags pass through
```

Results land in `results/` as `<tag>_timing.json` and `<tag>_detections.csv`.

## What it measures

Six stages per frame, reported as p50 and p95:

| stage | what |
| --- | --- |
| `read` | JPEG off disk into a numpy array |
| `preprocess` | letterbox and normalise |
| `inference` | the forward pass -- the only stage the format changes |
| `postprocess` | NMS and rescaling boxes |
| `depth` | box to distance, by ray-plane intersection |
| `total` | wall clock for the frame |

Five warm-up frames are discarded before measuring.

### Read the `depth` row first

It is identical numpy over the same boxes whatever the backend, so it has to
come out the same in all three columns. When it does not, the run measured CPU
contention rather than the export format, and the other rows cannot be trusted
either. `compare.py` prints a warning when it drifts more than 25%.

This is not hypothetical: onnxruntime defaults to using every core and
spin-waits between inferences, which starved the depth stage and made it read
1.5 ms under PyTorch and 16.2 ms under ONNX. The benchmark pins thread count and
process CPU affinity to stop that, which is also why `THREADS` should be the
same on every machine you compare.

## Contents

```
models/          the three export formats
data/            frames, manifest, calibration, class priors
src/             depth estimation and benchmark code
results/         written by the runners
```

## Depth estimation

Distance comes from geometry, not from a depth sensor or a learned model: a
pixel is back-projected to a ray and intersected with the road plane, using
focal length, principal point and camera height from `data/calib/`. See
`src/depth/__init__.py`.

Ground-contact classes (cone, barrier, barrel) use the ground plane, with the
class size prior as a cross-check. Stop signs are pole-mounted, so their box
bottom is not a ground contact and they use the size prior alone.

## If NCNN fails

It typically segfaults in the forward pass: a version mismatch between the pnnx
that wrote the export and the ncnn reading it. The runner detects this in a
child process and explains rather than dying silently.

The fix is to re-export on the machine that will run it:

```bash
./export_ncnn.sh models/yolo11n.pt
./run_ncnn.sh
```

This is the normal thing to do on a Pi, and it is why `export_ncnn.sh` ships
with the pack.

## Caveat on the accuracy numbers

The bundled weights are stock COCO YOLO11n, which has no cone, barrel or
barrier class. It detects cars and people on these frames, so the distances are
exercised and timed but are not measurements of anything. Swap in fine-tuned
weights and the same scripts become a real evaluation.
"""


def copy_code(destination: Path) -> None:
    for item in CODE:
        source = Path(item)
        target = destination / item
        target.parent.mkdir(parents=True, exist_ok=True)
        if source.is_dir():
            shutil.copytree(source, target, dirs_exist_ok=True,
                            ignore=shutil.ignore_patterns("__pycache__", "*.pyc"))
        else:
            shutil.copy(source, target)


def copy_models(destination: Path, models: list[str]) -> list[str]:
    (destination / "models").mkdir(parents=True, exist_ok=True)
    copied = []
    for model in models:
        source = Path(model)
        if not source.exists():
            print(f"  !! {source} missing, skipping")
            continue
        target = destination / "models" / source.name
        if source.is_dir():
            shutil.copytree(source, target, dirs_exist_ok=True)
        else:
            shutil.copy(source, target)
        copied.append(source.name)
        print(f"  models/{source.name}")
    return copied


def copy_data(root: Path, destination: Path, dataset: str, split: str,
              frames: int) -> int:
    """Copy a fixed slice of frames plus the metadata they need."""
    manifest = schema.read_manifest(root / "manifest.parquet")
    subset = manifest[(manifest["source"] == dataset) & (manifest["split"] == split)]

    # Sorted, not sampled: two builds of the same size must contain the same
    # frames or the machines are not comparable.
    keep = sorted(subset["image_path"].unique())[:frames]
    subset = subset[subset["image_path"].isin(keep)].copy()

    data = destination / "data"
    (data / "calib").mkdir(parents=True, exist_ok=True)

    for image_path in keep:
        source = paths.resolve_image(root, image_path).resolve()
        target = data / "images" / image_path
        target.parent.mkdir(parents=True, exist_ok=True)
        shutil.copy(source, target)

    schema.write_manifest(subset, data / "manifest.parquet")
    shutil.copy(root / "class_priors.yaml", data / "class_priors.yaml")

    for sensor_id in sorted(subset["sensor_id"].dropna().unique()):
        calibration = root / "calib" / f"{sensor_id}.yaml"
        if calibration.exists():
            shutil.copy(calibration, data / "calib" / calibration.name)

    print(f"  data/ {len(keep)} frames, {len(subset)} manifest rows, "
          f"{len(list((data / 'calib').glob('*.yaml')))} calibrations")
    return len(keep)


def write_scripts(destination: Path, frames: int, models: list[str]) -> None:
    (destination / "requirements.txt").write_text(REQUIREMENTS)
    (destination / "README.md").write_text(README)
    (destination / "setup.sh").write_text(SETUP)

    weights = {"pt": "models/yolo11n.pt", "onnx": "models/yolo11n.onnx",
               "ncnn": "models/yolo11n_ncnn_model"}
    labels = {"pt": "PyTorch", "onnx": "ONNX", "ncnn": "NCNN"}
    for tag, path in weights.items():
        (destination / f"run_{tag}.sh").write_text(
            RUNNER.format(tag=tag, weights=path, label=labels[tag], frames=frames))
    (destination / "run_all.sh").write_text(RUN_ALL)
    (destination / "export_ncnn.sh").write_text(EXPORT_NCNN)

    for script in destination.glob("*.sh"):
        script.chmod(0o755)
    (destination / "results").mkdir(exist_ok=True)


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__,
                                     formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("--out", type=Path, default=Path("evalpack"))
    parser.add_argument("--unified-root", type=Path, default=None)
    parser.add_argument("--dataset", default="nuscenes")
    parser.add_argument("--split", default="val")
    parser.add_argument("--frames", type=int, default=100)
    parser.add_argument("--models", nargs="+",
                        default=["yolo11n.pt", "yolo11n.onnx", "yolo11n_ncnn_model"])
    parser.add_argument("--tar", action="store_true", help="also write <out>.tar.gz")
    args = parser.parse_args()

    root = paths.unified_root(args.unified_root)
    if args.out.exists():
        shutil.rmtree(args.out)
    args.out.mkdir(parents=True)

    print(f"building {args.out}/")
    copy_code(args.out)
    copy_models(args.out, args.models)
    copy_data(root, args.out, args.dataset, args.split, args.frames)
    write_scripts(args.out, args.frames, args.models)

    size = sum(f.stat().st_size for f in args.out.rglob("*") if f.is_file())
    print(f"\n{args.out}/  {size / 1e6:.0f} MB")

    if args.tar:
        archive = f"{args.out}.tar.gz"
        # Excluded rather than assumed absent: the pack is usually tarred after
        # someone has already run setup.sh in it, and a 2 GB venv built for the
        # wrong architecture is worse than useless on the far end.
        subprocess.run(["tar", "-czf", archive,
                        "--exclude=.venv", "--exclude=results",
                        "--exclude=__pycache__",
                        "-C", str(args.out.parent), args.out.name], check=True)
        print(f"{archive}  {Path(archive).stat().st_size / 1e6:.0f} MB")


if __name__ == "__main__":
    main()