"""Argument parsing and the run sequence shared by the three format runners. Kept here so that `eval_pt.py`, `eval_onnx.py` and `eval_ncnn.py` differ only in which weights they point at. Any flag added here applies to all three at once, which is the property that makes their timings comparable. """ from __future__ import annotations import argparse import os from pathlib import Path from src.bench import pipeline from src.common import calib, paths def build_parser(description: str, default_weights: str) -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description=description, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--weights", default=default_weights) parser.add_argument("--unified-root", type=Path, default=None) parser.add_argument("--dataset", default="nuscenes") parser.add_argument("--split", default="val") parser.add_argument("--limit", type=int, default=60, help="frames to run, including %d warm-up" % pipeline.WARMUP_FRAMES) parser.add_argument("--imgsz", type=int, default=640) parser.add_argument("--conf", type=float, default=0.25) parser.add_argument("--device", default="cpu") parser.add_argument("--threads", type=int, default=None, help="CPU threads. Pin it when comparing formats -- the " "default differs between backends and an unpinned " "comparison measures thread count as much as format.") parser.add_argument("--out-dir", type=Path, default=Path("bench")) parser.add_argument("--tag", default=None, help="name for the output files. Defaults to the format, " "which means two runs of the same format with " "different weights overwrite each other -- set it " "when comparing checkpoints.") return parser def run(args, label: str, tag: str) -> dict: """Load, benchmark, report, save. The body of all three runners.""" if args.threads: # Three levers, because no single one constrains every backend. # # torch.set_num_threads covers PyTorch. It does NOT cover onnxruntime, # which runs its own pool and defaults to intra_op_num_threads = 0, # meaning every core on the machine. Worse, that pool spin-waits, so on # a 16-core box it starves the single-threaded numpy in the depth stage # -- measured at 4.4 ms per frame under PyTorch and 13.4 ms under ONNX, # for identical code on the same boxes. Left unpinned, the comparison # measures thread budget rather than export format. # # CPU affinity is the lever that actually binds all of them: it caps the # whole process regardless of which library spawned the thread. It also # makes an x86 run a better proxy for a Pi 5, which has four cores. os.environ["OMP_NUM_THREADS"] = str(args.threads) os.environ["MKL_NUM_THREADS"] = str(args.threads) try: os.sched_setaffinity(0, set(range(args.threads))) except (AttributeError, OSError): pass # not Linux, or not permitted; fall back to the above import torch torch.set_num_threads(args.threads) import torch threads = torch.get_num_threads() try: affinity = len(os.sched_getaffinity(0)) except AttributeError: affinity = None root = paths.unified_root(args.unified_root) image_paths, sensor_ids = pipeline.frames_from_manifest( root, args.dataset, args.split, args.limit) if not image_paths: raise SystemExit(f"no {args.dataset} {args.split} frames in the manifest") priors = pipeline.load_priors(root) calibrations = calib.load_all(root) # Must happen before the model is built: the session is constructed during # load, and onnxruntime reads its thread settings once, at that point. if str(args.weights).endswith(".onnx"): pipeline.configure_onnxruntime(args.threads) model, load_seconds, info = pipeline.load_model(args.weights, args.device) class_names = model.names stages, detections = pipeline.run( model, image_paths, root, calibrations, sensor_ids, priors, args.imgsz, args.conf, args.device, class_names) measured = max(len(image_paths) - pipeline.WARMUP_FRAMES, 0) summary = pipeline.report(label, info, load_seconds, stages, detections, measured, threads, affinity) pipeline.save(summary, detections, args.out_dir, args.tag or tag) return summary