File size: 4,713 Bytes
0b910fa a1e201c 0b910fa a1e201c 0b910fa | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 | """Argument parsing and the run sequence shared by the three format runners.
Kept here so that `eval_pt.py`, `eval_onnx.py` and `eval_ncnn.py` differ only in
which weights they point at. Any flag added here applies to all three at once,
which is the property that makes their timings comparable.
"""
from __future__ import annotations
import argparse
import os
from pathlib import Path
from src.bench import pipeline
from src.common import calib, paths
def build_parser(description: str, default_weights: str) -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description=description, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--weights", default=default_weights)
parser.add_argument("--unified-root", type=Path, default=None)
parser.add_argument("--dataset", default="nuscenes")
parser.add_argument("--split", default="val")
parser.add_argument("--limit", type=int, default=60,
help="frames to run, including %d warm-up"
% pipeline.WARMUP_FRAMES)
parser.add_argument("--imgsz", type=int, default=640)
parser.add_argument("--conf", type=float, default=0.25)
parser.add_argument("--device", default="cpu")
parser.add_argument("--threads", type=int, default=None,
help="CPU threads. Pin it when comparing formats -- the "
"default differs between backends and an unpinned "
"comparison measures thread count as much as format.")
parser.add_argument("--out-dir", type=Path, default=Path("bench"))
parser.add_argument("--tag", default=None,
help="name for the output files. Defaults to the format, "
"which means two runs of the same format with "
"different weights overwrite each other -- set it "
"when comparing checkpoints.")
return parser
def run(args, label: str, tag: str) -> dict:
"""Load, benchmark, report, save. The body of all three runners."""
if args.threads:
# Three levers, because no single one constrains every backend.
#
# torch.set_num_threads covers PyTorch. It does NOT cover onnxruntime,
# which runs its own pool and defaults to intra_op_num_threads = 0,
# meaning every core on the machine. Worse, that pool spin-waits, so on
# a 16-core box it starves the single-threaded numpy in the depth stage
# -- measured at 4.4 ms per frame under PyTorch and 13.4 ms under ONNX,
# for identical code on the same boxes. Left unpinned, the comparison
# measures thread budget rather than export format.
#
# CPU affinity is the lever that actually binds all of them: it caps the
# whole process regardless of which library spawned the thread. It also
# makes an x86 run a better proxy for a Pi 5, which has four cores.
os.environ["OMP_NUM_THREADS"] = str(args.threads)
os.environ["MKL_NUM_THREADS"] = str(args.threads)
try:
os.sched_setaffinity(0, set(range(args.threads)))
except (AttributeError, OSError):
pass # not Linux, or not permitted; fall back to the above
import torch
torch.set_num_threads(args.threads)
import torch
threads = torch.get_num_threads()
try:
affinity = len(os.sched_getaffinity(0))
except AttributeError:
affinity = None
root = paths.unified_root(args.unified_root)
image_paths, sensor_ids = pipeline.frames_from_manifest(
root, args.dataset, args.split, args.limit)
if not image_paths:
raise SystemExit(f"no {args.dataset} {args.split} frames in the manifest")
priors = pipeline.load_priors(root)
calibrations = calib.load_all(root)
# Must happen before the model is built: the session is constructed during
# load, and onnxruntime reads its thread settings once, at that point.
if str(args.weights).endswith(".onnx"):
pipeline.configure_onnxruntime(args.threads)
model, load_seconds, info = pipeline.load_model(args.weights, args.device)
class_names = model.names
stages, detections = pipeline.run(
model, image_paths, root, calibrations, sensor_ids, priors,
args.imgsz, args.conf, args.device, class_names)
measured = max(len(image_paths) - pipeline.WARMUP_FRAMES, 0)
summary = pipeline.report(label, info, load_seconds, stages, detections,
measured, threads, affinity)
pipeline.save(summary, detections, args.out_dir, args.tag or tag)
return summary
|