cone-distance / src /bench /common_cli.py
Aryan Sethi
Claude Opus 5 (1M context)
Score checkpoints end to end and keep the results
a1e201c
Raw History Blame Contribute Delete
4.71 kB
"""Argument parsing and the run sequence shared by the three format runners.
Kept here so that `eval_pt.py`, `eval_onnx.py` and `eval_ncnn.py` differ only in
which weights they point at. Any flag added here applies to all three at once,
which is the property that makes their timings comparable.
"""
from __future__ import annotations
import argparse
import os
from pathlib import Path
from src.bench import pipeline
from src.common import calib, paths
def build_parser(description: str, default_weights: str) -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description=description, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--weights", default=default_weights)
parser.add_argument("--unified-root", type=Path, default=None)
parser.add_argument("--dataset", default="nuscenes")
parser.add_argument("--split", default="val")
parser.add_argument("--limit", type=int, default=60,
help="frames to run, including %d warm-up"
% pipeline.WARMUP_FRAMES)
parser.add_argument("--imgsz", type=int, default=640)
parser.add_argument("--conf", type=float, default=0.25)
parser.add_argument("--device", default="cpu")
parser.add_argument("--threads", type=int, default=None,
help="CPU threads. Pin it when comparing formats -- the "
"default differs between backends and an unpinned "
"comparison measures thread count as much as format.")
parser.add_argument("--out-dir", type=Path, default=Path("bench"))
parser.add_argument("--tag", default=None,
help="name for the output files. Defaults to the format, "
"which means two runs of the same format with "
"different weights overwrite each other -- set it "
"when comparing checkpoints.")
return parser
def run(args, label: str, tag: str) -> dict:
"""Load, benchmark, report, save. The body of all three runners."""
if args.threads:
# Three levers, because no single one constrains every backend.
#
# torch.set_num_threads covers PyTorch. It does NOT cover onnxruntime,
# which runs its own pool and defaults to intra_op_num_threads = 0,
# meaning every core on the machine. Worse, that pool spin-waits, so on
# a 16-core box it starves the single-threaded numpy in the depth stage
# -- measured at 4.4 ms per frame under PyTorch and 13.4 ms under ONNX,
# for identical code on the same boxes. Left unpinned, the comparison
# measures thread budget rather than export format.
#
# CPU affinity is the lever that actually binds all of them: it caps the
# whole process regardless of which library spawned the thread. It also
# makes an x86 run a better proxy for a Pi 5, which has four cores.
os.environ["OMP_NUM_THREADS"] = str(args.threads)
os.environ["MKL_NUM_THREADS"] = str(args.threads)
try:
os.sched_setaffinity(0, set(range(args.threads)))
except (AttributeError, OSError):
pass # not Linux, or not permitted; fall back to the above
import torch
torch.set_num_threads(args.threads)
import torch
threads = torch.get_num_threads()
try:
affinity = len(os.sched_getaffinity(0))
except AttributeError:
affinity = None
root = paths.unified_root(args.unified_root)
image_paths, sensor_ids = pipeline.frames_from_manifest(
root, args.dataset, args.split, args.limit)
if not image_paths:
raise SystemExit(f"no {args.dataset} {args.split} frames in the manifest")
priors = pipeline.load_priors(root)
calibrations = calib.load_all(root)
# Must happen before the model is built: the session is constructed during
# load, and onnxruntime reads its thread settings once, at that point.
if str(args.weights).endswith(".onnx"):
pipeline.configure_onnxruntime(args.threads)
model, load_seconds, info = pipeline.load_model(args.weights, args.device)
class_names = model.names
stages, detections = pipeline.run(
model, image_paths, root, calibrations, sensor_ids, priors,
args.imgsz, args.conf, args.device, class_names)
measured = max(len(image_paths) - pipeline.WARMUP_FRAMES, 0)
summary = pipeline.report(label, info, load_seconds, stages, detections,
measured, threads, affinity)
pipeline.save(summary, detections, args.out_dir, args.tag or tag)
return summary