cone-distance / src /depth /accuracy.py
Aryan Sethi
Claude Opus 5 (1M context)
Quantise the ONNX export, and record how the model performs
35206ab
Raw History Blame Contribute Delete
13.1 kB
"""Score one model's detections *and* their distances, and save the result.
`evaluate.py` answers "does the depth maths work" and prints. This answers "how
good is this particular model end to end" and writes the answer to disk, keyed
by model, so two checkpoints can be put side by side weeks apart without
re-running the first.
What gets saved, under `--out-dir` (default `results/depth_accuracy/`):
<tag>_detections.csv one row per detection: class, confidence, predicted
distance, true distance where one was matched, error
<tag>_summary.json the aggregates, plus enough metadata to know what
produced them
Matching
--------
Detections are matched to ground truth greedily, most confident first, and each
ground-truth object can be claimed once. Without that rule two overlapping
detections of the same cone both count as correct and "objects found" can
exceed 100%, which it did before this was fixed.
A detection counts as correct only if it overlaps a ground-truth box at
IoU >= 0.5 *and* has the right class. Overlapping the right pixels with the
wrong label is not a detection of that object.
"""
from __future__ import annotations
import argparse
import json
import platform
from datetime import datetime, timezone
from pathlib import Path
import numpy as np
import pandas as pd
import yaml
from src.common import calib, paths, schema
from src.depth import estimate as depth_estimate
from src.depth.evaluate import iou
CONFIDENCE_BANDS = [(0.05, 0.25), (0.25, 0.50), (0.50, 0.75), (0.75, 1.01)]
CONFIDENCE_CUTS = [0.05, 0.10, 0.25, 0.40, 0.50, 0.60, 0.75, 0.85]
DISTANCE_BANDS = [(0, 10), (10, 20), (20, 30), (30, 50), (50, 200)]
def collect(model, manifest: pd.DataFrame, root: Path, calibrations: dict,
priors: dict, conf: float, min_iou: float, device: str) -> pd.DataFrame:
"""Run the model over every frame and pair each detection with the truth."""
rows = []
for image_path, truth in manifest.groupby("image_path"):
calibration = calibrations.get(truth.iloc[0]["sensor_id"])
if calibration is None:
continue
result = model.predict(str(paths.resolve_image(root, image_path)),
conf=conf, device=device, verbose=False)[0]
if not len(result.boxes):
continue
boxes = result.boxes.xyxy.cpu().numpy()
classes = result.boxes.cls.cpu().numpy().astype(int)
confidences = result.boxes.conf.cpu().numpy()
truth_boxes = truth[["x1", "y1", "x2", "y2"]].to_numpy()
claimed: set[int] = set()
# Most confident first, so the best detection of an object claims it.
for index in np.argsort(-confidences):
box = boxes[index]
name = model.names[int(classes[index])]
estimate = depth_estimate.estimate(tuple(box), name, calibration, priors)
scores = iou(box, truth_boxes)
candidates = [j for j in np.argsort(-scores)
if scores[j] >= min_iou and j not in claimed]
matched = bool(candidates)
j = candidates[0] if matched else -1
if matched:
claimed.add(j)
rows.append({
"image_path": image_path,
"predicted_class": name,
"confidence": float(confidences[index]),
# The box is saved so the CSV stands on its own: anything
# downstream -- crops, overlays, a re-scored IoU -- can work
# from the file without re-running the model.
"x1": float(box[0]), "y1": float(box[1]),
"x2": float(box[2]), "y2": float(box[3]),
"predicted_m": estimate.distance_m,
"method": estimate.method,
"spread_m": estimate.spread_m,
"matched": matched,
"truth_class": truth.iloc[j]["class"] if matched else None,
"truth_m": float(truth.iloc[j]["gt_distance_m"]) if matched else np.nan,
"scene_id": truth.iloc[j]["scene_id"] if matched
else truth.iloc[0]["scene_id"],
"iou": float(scores[j]) if matched else 0.0,
})
frame = pd.DataFrame(rows)
if frame.empty:
return frame
# Correct means right place AND right label.
frame["correct"] = frame["matched"] & (frame["predicted_class"] == frame["truth_class"])
frame["error_m"] = frame["predicted_m"] - frame["truth_m"]
frame.loc[~frame["correct"], "error_m"] = np.nan
frame["percent_off"] = 100 * frame["error_m"].abs() / frame["truth_m"]
return frame
def show_samples(frame: pd.DataFrame, per_band: int) -> None:
"""Print evenly spaced real detections from each confidence band.
Evenly spaced through the band rather than the first N, which would all come
from the same few frames, or a random N, which would not reproduce.
"""
if per_band <= 0: # --samples 0 means "just save, do not print"
return
correct = frame[frame["correct"]]
for low, high in CONFIDENCE_BANDS:
band = correct[(correct.confidence >= low) & (correct.confidence < high)]
band = band.sort_values("confidence", ascending=False)
print(f"\n{'=' * 70}")
print(f"confidence {low:.2f} - {min(high, 1.0):.2f} "
f"{len(band)} correct detections")
print("=" * 70)
if band.empty:
print(" none")
continue
step = max(1, len(band) // per_band)
sample = band.iloc[::step][:per_band]
print(f" {'class':<11}{'conf':>7}{'ours':>9}{'truth':>9}{'off by':>10}{'%':>7}")
print(" " + "-" * 51)
for _, row in sample.iterrows():
print(f" {row['predicted_class']:<11}{row['confidence']:>7.2f}"
f"{row['predicted_m']:>8.1f}m{row['truth_m']:>8.1f}m"
f"{row['error_m']:>+9.1f}m{row['percent_off']:>6.0f}%")
def summarise(frame: pd.DataFrame, n_truth: int) -> dict:
"""The aggregates, as plain numbers so they survive to a JSON file."""
correct = frame[frame["correct"]]
by_cut = []
for cut in CONFIDENCE_CUTS:
kept = frame[frame.confidence >= cut]
good = kept[kept["correct"]]
by_cut.append({
"confidence": cut,
"detections": int(len(kept)),
"percent_real": float(100 * kept["correct"].mean()) if len(kept) else 0.0,
"objects_found_percent": float(100 * len(good) / n_truth),
"median_abs_error_m": float(good["error_m"].abs().median()) if len(good) else None,
})
by_confidence = []
for low, high in CONFIDENCE_BANDS:
band = correct[(correct.confidence >= low) & (correct.confidence < high)]
if band.empty:
continue
by_confidence.append({
"band": f"{low:.2f}-{min(high, 1.0):.2f}",
"n": int(len(band)),
"median_abs_error_m": float(band["error_m"].abs().median()),
"within_10_percent": float(100 * (band["percent_off"] <= 10).mean()),
"within_25_percent": float(100 * (band["percent_off"] <= 25).mean()),
})
by_distance = []
for low, high in DISTANCE_BANDS:
band = correct[(correct.truth_m >= low) & (correct.truth_m < high)]
if len(band) < 5:
continue
by_distance.append({
"band_m": f"{low}-{high}",
"n": int(len(band)),
"median_abs_error_m": float(band["error_m"].abs().median()),
"median_percent_off": float(band["percent_off"].median()),
"median_bias_m": float(band["error_m"].median()),
})
# Per scene, because the dominant error is a road-grade offset that is close
# to constant within a scene and swings sign between them. An overall figure
# averages those out and looks better than any individual scene.
by_scene = []
for scene, band in correct.groupby("scene_id"):
if len(band) < 10:
continue
by_scene.append({"scene_id": str(scene), "n": int(len(band)),
"median_percent_off": float(
(100 * band["error_m"] / band["truth_m"]).median())})
return {
"ground_truth_objects": int(n_truth),
"detections": int(len(frame)),
"correct": int(len(correct)),
"overall_median_abs_error_m": float(correct["error_m"].abs().median())
if len(correct) else None,
"by_confidence_cut": by_cut,
"by_confidence_band": by_confidence,
"by_distance": by_distance,
"by_scene": sorted(by_scene, key=lambda r: r["median_percent_off"]),
}
def print_summary(summary: dict) -> None:
print(f"\n{'=' * 70}")
print(f"{summary['ground_truth_objects']} real objects, "
f"{summary['detections']} detections, {summary['correct']} correct")
print("=" * 70)
print(f"\n{'keep above':>12}{'detections':>12}{'% real':>9}"
f"{'objects found':>15}{'error':>9}")
print("-" * 57)
for row in summary["by_confidence_cut"]:
error = f"{row['median_abs_error_m']:.1f}m" if row["median_abs_error_m"] else "-"
print(f"{row['confidence']:>12.2f}{row['detections']:>12}"
f"{row['percent_real']:>8.0f}%{row['objects_found_percent']:>14.0f}%"
f"{error:>9}")
print(f"\n{'confidence':>12}{'n':>7}{'error':>9}{'within 10%':>13}{'within 25%':>13}")
print("-" * 54)
for row in summary["by_confidence_band"]:
print(f"{row['band']:>12}{row['n']:>7}{row['median_abs_error_m']:>8.1f}m"
f"{row['within_10_percent']:>12.0f}%{row['within_25_percent']:>12.0f}%")
print(f"\n{'distance':>12}{'n':>7}{'error':>9}{'% off':>8}{'bias':>9}")
print("-" * 45)
for row in summary["by_distance"]:
print(f"{row['band_m'] + ' m':>12}{row['n']:>7}"
f"{row['median_abs_error_m']:>8.1f}m{row['median_percent_off']:>7.0f}%"
f"{row['median_bias_m']:>+8.1f}m")
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--weights", required=True)
parser.add_argument("--tag", default=None, help="output name; defaults to the "
"weights file stem")
parser.add_argument("--unified-root", type=Path, default=None)
parser.add_argument("--dataset", default="nuscenes")
parser.add_argument("--split", default="val")
parser.add_argument("--conf", type=float, default=0.05,
help="lowest confidence collected. Keep it low: the "
"thresholds in the report are applied afterwards, "
"so one run covers every operating point.")
parser.add_argument("--min-iou", type=float, default=0.5)
parser.add_argument("--device", default="cpu")
parser.add_argument("--samples", type=int, default=20,
help="detections to print per confidence band")
parser.add_argument("--out-dir", type=Path, default=Path("results/depth_accuracy"))
args = parser.parse_args()
from ultralytics import YOLO
root = paths.unified_root(args.unified_root)
manifest = schema.objects_only(schema.read_manifest(root / "manifest.parquet"))
manifest = manifest[(manifest["source"] == args.dataset)
& (manifest["split"] == args.split)
& manifest["gt_distance_m"].notna()].reset_index(drop=True)
priors = yaml.safe_load((root / "class_priors.yaml").read_text())
calibrations = calib.load_all(root)
tag = args.tag or Path(args.weights).stem
model = YOLO(args.weights)
print(f"{tag}: {len(manifest)} ground-truth objects across "
f"{manifest.image_path.nunique()} frames")
frame = collect(model, manifest, root, calibrations, priors,
args.conf, args.min_iou, args.device)
if frame.empty:
raise SystemExit("no detections")
show_samples(frame, args.samples)
summary = summarise(frame, len(manifest))
print_summary(summary)
summary["meta"] = {
"tag": tag, "weights": str(args.weights), "dataset": args.dataset,
"split": args.split, "min_iou": args.min_iou, "collect_conf": args.conf,
"classes": list(model.names.values()),
"created": datetime.now(timezone.utc).isoformat(timespec="seconds"),
"host": platform.node(), "machine": platform.machine(),
}
args.out_dir.mkdir(parents=True, exist_ok=True)
frame.to_csv(args.out_dir / f"{tag}_detections.csv", index=False)
(args.out_dir / f"{tag}_summary.json").write_text(json.dumps(summary, indent=2))
print(f"\nsaved {args.out_dir / f'{tag}_detections.csv'}")
print(f" {args.out_dir / f'{tag}_summary.json'}")
if __name__ == "__main__":
main()