"""Score one model's detections *and* their distances, and save the result. `evaluate.py` answers "does the depth maths work" and prints. This answers "how good is this particular model end to end" and writes the answer to disk, keyed by model, so two checkpoints can be put side by side weeks apart without re-running the first. What gets saved, under `--out-dir` (default `results/depth_accuracy/`): _detections.csv one row per detection: class, confidence, predicted distance, true distance where one was matched, error _summary.json the aggregates, plus enough metadata to know what produced them Matching -------- Detections are matched to ground truth greedily, most confident first, and each ground-truth object can be claimed once. Without that rule two overlapping detections of the same cone both count as correct and "objects found" can exceed 100%, which it did before this was fixed. A detection counts as correct only if it overlaps a ground-truth box at IoU >= 0.5 *and* has the right class. Overlapping the right pixels with the wrong label is not a detection of that object. """ from __future__ import annotations import argparse import json import platform from datetime import datetime, timezone from pathlib import Path import numpy as np import pandas as pd import yaml from src.common import calib, paths, schema from src.depth import estimate as depth_estimate from src.depth.evaluate import iou CONFIDENCE_BANDS = [(0.05, 0.25), (0.25, 0.50), (0.50, 0.75), (0.75, 1.01)] CONFIDENCE_CUTS = [0.05, 0.10, 0.25, 0.40, 0.50, 0.60, 0.75, 0.85] DISTANCE_BANDS = [(0, 10), (10, 20), (20, 30), (30, 50), (50, 200)] def collect(model, manifest: pd.DataFrame, root: Path, calibrations: dict, priors: dict, conf: float, min_iou: float, device: str) -> pd.DataFrame: """Run the model over every frame and pair each detection with the truth.""" rows = [] for image_path, truth in manifest.groupby("image_path"): calibration = calibrations.get(truth.iloc[0]["sensor_id"]) if calibration is None: continue result = model.predict(str(paths.resolve_image(root, image_path)), conf=conf, device=device, verbose=False)[0] if not len(result.boxes): continue boxes = result.boxes.xyxy.cpu().numpy() classes = result.boxes.cls.cpu().numpy().astype(int) confidences = result.boxes.conf.cpu().numpy() truth_boxes = truth[["x1", "y1", "x2", "y2"]].to_numpy() claimed: set[int] = set() # Most confident first, so the best detection of an object claims it. for index in np.argsort(-confidences): box = boxes[index] name = model.names[int(classes[index])] estimate = depth_estimate.estimate(tuple(box), name, calibration, priors) scores = iou(box, truth_boxes) candidates = [j for j in np.argsort(-scores) if scores[j] >= min_iou and j not in claimed] matched = bool(candidates) j = candidates[0] if matched else -1 if matched: claimed.add(j) rows.append({ "image_path": image_path, "predicted_class": name, "confidence": float(confidences[index]), # The box is saved so the CSV stands on its own: anything # downstream -- crops, overlays, a re-scored IoU -- can work # from the file without re-running the model. "x1": float(box[0]), "y1": float(box[1]), "x2": float(box[2]), "y2": float(box[3]), "predicted_m": estimate.distance_m, "method": estimate.method, "spread_m": estimate.spread_m, "matched": matched, "truth_class": truth.iloc[j]["class"] if matched else None, "truth_m": float(truth.iloc[j]["gt_distance_m"]) if matched else np.nan, "scene_id": truth.iloc[j]["scene_id"] if matched else truth.iloc[0]["scene_id"], "iou": float(scores[j]) if matched else 0.0, }) frame = pd.DataFrame(rows) if frame.empty: return frame # Correct means right place AND right label. frame["correct"] = frame["matched"] & (frame["predicted_class"] == frame["truth_class"]) frame["error_m"] = frame["predicted_m"] - frame["truth_m"] frame.loc[~frame["correct"], "error_m"] = np.nan frame["percent_off"] = 100 * frame["error_m"].abs() / frame["truth_m"] return frame def show_samples(frame: pd.DataFrame, per_band: int) -> None: """Print evenly spaced real detections from each confidence band. Evenly spaced through the band rather than the first N, which would all come from the same few frames, or a random N, which would not reproduce. """ if per_band <= 0: # --samples 0 means "just save, do not print" return correct = frame[frame["correct"]] for low, high in CONFIDENCE_BANDS: band = correct[(correct.confidence >= low) & (correct.confidence < high)] band = band.sort_values("confidence", ascending=False) print(f"\n{'=' * 70}") print(f"confidence {low:.2f} - {min(high, 1.0):.2f} " f"{len(band)} correct detections") print("=" * 70) if band.empty: print(" none") continue step = max(1, len(band) // per_band) sample = band.iloc[::step][:per_band] print(f" {'class':<11}{'conf':>7}{'ours':>9}{'truth':>9}{'off by':>10}{'%':>7}") print(" " + "-" * 51) for _, row in sample.iterrows(): print(f" {row['predicted_class']:<11}{row['confidence']:>7.2f}" f"{row['predicted_m']:>8.1f}m{row['truth_m']:>8.1f}m" f"{row['error_m']:>+9.1f}m{row['percent_off']:>6.0f}%") def summarise(frame: pd.DataFrame, n_truth: int) -> dict: """The aggregates, as plain numbers so they survive to a JSON file.""" correct = frame[frame["correct"]] by_cut = [] for cut in CONFIDENCE_CUTS: kept = frame[frame.confidence >= cut] good = kept[kept["correct"]] by_cut.append({ "confidence": cut, "detections": int(len(kept)), "percent_real": float(100 * kept["correct"].mean()) if len(kept) else 0.0, "objects_found_percent": float(100 * len(good) / n_truth), "median_abs_error_m": float(good["error_m"].abs().median()) if len(good) else None, }) by_confidence = [] for low, high in CONFIDENCE_BANDS: band = correct[(correct.confidence >= low) & (correct.confidence < high)] if band.empty: continue by_confidence.append({ "band": f"{low:.2f}-{min(high, 1.0):.2f}", "n": int(len(band)), "median_abs_error_m": float(band["error_m"].abs().median()), "within_10_percent": float(100 * (band["percent_off"] <= 10).mean()), "within_25_percent": float(100 * (band["percent_off"] <= 25).mean()), }) by_distance = [] for low, high in DISTANCE_BANDS: band = correct[(correct.truth_m >= low) & (correct.truth_m < high)] if len(band) < 5: continue by_distance.append({ "band_m": f"{low}-{high}", "n": int(len(band)), "median_abs_error_m": float(band["error_m"].abs().median()), "median_percent_off": float(band["percent_off"].median()), "median_bias_m": float(band["error_m"].median()), }) # Per scene, because the dominant error is a road-grade offset that is close # to constant within a scene and swings sign between them. An overall figure # averages those out and looks better than any individual scene. by_scene = [] for scene, band in correct.groupby("scene_id"): if len(band) < 10: continue by_scene.append({"scene_id": str(scene), "n": int(len(band)), "median_percent_off": float( (100 * band["error_m"] / band["truth_m"]).median())}) return { "ground_truth_objects": int(n_truth), "detections": int(len(frame)), "correct": int(len(correct)), "overall_median_abs_error_m": float(correct["error_m"].abs().median()) if len(correct) else None, "by_confidence_cut": by_cut, "by_confidence_band": by_confidence, "by_distance": by_distance, "by_scene": sorted(by_scene, key=lambda r: r["median_percent_off"]), } def print_summary(summary: dict) -> None: print(f"\n{'=' * 70}") print(f"{summary['ground_truth_objects']} real objects, " f"{summary['detections']} detections, {summary['correct']} correct") print("=" * 70) print(f"\n{'keep above':>12}{'detections':>12}{'% real':>9}" f"{'objects found':>15}{'error':>9}") print("-" * 57) for row in summary["by_confidence_cut"]: error = f"{row['median_abs_error_m']:.1f}m" if row["median_abs_error_m"] else "-" print(f"{row['confidence']:>12.2f}{row['detections']:>12}" f"{row['percent_real']:>8.0f}%{row['objects_found_percent']:>14.0f}%" f"{error:>9}") print(f"\n{'confidence':>12}{'n':>7}{'error':>9}{'within 10%':>13}{'within 25%':>13}") print("-" * 54) for row in summary["by_confidence_band"]: print(f"{row['band']:>12}{row['n']:>7}{row['median_abs_error_m']:>8.1f}m" f"{row['within_10_percent']:>12.0f}%{row['within_25_percent']:>12.0f}%") print(f"\n{'distance':>12}{'n':>7}{'error':>9}{'% off':>8}{'bias':>9}") print("-" * 45) for row in summary["by_distance"]: print(f"{row['band_m'] + ' m':>12}{row['n']:>7}" f"{row['median_abs_error_m']:>8.1f}m{row['median_percent_off']:>7.0f}%" f"{row['median_bias_m']:>+8.1f}m") def main() -> None: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--weights", required=True) parser.add_argument("--tag", default=None, help="output name; defaults to the " "weights file stem") parser.add_argument("--unified-root", type=Path, default=None) parser.add_argument("--dataset", default="nuscenes") parser.add_argument("--split", default="val") parser.add_argument("--conf", type=float, default=0.05, help="lowest confidence collected. Keep it low: the " "thresholds in the report are applied afterwards, " "so one run covers every operating point.") parser.add_argument("--min-iou", type=float, default=0.5) parser.add_argument("--device", default="cpu") parser.add_argument("--samples", type=int, default=20, help="detections to print per confidence band") parser.add_argument("--out-dir", type=Path, default=Path("results/depth_accuracy")) args = parser.parse_args() from ultralytics import YOLO root = paths.unified_root(args.unified_root) manifest = schema.objects_only(schema.read_manifest(root / "manifest.parquet")) manifest = manifest[(manifest["source"] == args.dataset) & (manifest["split"] == args.split) & manifest["gt_distance_m"].notna()].reset_index(drop=True) priors = yaml.safe_load((root / "class_priors.yaml").read_text()) calibrations = calib.load_all(root) tag = args.tag or Path(args.weights).stem model = YOLO(args.weights) print(f"{tag}: {len(manifest)} ground-truth objects across " f"{manifest.image_path.nunique()} frames") frame = collect(model, manifest, root, calibrations, priors, args.conf, args.min_iou, args.device) if frame.empty: raise SystemExit("no detections") show_samples(frame, args.samples) summary = summarise(frame, len(manifest)) print_summary(summary) summary["meta"] = { "tag": tag, "weights": str(args.weights), "dataset": args.dataset, "split": args.split, "min_iou": args.min_iou, "collect_conf": args.conf, "classes": list(model.names.values()), "created": datetime.now(timezone.utc).isoformat(timespec="seconds"), "host": platform.node(), "machine": platform.machine(), } args.out_dir.mkdir(parents=True, exist_ok=True) frame.to_csv(args.out_dir / f"{tag}_detections.csv", index=False) (args.out_dir / f"{tag}_summary.json").write_text(json.dumps(summary, indent=2)) print(f"\nsaved {args.out_dir / f'{tag}_detections.csv'}") print(f" {args.out_dir / f'{tag}_summary.json'}") if __name__ == "__main__": main()