Aryan Sethi
Claude Opus 5 (1M context)
Drop barrel, add held-out stop-sign distance, add the GPU runbook
5d449ff Download src/depth/evaluate.py from Aryan006/cone-distance: direct link, hf CLI and curl.
- Browser
- Download file 14.4 kB
-
https://huggingface.co/Aryan006/cone-distance/resolve/main/src/depth/evaluate.py
- Command line
-
hf download hf://Aryan006/cone-distance/src/depth/evaluate.py
-
curl -L -o evaluate.py https://huggingface.co/Aryan006/cone-distance/resolve/main/src/depth/evaluate.py
14.4 kB
| """Score the depth estimator against LiDAR ground truth on the nuScenes val split. | |
| Two modes, and the difference between them matters: | |
| --source gt estimate from the ground-truth boxes. Measures the depth | |
| module alone, with detector error removed. These are the | |
| numbers that say whether the geometry works. | |
| --source detect run a detector, match its boxes to ground truth by IoU, and | |
| estimate from those. Measures the system end to end, so it | |
| folds in every box the detector missed or misplaced. | |
| Run `gt` to develop against and `detect` to report. A `detect` number that is | |
| worse than its `gt` counterpart is the detector's contribution, and separating | |
| the two is the only way to know which to work on. | |
| Usage: | |
| python -m src.depth.evaluate --source gt | |
| python -m src.depth.evaluate --source detect --weights yolo11n.pt | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| from pathlib import Path | |
| import numpy as np | |
| import pandas as pd | |
| import yaml | |
| from src.common import calib, paths, schema | |
| from src.depth import estimate as depth_estimate | |
| BUCKETS = [(0, 10), (10, 20), (20, 30), (30, 45), (45, 1000)] | |
| def load_priors(root: Path) -> dict: | |
| """Read class_priors.yaml -- the per-class physical sizes. | |
| Written by `src.data.merge` from the datasets' own 3D boxes rather than | |
| looked up from a standards document, so it describes the objects actually in | |
| this data. `{class: {height_m, width_m, length_m, height_std_m, n_samples}}`. | |
| Read `n_samples` before trusting any of it. Fitted on nuScenes mini alone | |
| the cone height came out 0.78 m; across all 850 trainval scenes it is | |
| 1.07 m. A prior derived from too small a sample looks authoritative and is | |
| 27% wrong, which is worse than an honest lookup. | |
| """ | |
| with open(root / "class_priors.yaml") as handle: | |
| return yaml.safe_load(handle) | |
| def boxes_from_manifest(root: Path, source: str, split: str) -> pd.DataFrame: | |
| """The objects we can actually score: one source, one split, with truth. | |
| Three filters, each load-bearing: | |
| * `objects_only` drops the negative rows -- images carrying no objects, | |
| which exist so the YOLO export gets empty label files. They have no box. | |
| * `source` and `split` restrict to one dataset's val set. Mixing sources | |
| would average over different cameras and different ego-frame conventions; | |
| mixing splits would score on frames the detector trained on. | |
| * `gt_distance_m.notna()` drops COCO and BDD, which have no depth truth at | |
| all. They train the detector and are invisible here, by design. | |
| `.copy()` because the caller adds columns, and a slice of a DataFrame is a | |
| view -- assigning into it raises SettingWithCopyWarning and may not stick. | |
| """ | |
| frame = schema.objects_only(schema.read_manifest(root / "manifest.parquet")) | |
| frame = frame[(frame["source"] == source) & (frame["split"] == split)] | |
| return frame[frame["gt_distance_m"].notna()].copy() | |
| # --------------------------------------------------------------------------- | |
| # Detection | |
| # --------------------------------------------------------------------------- | |
| def iou(box, others: np.ndarray) -> np.ndarray: | |
| """Intersection over union of one box against many. Returns (N,). | |
| Used to decide which detection corresponds to which ground-truth object. | |
| Intersection is the overlap rectangle, which `np.clip(..., 0, None)` forces | |
| to zero area when the boxes miss entirely rather than letting a negative | |
| width multiply a negative height into a spurious positive. Union is the sum | |
| of areas minus the double-counted overlap. | |
| Vectorised over `others` because this runs once per ground-truth object per | |
| frame, and a Python loop over detections would dominate the run. | |
| """ | |
| x1 = np.maximum(box[0], others[:, 0]) | |
| y1 = np.maximum(box[1], others[:, 1]) | |
| x2 = np.minimum(box[2], others[:, 2]) | |
| y2 = np.minimum(box[3], others[:, 3]) | |
| overlap = np.clip(x2 - x1, 0, None) * np.clip(y2 - y1, 0, None) | |
| area = (box[2] - box[0]) * (box[3] - box[1]) | |
| areas = (others[:, 2] - others[:, 0]) * (others[:, 3] - others[:, 1]) | |
| union = area + areas - overlap | |
| return np.where(union > 0, overlap / union, 0.0) | |
| def detect_boxes(frame: pd.DataFrame, root: Path, weights: str, device: str, | |
| conf: float, min_iou: float) -> pd.DataFrame: | |
| """Replace each ground-truth box with the detected box that best overlaps it. | |
| Matching is class-agnostic on purpose. Stock YOLO11n knows COCO, which has no | |
| cone or barrier class, so requiring a class match would leave nothing | |
| to measure. Treating the detector as a box proposer still exercises the whole | |
| path and gives a real localisation error to fold in; once the fine-tuned | |
| weights exist, the class labels become meaningful and this can tighten. | |
| """ | |
| from ultralytics import YOLO | |
| model = YOLO(weights) | |
| rows, matched, total = [], 0, 0 | |
| for image_path, group in frame.groupby("image_path"): | |
| full_path = paths.resolve_image(root, image_path) | |
| result = model.predict(str(full_path), conf=conf, device=device, verbose=False)[0] | |
| predicted = result.boxes.xyxy.cpu().numpy() if len(result.boxes) else np.zeros((0, 4)) | |
| total += len(group) | |
| for _, row in group.iterrows(): | |
| if len(predicted) == 0: | |
| continue | |
| truth = np.array([row.x1, row.y1, row.x2, row.y2]) | |
| scores = iou(truth, predicted) | |
| best = int(np.argmax(scores)) | |
| if scores[best] < min_iou: | |
| continue | |
| matched += 1 | |
| new = row.copy() | |
| new["x1"], new["y1"], new["x2"], new["y2"] = predicted[best] | |
| new["iou"] = float(scores[best]) | |
| rows.append(new) | |
| print(f"detector matched {matched}/{total} ground-truth boxes " | |
| f"at IoU >= {min_iou}") | |
| if not rows: | |
| return pd.DataFrame(columns=list(frame.columns) + ["iou"]) | |
| return pd.DataFrame(rows) | |
| # --------------------------------------------------------------------------- | |
| # Scoring | |
| # --------------------------------------------------------------------------- | |
| def score(frame: pd.DataFrame, root: Path, priors: dict) -> pd.DataFrame: | |
| """Estimate a distance for every box and pair it with the truth. | |
| One row out per row in, carrying the prediction, the truth, and the evidence | |
| fields from `DepthEstimate` -- so `report` can slice by method, by | |
| confidence, or by scene without re-running anything. | |
| Calibrations are loaded once and looked up per row by `sensor_id` rather | |
| than assumed constant. That matters: Argoverse 2 has a separate calibration | |
| per log, since its focal length varies ~1% and its camera height ~6% between | |
| them. A row whose sensor has no calibration file is skipped rather than | |
| guessed at -- COCO rows are deliberately in that position, since those photos | |
| have no shared intrinsics and must never be handed to a geometric estimator. | |
| """ | |
| calibrations = calib.load_all(root) | |
| records = [] | |
| for _, row in frame.iterrows(): | |
| calibration = calibrations.get(row["sensor_id"]) | |
| if calibration is None: | |
| continue | |
| result = depth_estimate.estimate( | |
| (row.x1, row.y1, row.x2, row.y2), row["class"], calibration, priors) | |
| records.append({ | |
| "scene_id": row["scene_id"], | |
| "class": row["class"], | |
| "truth_m": row["gt_distance_m"], | |
| "predicted_m": result.distance_m, | |
| "method": result.method, | |
| "spread_m": result.spread_m, | |
| "disagreement": result.disagreement, | |
| "confident": result.confident, | |
| "reason": result.reason, | |
| }) | |
| return pd.DataFrame(records) | |
| def report(results: pd.DataFrame) -> None: | |
| """Print the evaluation, sliced the ways that actually distinguish causes. | |
| Four views, because a single MAE is close to meaningless here: | |
| * **By range.** Ground-plane error grows as Z^2, so a number pooled over all | |
| ranges mostly reports the range distribution of the test set. | |
| * **By class.** Shows which classes are routed to which estimator -- but see | |
| the scene view before concluding a class is the problem. | |
| * **By scene.** The one that changed how this project reads its own results. | |
| Within a scene the relative error is close to constant, and between scenes | |
| it runs -35% to +35%; that is the flat-road assumption meeting a grade, not | |
| a noisy estimator. Pooling across scenes averages errors of opposite sign | |
| and hides it, and it makes a class look biased when really one scene held | |
| most of that class's objects. | |
| * **Agreement filter.** What MAE would be if estimates the two estimators | |
| disagree on were dropped -- the operating point you would actually deploy. | |
| Then the reasons for missing estimates, which say what limits coverage | |
| rather than leaving it as a bare percentage. | |
| Bias is reported as a median rather than a mean: the far-field tail is heavy | |
| enough that a mean bias mostly reports its worst few members. | |
| """ | |
| total = len(results) | |
| usable = results[results["predicted_m"].notna()].copy() | |
| usable["error"] = usable["predicted_m"] - usable["truth_m"] | |
| print(f"\n{'=' * 66}") | |
| print(f"objects: {total} with an estimate: {len(usable)} " | |
| f"({100 * len(usable) / max(total, 1):.0f}%)") | |
| print("=" * 66) | |
| if usable.empty: | |
| print("no estimates were produced") | |
| if total: | |
| print("\nreasons:") | |
| print(results["reason"].value_counts().head(10).to_string()) | |
| return | |
| print(f"\n{'range':>10} {'n':>6} {'MAE':>9} {'bias':>9} {'p90|err|':>10} {'rel':>7}") | |
| print("-" * 56) | |
| for low, high in BUCKETS: | |
| bucket = usable[(usable.truth_m >= low) & (usable.truth_m < high)] | |
| if len(bucket) < 5: | |
| continue | |
| error = bucket["error"] | |
| label = f"{low}-{high} m" if high < 1000 else f"{low}+ m" | |
| print(f"{label:>10} {len(bucket):6d} {error.abs().mean():8.2f}m " | |
| f"{error.median():+8.2f}m {np.percentile(error.abs(), 90):9.2f}m " | |
| f"{100 * (error.abs() / bucket.truth_m).mean():6.1f}%") | |
| print(f"\n{'class':>12} {'n':>6} {'MAE':>9} {'bias':>9} method") | |
| print("-" * 56) | |
| for class_name, group in usable.groupby("class"): | |
| error = group["error"] | |
| methods = "/".join(sorted(group["method"].unique())) | |
| print(f"{class_name:>12} {len(group):6d} {error.abs().mean():8.2f}m " | |
| f"{error.median():+8.2f}m {methods}") | |
| # Per scene, because the dominant error term is not pixel noise. A single | |
| # global MAE averages over scenes whose errors have opposite signs and hides | |
| # the fact that within a scene the error is close to a constant fraction -- | |
| # the signature of the flat-road assumption failing on a grade, not of a | |
| # noisy estimator. | |
| scenes = usable.copy() | |
| scenes["relative"] = scenes["error"] / scenes["truth_m"] | |
| per_scene = (scenes.groupby("scene_id") | |
| .agg(n=("relative", "size"), median_rel=("relative", "median")) | |
| .query("n >= 10") | |
| .sort_values("median_rel")) | |
| if len(per_scene) > 1: | |
| print(f"\nmedian relative error per scene ({len(per_scene)} scenes with n >= 10)") | |
| print(f" best {per_scene['median_rel'].iloc[0]:+.1%}" | |
| f" worst {per_scene['median_rel'].iloc[-1]:+.1%}" | |
| f" spread (std) {per_scene['median_rel'].std():.1%}") | |
| for scene_id, row in per_scene.iterrows(): | |
| print(f" {str(scene_id)[-8:]} n={int(row['n']):4d} {row['median_rel']:+7.1%}") | |
| confident = usable[usable["confident"]] | |
| if len(confident) and len(confident) < len(usable): | |
| print(f"\nagreement filter (the two estimators within " | |
| f"{depth_estimate.DISAGREEMENT_LIMIT:.0%}):") | |
| print(f" keeps {len(confident)}/{len(usable)} " | |
| f"({100 * len(confident) / len(usable):.0f}%), " | |
| f"MAE {confident['error'].abs().mean():.2f}m " | |
| f"vs {usable['error'].abs().mean():.2f}m over all") | |
| dropped = results[results["predicted_m"].isna()] | |
| if len(dropped): | |
| print(f"\nno estimate ({len(dropped)}):") | |
| print(dropped["reason"].value_counts().head(5).to_string()) | |
| def main() -> None: | |
| """Wire up the two modes and run one. | |
| `--source gt` scores the depth module alone; `--source detect` scores the | |
| system. Run the first while developing the geometry and the second to | |
| report, and read the gap between them as the detector's contribution. | |
| Collapsing them into one number would make it impossible to tell which half | |
| to work on. | |
| `--limit` caps objects for a quick pass. Note it truncates the object list | |
| before grouping by image, so it also reduces the number of frames the | |
| detector has to run on -- which is the point on CPU. | |
| """ | |
| parser = argparse.ArgumentParser(description=__doc__, | |
| formatter_class=argparse.RawDescriptionHelpFormatter) | |
| parser.add_argument("--unified-root", type=Path, default=None) | |
| parser.add_argument("--dataset", default="nuscenes") | |
| parser.add_argument("--split", default="val") | |
| parser.add_argument("--source", choices=["gt", "detect"], default="gt") | |
| parser.add_argument("--weights", default="yolo11n.pt") | |
| parser.add_argument("--device", default="cpu") | |
| parser.add_argument("--conf", type=float, default=0.25) | |
| parser.add_argument("--min-iou", type=float, default=0.5) | |
| parser.add_argument("--limit", type=int, default=None, | |
| help="cap the number of objects, for a quick run") | |
| args = parser.parse_args() | |
| root = paths.unified_root(args.unified_root) | |
| priors = load_priors(root) | |
| frame = boxes_from_manifest(root, args.dataset, args.split) | |
| if args.limit: | |
| frame = frame.head(args.limit) | |
| print(f"{args.dataset} {args.split}: {len(frame)} objects with ground-truth distance") | |
| if args.source == "detect": | |
| frame = detect_boxes(frame, root, args.weights, args.device, | |
| args.conf, args.min_iou) | |
| report(score(frame, root, priors)) | |
| if __name__ == "__main__": | |
| main() | |