Simam3D / evaluate_depth.py
junaid-simamdigital's picture
eval: add reproducible depth metrics
de70edb verified
Raw History Blame Contribute Delete
3.63 kB
"""Evaluate Simam3D depth predictions against ground-truth arrays.
The evaluator is deliberately dataset-agnostic. It reports raw metrics when
the prediction is metric depth and median-scaled metrics for monocular relative
depth. It does not claim that a relative-depth score is metric reconstruction.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
import numpy as np
def _valid_values(prediction: np.ndarray, ground_truth: np.ndarray, valid_mask=None):
prediction = np.asarray(prediction, dtype=np.float64)
ground_truth = np.asarray(ground_truth, dtype=np.float64)
if prediction.shape != ground_truth.shape or prediction.ndim != 2:
raise ValueError("prediction and ground_truth must be matching HxW arrays")
valid = np.isfinite(prediction) & np.isfinite(ground_truth) & (ground_truth > 0) & (prediction > 0)
if valid_mask is not None:
mask = np.asarray(valid_mask, dtype=bool)
if mask.shape != prediction.shape:
raise ValueError("valid_mask must match prediction shape")
valid &= mask
if not valid.any():
raise ValueError("no valid positive depth pixels")
return prediction[valid], ground_truth[valid]
def depth_metrics(prediction: np.ndarray, ground_truth: np.ndarray, valid_mask=None) -> dict[str, float | int]:
"""Return standard depth metrics for positive, finite pixels."""
pred, gt = _valid_values(prediction, ground_truth, valid_mask)
error = pred - gt
ratio = np.maximum(pred / gt, gt / pred)
return {
"valid_pixel_count": int(len(gt)),
"abs_rel": float(np.mean(np.abs(error) / gt)),
"sq_rel": float(np.mean((error ** 2) / gt)),
"rmse": float(np.sqrt(np.mean(error ** 2))),
"rmse_log": float(np.sqrt(np.mean((np.log(pred) - np.log(gt)) ** 2))),
"delta1": float(np.mean(ratio < 1.25)),
"delta2": float(np.mean(ratio < 1.25 ** 2)),
"delta3": float(np.mean(ratio < 1.25 ** 3)),
}
def evaluate_depth_pair(
prediction: np.ndarray,
ground_truth: np.ndarray,
valid_mask=None,
prediction_is_inverse_depth: bool = False,
) -> dict[str, object]:
"""Evaluate raw and median-scaled metrics without hiding scale ambiguity."""
prediction = np.asarray(prediction, dtype=np.float64)
if prediction_is_inverse_depth:
prediction = 1.0 / np.maximum(prediction, 1e-12)
raw = depth_metrics(prediction, ground_truth, valid_mask)
pred_values, gt_values = _valid_values(prediction, ground_truth, valid_mask)
scale = float(np.median(gt_values) / np.median(pred_values))
scaled = depth_metrics(prediction * scale, ground_truth, valid_mask)
return {
"raw": raw,
"median_scaled": scaled,
"median_scale": scale,
"interpretation": "median_scaled is appropriate for relative-depth comparison; raw requires metric scale",
}
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("prediction", type=Path, help=".npy prediction depth array")
parser.add_argument("ground_truth", type=Path, help=".npy ground-truth depth array")
parser.add_argument("--inverse-depth", action="store_true", help="invert prediction before evaluation")
args = parser.parse_args(argv)
prediction = np.load(args.prediction)
ground_truth = np.load(args.ground_truth)
print(json.dumps(evaluate_depth_pair(prediction, ground_truth, prediction_is_inverse_depth=args.inverse_depth), indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())