"""Pinhole projection helpers. Deliberately tiny and dependency-free so that nuScenes and AV2 go through exactly the same maths. If the two extractors disagree about what a box is, it will not be because they used two different projection functions. """ from __future__ import annotations import numpy as np Box2D = tuple[float, float, float, float] def project_to_image(K: np.ndarray, points_cam: np.ndarray) -> np.ndarray: """Project camera-frame points to pixels. K: (3, 3) intrinsics points_cam: (3, N) points in the camera frame, +Z forward returns: (2, N) pixel coordinates Points at or behind the image plane are the caller's problem -- check Z before calling. """ if points_cam.shape[0] != 3: raise ValueError(f"expected (3, N) points, got {points_cam.shape}") projected = K @ points_cam return projected[:2] / projected[2] def apply_radial_distortion(normalised: np.ndarray, k: tuple[float, float, float] ) -> np.ndarray: """Bend ideal normalised coordinates the way the lens does. (2, N) in, (2, N) out, using the three-term radial model `1 + k1*r^2 + k2*r^4 + k3*r^6`. Argoverse 2 cameras carry real coefficients (k1 near -0.24); nuScenes ships rectified images and passes zeros, for which this is the identity. """ k1, k2, k3 = k if k1 == 0.0 and k2 == 0.0 and k3 == 0.0: return normalised r2 = (normalised ** 2).sum(axis=0) return normalised * (1.0 + k1 * r2 + k2 * r2 ** 2 + k3 * r2 ** 3) def remove_radial_distortion(normalised: np.ndarray, k: tuple[float, float, float], iterations: int = 10) -> np.ndarray: """Inverse of apply_radial_distortion, by fixed-point iteration. The forward model has no closed-form inverse. Iterating `x <- x_distorted / (1 + k1*r^2 + ...)` converges in a handful of steps for the coefficient magnitudes these cameras use. """ k1, k2, k3 = k if k1 == 0.0 and k2 == 0.0 and k3 == 0.0: return normalised undistorted = normalised.copy() for _ in range(iterations): r2 = (undistorted ** 2).sum(axis=0) undistorted = normalised / (1.0 + k1 * r2 + k2 * r2 ** 2 + k3 * r2 ** 3) return undistorted def box_from_points(uv: np.ndarray) -> Box2D: """Axis-aligned box enclosing (2, N) pixel points.""" return ( float(uv[0].min()), float(uv[1].min()), float(uv[0].max()), float(uv[1].max()), ) def box_area(box: Box2D) -> float: x1, y1, x2, y2 = box return max(0.0, x2 - x1) * max(0.0, y2 - y1) def clip_box(box: Box2D, width: int, height: int) -> tuple[Box2D, float]: """Clip a box to the image and report how much of it was lost. Returns (clipped_box, truncation) where truncation is the fraction of the original box area that fell outside the frame, 0.0 for a fully visible box. """ x1, y1, x2, y2 = box clipped = ( float(np.clip(x1, 0.0, width)), float(np.clip(y1, 0.0, height)), float(np.clip(x2, 0.0, width)), float(np.clip(y2, 0.0, height)), ) original_area = box_area(box) if original_area <= 0.0: return clipped, 1.0 return clipped, 1.0 - box_area(clipped) / original_area def point_in_box(point: tuple[float, float], box: Box2D, tolerance: float = 1.0) -> bool: """Is a pixel inside a box, allowing a little slack for rounding?""" u, v = point x1, y1, x2, y2 = box return ( x1 - tolerance <= u <= x2 + tolerance and y1 - tolerance <= v <= y2 + tolerance ) def quaternion_to_rotation_matrix(qw: float, qx: float, qy: float, qz: float) -> np.ndarray: """(w, x, y, z) unit quaternion -> (3, 3) rotation matrix.""" quaternion = np.array([qw, qx, qy, qz], dtype=float) quaternion /= np.linalg.norm(quaternion) w, x, y, z = quaternion return np.array([ [1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y)], [2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x)], [2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y)], ]) def cuboid_corners(center: np.ndarray, rotation: np.ndarray, length: float, width: float, height: float) -> np.ndarray: """Eight corners of a 3D box, in the frame `center` and `rotation` are given in. The object frame is +X along length, +Y along width, +Z along height, with the origin at the cuboid centre -- the convention used by both nuScenes and Argoverse 2 cuboid records. Returns (3, 8). """ half = np.array([length, width, height], dtype=float) / 2.0 signs = np.array([ [+1, +1, +1], [+1, +1, -1], [+1, -1, +1], [+1, -1, -1], [-1, +1, +1], [-1, +1, -1], [-1, -1, +1], [-1, -1, -1], ], dtype=float) corners_object = (signs * half).T # (3, 8) return rotation @ corners_object + np.asarray(center, dtype=float).reshape(3, 1)