Aryan Sethi
Claude Opus 5 (1M context)
Fix two cross-dataset convention bugs found by scoring against ground truth
b9ff8f9 Download src/common/geometry.py from Aryan006/cone-distance: direct link, hf CLI and curl.
- Browser
- Download file 5.01 kB
-
https://huggingface.co/Aryan006/cone-distance/resolve/main/src/common/geometry.py
- Command line
-
hf download hf://Aryan006/cone-distance/src/common/geometry.py
-
curl -L -o geometry.py https://huggingface.co/Aryan006/cone-distance/resolve/main/src/common/geometry.py
5.01 kB
| """Pinhole projection helpers. | |
| Deliberately tiny and dependency-free so that nuScenes and AV2 go through | |
| exactly the same maths. If the two extractors disagree about what a box is, | |
| it will not be because they used two different projection functions. | |
| """ | |
| from __future__ import annotations | |
| import numpy as np | |
| Box2D = tuple[float, float, float, float] | |
| def project_to_image(K: np.ndarray, points_cam: np.ndarray) -> np.ndarray: | |
| """Project camera-frame points to pixels. | |
| K: (3, 3) intrinsics | |
| points_cam: (3, N) points in the camera frame, +Z forward | |
| returns: (2, N) pixel coordinates | |
| Points at or behind the image plane are the caller's problem -- check Z | |
| before calling. | |
| """ | |
| if points_cam.shape[0] != 3: | |
| raise ValueError(f"expected (3, N) points, got {points_cam.shape}") | |
| projected = K @ points_cam | |
| return projected[:2] / projected[2] | |
| def apply_radial_distortion(normalised: np.ndarray, k: tuple[float, float, float] | |
| ) -> np.ndarray: | |
| """Bend ideal normalised coordinates the way the lens does. | |
| (2, N) in, (2, N) out, using the three-term radial model | |
| `1 + k1*r^2 + k2*r^4 + k3*r^6`. Argoverse 2 cameras carry real coefficients | |
| (k1 near -0.24); nuScenes ships rectified images and passes zeros, for which | |
| this is the identity. | |
| """ | |
| k1, k2, k3 = k | |
| if k1 == 0.0 and k2 == 0.0 and k3 == 0.0: | |
| return normalised | |
| r2 = (normalised ** 2).sum(axis=0) | |
| return normalised * (1.0 + k1 * r2 + k2 * r2 ** 2 + k3 * r2 ** 3) | |
| def remove_radial_distortion(normalised: np.ndarray, k: tuple[float, float, float], | |
| iterations: int = 10) -> np.ndarray: | |
| """Inverse of apply_radial_distortion, by fixed-point iteration. | |
| The forward model has no closed-form inverse. Iterating | |
| `x <- x_distorted / (1 + k1*r^2 + ...)` converges in a handful of steps for | |
| the coefficient magnitudes these cameras use. | |
| """ | |
| k1, k2, k3 = k | |
| if k1 == 0.0 and k2 == 0.0 and k3 == 0.0: | |
| return normalised | |
| undistorted = normalised.copy() | |
| for _ in range(iterations): | |
| r2 = (undistorted ** 2).sum(axis=0) | |
| undistorted = normalised / (1.0 + k1 * r2 + k2 * r2 ** 2 + k3 * r2 ** 3) | |
| return undistorted | |
| def box_from_points(uv: np.ndarray) -> Box2D: | |
| """Axis-aligned box enclosing (2, N) pixel points.""" | |
| return ( | |
| float(uv[0].min()), | |
| float(uv[1].min()), | |
| float(uv[0].max()), | |
| float(uv[1].max()), | |
| ) | |
| def box_area(box: Box2D) -> float: | |
| x1, y1, x2, y2 = box | |
| return max(0.0, x2 - x1) * max(0.0, y2 - y1) | |
| def clip_box(box: Box2D, width: int, height: int) -> tuple[Box2D, float]: | |
| """Clip a box to the image and report how much of it was lost. | |
| Returns (clipped_box, truncation) where truncation is the fraction of the | |
| original box area that fell outside the frame, 0.0 for a fully visible box. | |
| """ | |
| x1, y1, x2, y2 = box | |
| clipped = ( | |
| float(np.clip(x1, 0.0, width)), | |
| float(np.clip(y1, 0.0, height)), | |
| float(np.clip(x2, 0.0, width)), | |
| float(np.clip(y2, 0.0, height)), | |
| ) | |
| original_area = box_area(box) | |
| if original_area <= 0.0: | |
| return clipped, 1.0 | |
| return clipped, 1.0 - box_area(clipped) / original_area | |
| def point_in_box(point: tuple[float, float], box: Box2D, tolerance: float = 1.0) -> bool: | |
| """Is a pixel inside a box, allowing a little slack for rounding?""" | |
| u, v = point | |
| x1, y1, x2, y2 = box | |
| return ( | |
| x1 - tolerance <= u <= x2 + tolerance | |
| and y1 - tolerance <= v <= y2 + tolerance | |
| ) | |
| def quaternion_to_rotation_matrix(qw: float, qx: float, qy: float, qz: float) -> np.ndarray: | |
| """(w, x, y, z) unit quaternion -> (3, 3) rotation matrix.""" | |
| quaternion = np.array([qw, qx, qy, qz], dtype=float) | |
| quaternion /= np.linalg.norm(quaternion) | |
| w, x, y, z = quaternion | |
| return np.array([ | |
| [1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y)], | |
| [2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x)], | |
| [2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y)], | |
| ]) | |
| def cuboid_corners(center: np.ndarray, rotation: np.ndarray, | |
| length: float, width: float, height: float) -> np.ndarray: | |
| """Eight corners of a 3D box, in the frame `center` and `rotation` are given in. | |
| The object frame is +X along length, +Y along width, +Z along height, with | |
| the origin at the cuboid centre -- the convention used by both nuScenes and | |
| Argoverse 2 cuboid records. | |
| Returns (3, 8). | |
| """ | |
| half = np.array([length, width, height], dtype=float) / 2.0 | |
| signs = np.array([ | |
| [+1, +1, +1], [+1, +1, -1], [+1, -1, +1], [+1, -1, -1], | |
| [-1, +1, +1], [-1, +1, -1], [-1, -1, +1], [-1, -1, -1], | |
| ], dtype=float) | |
| corners_object = (signs * half).T # (3, 8) | |
| return rotation @ corners_object + np.asarray(center, dtype=float).reshape(3, 1) | |