"""Build GT voxel grids for one (object, cam, frame) selection. Pipeline (all in the OBJECT's canonical frame — free alignment from GT pose): depth px (visible-mask bit) --K,c2w--> world (m) --T_world_obj^-1--> object/model frame GLB verts --glb_to_model--> object/model frame canonicalize with the GLB's model-frame bbox (oracle center/scale) voxelize 64^3: gt_vis (depth points), gt_full (GLB surface samples) free space: ray-carve from the camera position (same canonical coords) Usage: from metrics.gt_loader import load_gt g = load_gt(clip, sel_entry) # sel_entry = one dict from selection.json g.gt_vis, g.gt_full, g.free, g.center, g.scale, g.cam_pos_canon Self-test against real data: python metrics/gt_loader.py (uses selection.json from .debug/exp_faithfulness) """ from __future__ import annotations import json from dataclasses import dataclass from pathlib import Path import imageio.v3 as iio import numpy as np import trimesh from scipy import ndimage from faithfulness import canonicalize, voxelize_points, evaluate_faithfulness DEPTH_UNIT = 1e-3 # uint16 png is millimeters (verified vs GT poses) def _load_json(p: Path): return json.loads(Path(p).read_text()) def _camera(clip: Path, name: str) -> dict: cams = _load_json(clip / "cameras.json")["cameras"] return next(c for c in cams if c["name"] == name) def _pose(clip: Path, frame: int, obj: str) -> np.ndarray: poses = _load_json(clip / "poses.json") fr = poses["frames"][frame] assert fr["f"] == frame T = next(p["T_world_obj"] for p in fr["poses"] if p["name"] == obj) return np.array(T, np.float64), np.array(poses["glb_to_model"], np.float64) def _glb_model_frame(glb_path: str, glb_to_model: np.ndarray) -> trimesh.Trimesh: """Load GLB, merge geometry, map into the model (object) frame.""" scene = trimesh.load(glb_path, force="scene", process=False) mesh = scene.to_mesh() if hasattr(scene, "to_mesh") else scene mesh = trimesh.Trimesh(vertices=np.asarray(mesh.vertices), faces=np.asarray(mesh.faces), process=False) mesh.apply_transform(glb_to_model) return mesh def carve_free_space_depth(depth: np.ndarray, K: dict, c2w: np.ndarray, T_world_obj: np.ndarray, center: np.ndarray, scale: float, gt_vis: np.ndarray, n: int, margin_vox: float = 3.0, min_filter_px: int = 9) -> np.ndarray: """Exact free-space carving from the full scene depth map. A voxel is proven empty iff its center projects inside the image and its camera-space depth is at least `margin_vox` voxels IN FRONT of the observed scene depth at that pixel. Unlike per-ray marching this cannot leak behind foreground silhouette edges (a hidden voxel there projects onto the foreground pixel, whose depth is closer -> not free), and it covers the whole frustum, not just corridors to observed voxels. """ ii = (np.arange(n) + 0.5) / n - 0.5 gx, gy, gz = np.meshgrid(ii, ii, ii, indexing="ij") pc_model = np.stack([gx, gy, gz], -1).reshape(-1, 3) * scale + center T = np.linalg.inv(c2w) @ T_world_obj # model -> camera p_cam = pc_model @ T[:3, :3].T + T[:3, 3] z = p_cam[:, 2] h, w = depth.shape with np.errstate(divide="ignore", invalid="ignore"): u = np.round(p_cam[:, 0] / z * K["fx"] + K["cx"]).astype(np.int64) v = np.round(p_cam[:, 1] / z * K["fy"] + K["cy"]).astype(np.int64) ok = (z > 0) & (u >= 0) & (u < w) & (v >= 0) & (v < h) # conservative depth: minimum filter, so voxels projecting right next # to a silhouette edge (where the pixel sees the far background) are not # carved — hidden surface continues just behind the contour. 9px + 3 vox # margin leaves 3 / 738k false-free voxels across the 6 test objects. depth_min = ndimage.minimum_filter(depth, size=min_filter_px) dmap = np.zeros(len(z)) dmap[ok] = depth_min[v[ok], u[ok]] * DEPTH_UNIT margin_m = margin_vox * scale / n free = ok & (dmap > 0) & (z < dmap - margin_m) free = free.reshape(n, n, n) # never mark observed surface (or its 1-voxel shell) as free free &= ~ndimage.maximum_filter(gt_vis, size=3) return free @dataclass class GTGrids: gt_vis: np.ndarray # (n,n,n) bool, from depth px gt_full: np.ndarray # (n,n,n) bool, from GLB surface free: np.ndarray # (n,n,n) bool, ray-carved center: np.ndarray # canonicalization center (model frame, m) scale: float # canonicalization scale (m) cam_pos_canon: np.ndarray # camera position in canonical coords points_canon: np.ndarray # depth points, canonical coords (for debug) kept_frac: float # fraction of depth points inside box (+tol) mesh_canon: trimesh.Trimesh # GLB mesh in canonical coords (for renders) def load_gt(clip: Path, sel: dict, n: int = 64, mesh_samples: int = 1_000_000, mask_erode: int = 2) -> GTGrids: clip = Path(clip) cam = _camera(clip, sel["cam"]) K = cam["intrinsics"] c2w = np.array(cam["c2w"], np.float64) T_world_obj, glb_to_model = _pose(clip, sel["frame"], sel["object"]) # depth px of the object -> world -> object/model frame d = iio.imread(clip / "depth" / sel["cam"] / f"{sel['frame']:04d}.png") m = iio.imread(clip / "mask" / sel["cam"] / f"{sel['frame']:04d}.png") obj_px = ((m >> sel["bit"]) & 1).astype(bool) & (d > 0) if mask_erode: # silhouette-edge px mix object and background depth; drop them from scipy import ndimage obj_px &= ndimage.binary_erosion(obj_px, iterations=mask_erode) ys, xs = np.nonzero(obj_px) z = d[ys, xs].astype(np.float64) * DEPTH_UNIT pc = np.stack([(xs - K["cx"]) / K["fx"] * z, (ys - K["cy"]) / K["fy"] * z, z, np.ones_like(z)], 1) p_world = (c2w @ pc.T).T T_obj_world = np.linalg.inv(T_world_obj) p_model = (T_obj_world @ p_world.T).T[:, :3] # GLB in model frame; oracle canonicalization from its bbox mesh = _glb_model_frame(sel["glb"], glb_to_model) _, center, scale = canonicalize(mesh.vertices) # seeded so every run scores against the identical GT grid surf, _ = trimesh.sample.sample_surface(mesh, mesh_samples, seed=0) surf_c, _, _ = canonicalize(surf, center, scale) pts_c, _, _ = canonicalize(p_model, center, scale) # depth is quantized to 1 mm: points on a bbox face can fall epsilon # outside [-0.5, 0.5]. Clamp near-misses onto the box, drop true outliers. # 0.012 ~ 0.77 voxel at n=64: sub-voxel, so a clamped point stays in the # correct boundary voxel. Measured worst real overshoot 0.0105 (glazed # squirrel, z-only mm-scale render/depth bias); everything else < 0.005. tol = 0.012 near = (np.abs(pts_c) <= 0.5 + tol).all(1) pts_c = np.clip(pts_c[near], -0.5, 0.5 - 1e-9) gt_full = voxelize_points(surf_c, n) gt_vis = voxelize_points(pts_c, n) assert gt_vis.sum() > 0, ( f"{sel['object']}: empty gt_vis (mask erosion wiped a tiny object?)") cam_world = c2w[:3, 3] cam_model = (T_obj_world @ np.append(cam_world, 1.0))[:3] cam_canon = (cam_model - center) / scale free = carve_free_space_depth(d, K, c2w, T_world_obj, center, scale, gt_vis, n) mesh_c = mesh.copy() mesh_c.apply_translation(-center) mesh_c.apply_scale(1.0 / scale) return GTGrids(gt_vis, gt_full, free, center, float(scale), cam_canon, pts_c, float(near.mean()), mesh_c) # ---------------------------------------------------------------- self-test def _self_test(): root = Path(__file__).resolve().parent.parent / ".debug" / "exp_faithfulness" sel = _load_json(root / "selection.json") clip = Path(sel["clip"]) ok = True for s in sel["selections"]: g = load_gt(clip, s) nv, nf = int(g.gt_vis.sum()), int(g.gt_full.sum()) # depth points must lie ON the GLB surface: gt_vis vs gt_full recall@1 rep = evaluate_faithfulness(g.gt_full, g.gt_vis, radii=(0, 1, 2)) # fraction of gt_vis voxels within 1 voxel of gt_full r1 = rep.f1[1]["precision"] # pred=gt_vis voxels near gt_full inb = g.kept_frac vis_frac = nv / nf line = (f"{s['object']:32s} vis={nv:5d} full={nf:6d} " f"vis/full={vis_frac:.2f} onSurf@1={r1:.3f} inbox={inb:.3f} " f"free={int(g.free.sum())}") good = r1 > 0.95 and inb > 0.99 and 0.05 < vis_frac < 1.0 print(("PASS " if good else "FAIL ") + line) ok &= good assert ok, "gt_loader self-test failed" print("ALL GT-LOADER SELF-TESTS PASSED") if __name__ == "__main__": _self_test()