forgebench: add missing local modules (RVG infer.py, gt_loader, faithfulness), MV-SAM3D diff, pinned commits
e4e030b verified Download forgebench/code/eval/gt_loader.py from Ronaldo-GOAT/bert_simpson: direct link, hf CLI and curl.
- Browser
- Download file 8.86 kB
-
https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/eval/gt_loader.py
- Command line
-
hf download hf://Ronaldo-GOAT/bert_simpson/forgebench/code/eval/gt_loader.py
-
curl -L -o gt_loader.py https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/eval/gt_loader.py
8.86 kB
| """Build GT voxel grids for one (object, cam, frame) selection. | |
| Pipeline (all in the OBJECT's canonical frame — free alignment from GT pose): | |
| depth px (visible-mask bit) --K,c2w--> world (m) | |
| --T_world_obj^-1--> object/model frame | |
| GLB verts --glb_to_model--> object/model frame | |
| canonicalize with the GLB's model-frame bbox (oracle center/scale) | |
| voxelize 64^3: gt_vis (depth points), gt_full (GLB surface samples) | |
| free space: ray-carve from the camera position (same canonical coords) | |
| Usage: | |
| from metrics.gt_loader import load_gt | |
| g = load_gt(clip, sel_entry) # sel_entry = one dict from selection.json | |
| g.gt_vis, g.gt_full, g.free, g.center, g.scale, g.cam_pos_canon | |
| Self-test against real data: | |
| python metrics/gt_loader.py (uses selection.json from .debug/exp_faithfulness) | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from dataclasses import dataclass | |
| from pathlib import Path | |
| import imageio.v3 as iio | |
| import numpy as np | |
| import trimesh | |
| from scipy import ndimage | |
| from faithfulness import canonicalize, voxelize_points, evaluate_faithfulness | |
| DEPTH_UNIT = 1e-3 # uint16 png is millimeters (verified vs GT poses) | |
| def _load_json(p: Path): | |
| return json.loads(Path(p).read_text()) | |
| def _camera(clip: Path, name: str) -> dict: | |
| cams = _load_json(clip / "cameras.json")["cameras"] | |
| return next(c for c in cams if c["name"] == name) | |
| def _pose(clip: Path, frame: int, obj: str) -> np.ndarray: | |
| poses = _load_json(clip / "poses.json") | |
| fr = poses["frames"][frame] | |
| assert fr["f"] == frame | |
| T = next(p["T_world_obj"] for p in fr["poses"] if p["name"] == obj) | |
| return np.array(T, np.float64), np.array(poses["glb_to_model"], np.float64) | |
| def _glb_model_frame(glb_path: str, glb_to_model: np.ndarray) -> trimesh.Trimesh: | |
| """Load GLB, merge geometry, map into the model (object) frame.""" | |
| scene = trimesh.load(glb_path, force="scene", process=False) | |
| mesh = scene.to_mesh() if hasattr(scene, "to_mesh") else scene | |
| mesh = trimesh.Trimesh(vertices=np.asarray(mesh.vertices), | |
| faces=np.asarray(mesh.faces), process=False) | |
| mesh.apply_transform(glb_to_model) | |
| return mesh | |
| def carve_free_space_depth(depth: np.ndarray, K: dict, c2w: np.ndarray, | |
| T_world_obj: np.ndarray, center: np.ndarray, | |
| scale: float, gt_vis: np.ndarray, n: int, | |
| margin_vox: float = 3.0, | |
| min_filter_px: int = 9) -> np.ndarray: | |
| """Exact free-space carving from the full scene depth map. | |
| A voxel is proven empty iff its center projects inside the image and its | |
| camera-space depth is at least `margin_vox` voxels IN FRONT of the | |
| observed scene depth at that pixel. Unlike per-ray marching this cannot | |
| leak behind foreground silhouette edges (a hidden voxel there projects | |
| onto the foreground pixel, whose depth is closer -> not free), and it | |
| covers the whole frustum, not just corridors to observed voxels. | |
| """ | |
| ii = (np.arange(n) + 0.5) / n - 0.5 | |
| gx, gy, gz = np.meshgrid(ii, ii, ii, indexing="ij") | |
| pc_model = np.stack([gx, gy, gz], -1).reshape(-1, 3) * scale + center | |
| T = np.linalg.inv(c2w) @ T_world_obj # model -> camera | |
| p_cam = pc_model @ T[:3, :3].T + T[:3, 3] | |
| z = p_cam[:, 2] | |
| h, w = depth.shape | |
| with np.errstate(divide="ignore", invalid="ignore"): | |
| u = np.round(p_cam[:, 0] / z * K["fx"] + K["cx"]).astype(np.int64) | |
| v = np.round(p_cam[:, 1] / z * K["fy"] + K["cy"]).astype(np.int64) | |
| ok = (z > 0) & (u >= 0) & (u < w) & (v >= 0) & (v < h) | |
| # conservative depth: minimum filter, so voxels projecting right next | |
| # to a silhouette edge (where the pixel sees the far background) are not | |
| # carved — hidden surface continues just behind the contour. 9px + 3 vox | |
| # margin leaves 3 / 738k false-free voxels across the 6 test objects. | |
| depth_min = ndimage.minimum_filter(depth, size=min_filter_px) | |
| dmap = np.zeros(len(z)) | |
| dmap[ok] = depth_min[v[ok], u[ok]] * DEPTH_UNIT | |
| margin_m = margin_vox * scale / n | |
| free = ok & (dmap > 0) & (z < dmap - margin_m) | |
| free = free.reshape(n, n, n) | |
| # never mark observed surface (or its 1-voxel shell) as free | |
| free &= ~ndimage.maximum_filter(gt_vis, size=3) | |
| return free | |
| class GTGrids: | |
| gt_vis: np.ndarray # (n,n,n) bool, from depth px | |
| gt_full: np.ndarray # (n,n,n) bool, from GLB surface | |
| free: np.ndarray # (n,n,n) bool, ray-carved | |
| center: np.ndarray # canonicalization center (model frame, m) | |
| scale: float # canonicalization scale (m) | |
| cam_pos_canon: np.ndarray # camera position in canonical coords | |
| points_canon: np.ndarray # depth points, canonical coords (for debug) | |
| kept_frac: float # fraction of depth points inside box (+tol) | |
| mesh_canon: trimesh.Trimesh # GLB mesh in canonical coords (for renders) | |
| def load_gt(clip: Path, sel: dict, n: int = 64, | |
| mesh_samples: int = 1_000_000, mask_erode: int = 2) -> GTGrids: | |
| clip = Path(clip) | |
| cam = _camera(clip, sel["cam"]) | |
| K = cam["intrinsics"] | |
| c2w = np.array(cam["c2w"], np.float64) | |
| T_world_obj, glb_to_model = _pose(clip, sel["frame"], sel["object"]) | |
| # depth px of the object -> world -> object/model frame | |
| d = iio.imread(clip / "depth" / sel["cam"] / f"{sel['frame']:04d}.png") | |
| m = iio.imread(clip / "mask" / sel["cam"] / f"{sel['frame']:04d}.png") | |
| obj_px = ((m >> sel["bit"]) & 1).astype(bool) & (d > 0) | |
| if mask_erode: | |
| # silhouette-edge px mix object and background depth; drop them | |
| from scipy import ndimage | |
| obj_px &= ndimage.binary_erosion(obj_px, iterations=mask_erode) | |
| ys, xs = np.nonzero(obj_px) | |
| z = d[ys, xs].astype(np.float64) * DEPTH_UNIT | |
| pc = np.stack([(xs - K["cx"]) / K["fx"] * z, | |
| (ys - K["cy"]) / K["fy"] * z, | |
| z, np.ones_like(z)], 1) | |
| p_world = (c2w @ pc.T).T | |
| T_obj_world = np.linalg.inv(T_world_obj) | |
| p_model = (T_obj_world @ p_world.T).T[:, :3] | |
| # GLB in model frame; oracle canonicalization from its bbox | |
| mesh = _glb_model_frame(sel["glb"], glb_to_model) | |
| _, center, scale = canonicalize(mesh.vertices) | |
| # seeded so every run scores against the identical GT grid | |
| surf, _ = trimesh.sample.sample_surface(mesh, mesh_samples, seed=0) | |
| surf_c, _, _ = canonicalize(surf, center, scale) | |
| pts_c, _, _ = canonicalize(p_model, center, scale) | |
| # depth is quantized to 1 mm: points on a bbox face can fall epsilon | |
| # outside [-0.5, 0.5]. Clamp near-misses onto the box, drop true outliers. | |
| # 0.012 ~ 0.77 voxel at n=64: sub-voxel, so a clamped point stays in the | |
| # correct boundary voxel. Measured worst real overshoot 0.0105 (glazed | |
| # squirrel, z-only mm-scale render/depth bias); everything else < 0.005. | |
| tol = 0.012 | |
| near = (np.abs(pts_c) <= 0.5 + tol).all(1) | |
| pts_c = np.clip(pts_c[near], -0.5, 0.5 - 1e-9) | |
| gt_full = voxelize_points(surf_c, n) | |
| gt_vis = voxelize_points(pts_c, n) | |
| assert gt_vis.sum() > 0, ( | |
| f"{sel['object']}: empty gt_vis (mask erosion wiped a tiny object?)") | |
| cam_world = c2w[:3, 3] | |
| cam_model = (T_obj_world @ np.append(cam_world, 1.0))[:3] | |
| cam_canon = (cam_model - center) / scale | |
| free = carve_free_space_depth(d, K, c2w, T_world_obj, center, scale, | |
| gt_vis, n) | |
| mesh_c = mesh.copy() | |
| mesh_c.apply_translation(-center) | |
| mesh_c.apply_scale(1.0 / scale) | |
| return GTGrids(gt_vis, gt_full, free, center, float(scale), | |
| cam_canon, pts_c, float(near.mean()), mesh_c) | |
| # ---------------------------------------------------------------- self-test | |
| def _self_test(): | |
| root = Path(__file__).resolve().parent.parent / ".debug" / "exp_faithfulness" | |
| sel = _load_json(root / "selection.json") | |
| clip = Path(sel["clip"]) | |
| ok = True | |
| for s in sel["selections"]: | |
| g = load_gt(clip, s) | |
| nv, nf = int(g.gt_vis.sum()), int(g.gt_full.sum()) | |
| # depth points must lie ON the GLB surface: gt_vis vs gt_full recall@1 | |
| rep = evaluate_faithfulness(g.gt_full, g.gt_vis, radii=(0, 1, 2)) | |
| # fraction of gt_vis voxels within 1 voxel of gt_full | |
| r1 = rep.f1[1]["precision"] # pred=gt_vis voxels near gt_full | |
| inb = g.kept_frac | |
| vis_frac = nv / nf | |
| line = (f"{s['object']:32s} vis={nv:5d} full={nf:6d} " | |
| f"vis/full={vis_frac:.2f} onSurf@1={r1:.3f} inbox={inb:.3f} " | |
| f"free={int(g.free.sum())}") | |
| good = r1 > 0.95 and inb > 0.99 and 0.05 < vis_frac < 1.0 | |
| print(("PASS " if good else "FAIL ") + line) | |
| ok &= good | |
| assert ok, "gt_loader self-test failed" | |
| print("ALL GT-LOADER SELF-TESTS PASSED") | |
| if __name__ == "__main__": | |
| _self_test() | |