FORGE3DBench: final eval protocol + eval_final.py, batched Ours inference, held-out view tars, missing-object lists, README
cca6827 verified Download forgebench/code/eval/appeval/render.py from Ronaldo-GOAT/bert_simpson: direct link, hf CLI and curl.
- Browser
- Download file 12 kB
-
https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/eval/appeval/render.py
- Command line
-
hf download hf://Ronaldo-GOAT/bert_simpson/forgebench/code/eval/appeval/render.py
-
curl -L -o render.py https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/eval/appeval/render.py
12 kB
| """Unlit-albedo mesh renderer for the appearance-eval harness (nvdiffrast). | |
| LOCKED render rig (identical for pred and GT so no method is advantaged): | |
| * Renderer : nvdiffrast RasterizeCudaContext (headless, GPU). | |
| * Shading : WORLD-FIXED UNLIT ALBEDO. We rasterize the base colour / | |
| texture with NO lighting term (flat emission). SLAT-baked | |
| textures are albedo-like, so unlit is the fair choice and it | |
| is byte-for-byte identical between pred and GT. | |
| * Output : RGBA float32 in [0,1]; alpha == object mask (1 inside the | |
| silhouette, 0 outside). | |
| * Novel views: 24 = 8 azimuth (0..315 step 45) x 3 elevation {-30,0,+30}, | |
| look-at origin, radius ~2.6, vertical fov 40 deg, 512x512, | |
| world-up = +Z (toys4k canonical up). | |
| * Input view : render_input_view() reproduces a SPECIFIC OpenCV camera | |
| (K + c2w_cv exactly as stored in the exp .npz files) so the | |
| render lines up pixel-for-pixel with the input photo/crop. | |
| Mesh colour handling (common fragment-level representation): | |
| * TextureVisuals (UV + PBR baseColorTexture) -> UV interpolated, texture | |
| sampled per fragment (native full-res, nothing baked down). | |
| * ColorVisuals (per-vertex RGBA) -> vertex colour interpolated. | |
| * flat / material-only meshes -> constant base colour. | |
| Every mesh therefore reduces to "an unlit RGB per fragment", which is the | |
| single common representation the spec asks for. | |
| Camera conventions (must match metrics/evaluate_synth.py): | |
| c2w_cv is an OpenCV camera-to-world matrix (x right, y down, z forward into | |
| the scene). evaluate_synth back-projects depth with exactly this convention; | |
| render_input_view() inverts it and composes an OpenGL projection so the two | |
| agree. This is verified in selfcheck.py by overlapping the rendered alpha | |
| with the stored depth>0 mask (mask IoU must be high). | |
| """ | |
| from __future__ import annotations | |
| # FINAL: = metrics/appeval/render.py + render_input_view far-clip fix (see docstring). | |
| import numpy as np | |
| import torch | |
| import trimesh | |
| import nvdiffrast.torch as dr | |
| # ---------------------------------------------------------------------------- | |
| # global context (one CUDA raster context per process) | |
| # ---------------------------------------------------------------------------- | |
| _GLCTX = None | |
| def get_ctx(): | |
| global _GLCTX | |
| if _GLCTX is None: | |
| _GLCTX = dr.RasterizeCudaContext() | |
| return _GLCTX | |
| # ---------------------------------------------------------------------------- | |
| # mesh preparation -> GPU tensors + a per-fragment colour source | |
| # ---------------------------------------------------------------------------- | |
| class MeshGL: | |
| """A mesh prepared for nvdiffrast: verts, faces, and a colour source. | |
| colour source is exactly one of: | |
| mode == 'vertex' : self.vcol (V,3) float in [0,1] | |
| mode == 'uv' : self.uv (V,2), self.tex (Ht,Wt,3) float in [0,1] | |
| mode == 'flat' : self.flat (3,) float in [0,1] | |
| """ | |
| def __init__(self, verts, faces, device="cuda"): | |
| self.device = device | |
| self.verts = torch.as_tensor(verts, dtype=torch.float32, device=device) | |
| self.faces = torch.as_tensor(faces, dtype=torch.int32, device=device) | |
| self.mode = "flat" | |
| self.flat = torch.tensor([0.6, 0.6, 0.6], dtype=torch.float32, device=device) | |
| self.vcol = None | |
| self.uv = None | |
| self.tex = None | |
| def _extract_texture_image(mat): | |
| """Return an (H,W,3) float[0,1] array from a trimesh material, or None.""" | |
| img = None | |
| for attr in ("baseColorTexture", "image"): | |
| cand = getattr(mat, attr, None) | |
| if cand is not None: | |
| img = cand | |
| break | |
| if img is None: | |
| return None | |
| arr = np.asarray(img) | |
| if arr.ndim == 2: # grayscale | |
| arr = np.stack([arr] * 3, -1) | |
| if arr.shape[-1] == 4: | |
| arr = arr[..., :3] | |
| return arr.astype(np.float32) / 255.0 | |
| def prepare_mesh(mesh: trimesh.Trimesh, device="cuda") -> MeshGL: | |
| """Convert a trimesh mesh into a MeshGL with the right colour source.""" | |
| if not isinstance(mesh, trimesh.Trimesh): | |
| mesh = mesh.dump(concatenate=True) if hasattr(mesh, "dump") else \ | |
| trimesh.util.concatenate(mesh) | |
| g = MeshGL(np.asarray(mesh.vertices), np.asarray(mesh.faces), device) | |
| vis = mesh.visual | |
| # --- UV / textured path --- | |
| uv = getattr(vis, "uv", None) | |
| tex = None | |
| if uv is not None: | |
| mat = getattr(vis, "material", None) | |
| if mat is not None: | |
| tex = _extract_texture_image(mat) | |
| if uv is not None and tex is not None and len(uv) == len(mesh.vertices): | |
| g.mode = "uv" | |
| # trimesh's glTF loader already flips V to bottom-left origin, but dr.texture | |
| # indexes the (unflipped) texture array top-left -> flip V back so the texture | |
| # is sampled with the correct orientation (was rendering UV meshes upside-down). | |
| uv_arr = np.asarray(uv, dtype=np.float32).copy() | |
| uv_arr[:, 1] = 1.0 - uv_arr[:, 1] | |
| g.uv = torch.as_tensor(uv_arr, dtype=torch.float32, device=device) | |
| g.tex = torch.as_tensor(tex, dtype=torch.float32, device=device) | |
| return g | |
| # --- textured but no image: use flat baseColorFactor if any --- | |
| if uv is not None: | |
| mat = getattr(vis, "material", None) | |
| base = getattr(mat, "baseColorFactor", None) if mat is not None else None | |
| if base is not None: | |
| g.mode = "flat" | |
| g.flat = torch.as_tensor(np.asarray(base)[:3] / (255.0 if np.max(base) > 1.5 else 1.0), | |
| dtype=torch.float32, device=device) | |
| return g | |
| # --- vertex colour path --- | |
| vc = getattr(vis, "vertex_colors", None) | |
| if vc is not None and len(vc) == len(mesh.vertices): | |
| vc = np.asarray(vc)[:, :3].astype(np.float32) / 255.0 | |
| g.mode = "vertex" | |
| g.vcol = torch.as_tensor(vc, dtype=torch.float32, device=device) | |
| return g | |
| # --- fallback: convert whatever we have to per-vertex colour --- | |
| try: | |
| vc = np.asarray(vis.to_color().vertex_colors)[:, :3].astype(np.float32) / 255.0 | |
| g.mode = "vertex" | |
| g.vcol = torch.as_tensor(vc, dtype=torch.float32, device=device) | |
| except Exception: | |
| pass # keep flat grey | |
| return g | |
| # ---------------------------------------------------------------------------- | |
| # camera matrices | |
| # ---------------------------------------------------------------------------- | |
| def _normalize(v): | |
| return v / (np.linalg.norm(v) + 1e-12) | |
| def look_at(eye, at, up): | |
| """world->camera in OpenGL convention (camera looks down -Z, +Y up).""" | |
| eye = np.asarray(eye, float) | |
| at = np.asarray(at, float) | |
| up = np.asarray(up, float) | |
| f = _normalize(at - eye) # forward | |
| s = _normalize(np.cross(f, up)) # right | |
| u = np.cross(s, f) # true up | |
| V = np.eye(4) | |
| V[0, :3] = s | |
| V[1, :3] = u | |
| V[2, :3] = -f | |
| V[0, 3] = -s @ eye | |
| V[1, 3] = -u @ eye | |
| V[2, 3] = f @ eye | |
| return V | |
| def gl_perspective(fovy_deg, aspect, near, far): | |
| t = np.tan(np.radians(fovy_deg) / 2.0) | |
| P = np.zeros((4, 4)) | |
| P[0, 0] = 1.0 / (aspect * t) | |
| P[1, 1] = 1.0 / t | |
| P[2, 2] = -(far + near) / (far - near) | |
| P[2, 3] = -2.0 * far * near / (far - near) | |
| P[3, 2] = -1.0 | |
| return P | |
| def gl_proj_from_K(fx, fy, cx, cy, W, H, near, far): | |
| """OpenGL projection from OpenCV intrinsics. | |
| Intended to be applied AFTER transforming vertices into an OpenGL camera | |
| frame (see cv_extrinsic_to_gl). Principal-point offsets follow the OpenCV | |
| top-left origin; the y sign is handled by the extrinsic flip. | |
| """ | |
| P = np.zeros((4, 4)) | |
| P[0, 0] = 2.0 * fx / W | |
| P[1, 1] = 2.0 * fy / H | |
| P[0, 2] = 1.0 - 2.0 * cx / W | |
| P[1, 2] = 2.0 * cy / H - 1.0 | |
| P[2, 2] = -(far + near) / (far - near) | |
| P[2, 3] = -2.0 * far * near / (far - near) | |
| P[3, 2] = -1.0 | |
| return P | |
| # OpenCV cam (x right, y down, z forward) -> OpenGL cam (x right, y up, z back) | |
| _CV2GL = np.diag([1.0, -1.0, -1.0, 1.0]) | |
| def cv_extrinsic_to_gl(c2w_cv): | |
| """world->OpenGL-camera matrix from an OpenCV camera-to-world matrix.""" | |
| w2c_cv = np.linalg.inv(np.asarray(c2w_cv, float)) | |
| return _CV2GL @ w2c_cv | |
| def orbit_cameras(radius=2.6, elevs=(-30, 0, 30), | |
| azims=range(0, 360, 45), up=(0, 0, 1)): | |
| """Return list of dicts {name, eye, view} for the 24-view rig. | |
| Azimuth is measured in the world XY plane; elevation lifts along +Z. | |
| az=0 places the camera on -Y looking toward +Y (matches toys4k 'front'). | |
| """ | |
| up = np.asarray(up, float) | |
| cams = [] | |
| for el in elevs: | |
| for az in azims: | |
| ar = np.radians(az) | |
| er = np.radians(el) | |
| x = radius * np.cos(er) * np.sin(ar) | |
| y = -radius * np.cos(er) * np.cos(ar) | |
| z = radius * np.sin(er) | |
| eye = np.array([x, y, z]) | |
| cams.append({ | |
| "name": f"az{az:03d}_el{el:+03d}", | |
| "eye": eye, | |
| "view": look_at(eye, (0, 0, 0), up), | |
| }) | |
| return cams | |
| # ---------------------------------------------------------------------------- | |
| # core render | |
| # ---------------------------------------------------------------------------- | |
| def _render_mvp(g: MeshGL, mvp, H, W, ctx=None): | |
| """Rasterize mesh g under a 4x4 clip transform. Returns RGBA (H,W,4) [0,1]. | |
| The output is oriented conventionally (row 0 = top of image). | |
| """ | |
| ctx = ctx or get_ctx() | |
| device = g.verts.device | |
| mvp_t = torch.as_tensor(mvp, dtype=torch.float32, device=device) | |
| vh = torch.cat([g.verts, torch.ones(len(g.verts), 1, device=device)], 1) | |
| clip = (mvp_t @ vh.T).T.contiguous()[None] # (1,V,4) | |
| rast, _ = dr.rasterize(ctx, clip, g.faces, (H, W)) | |
| alpha = (rast[..., 3:4] > 0).float() # (1,H,W,1) | |
| if g.mode == "uv": | |
| uv_i, _ = dr.interpolate(g.uv[None], rast, g.faces) | |
| tex = g.tex[None] # (1,Ht,Wt,3) | |
| col = dr.texture(tex, uv_i, filter_mode="linear") # (1,H,W,3) | |
| elif g.mode == "vertex": | |
| col, _ = dr.interpolate(g.vcol[None], rast, g.faces) | |
| else: | |
| col = g.flat[None, None, None, :].expand(1, H, W, 3) | |
| col = col * alpha # zero the background | |
| col = dr.antialias(col.contiguous(), rast, clip, g.faces) | |
| alpha = dr.antialias(alpha.contiguous(), rast, clip, g.faces) | |
| rgba = torch.cat([col, alpha], -1)[0].clamp(0, 1) # (H,W,4) | |
| rgba = torch.flip(rgba, dims=[0]) # GL bottom-up -> top-down | |
| return rgba | |
| def render_orbit(g: MeshGL, cams, H=512, W=512, fovy=40.0, near=0.05, far=20.0, | |
| ctx=None): | |
| """Render mesh g from a list of orbit cameras. Returns (N,H,W,4) tensor.""" | |
| P = gl_perspective(fovy, W / H, near, far) | |
| out = [] | |
| for cam in cams: | |
| mvp = P @ cam["view"] | |
| out.append(_render_mvp(g, mvp, H, W, ctx=ctx)) | |
| return torch.stack(out, 0) | |
| def render_input_view(g: MeshGL, K, c2w_cv, H, W, near=0.05, far=None, ctx=None): | |
| """Render mesh g from a SPECIFIC OpenCV camera (K dict + c2w_cv 4x4). | |
| K = {'fx','fy','cx','cy'} in pixels at resolution (H,W). Returns (H,W,4). | |
| FAR-CLIP FIX (2026-09-25): far was a fixed 20.0 canonical units; FB150 | |
| tiny objects have cameras 19.2-21.6 units from the origin, so the whole | |
| object was clipped (empty render). far now defaults to | |
| max(20, |camera centre| + 10): unchanged (20) for every camera within 10 | |
| units (all Toys/Omni, most FB150); only the depth mapping, never the | |
| visible surface, changes otherwise. | |
| """ | |
| if far is None: | |
| far = max(20.0, float(np.linalg.norm(np.asarray(c2w_cv, float)[:3, 3])) + 10.0) | |
| P = gl_proj_from_K(K["fx"], K["fy"], K["cx"], K["cy"], W, H, near, far) | |
| Vgl = cv_extrinsic_to_gl(c2w_cv) | |
| mvp = P @ Vgl | |
| return _render_mvp(g, mvp, H, W, ctx=ctx) | |