Ronaldo-GOAT's picture
FORGE3DBench: final eval protocol + eval_final.py, batched Ours inference, held-out view tars, missing-object lists, README
cca6827 verified
Raw History Blame Contribute Delete
12 kB
"""Unlit-albedo mesh renderer for the appearance-eval harness (nvdiffrast).
LOCKED render rig (identical for pred and GT so no method is advantaged):
* Renderer : nvdiffrast RasterizeCudaContext (headless, GPU).
* Shading : WORLD-FIXED UNLIT ALBEDO. We rasterize the base colour /
texture with NO lighting term (flat emission). SLAT-baked
textures are albedo-like, so unlit is the fair choice and it
is byte-for-byte identical between pred and GT.
* Output : RGBA float32 in [0,1]; alpha == object mask (1 inside the
silhouette, 0 outside).
* Novel views: 24 = 8 azimuth (0..315 step 45) x 3 elevation {-30,0,+30},
look-at origin, radius ~2.6, vertical fov 40 deg, 512x512,
world-up = +Z (toys4k canonical up).
* Input view : render_input_view() reproduces a SPECIFIC OpenCV camera
(K + c2w_cv exactly as stored in the exp .npz files) so the
render lines up pixel-for-pixel with the input photo/crop.
Mesh colour handling (common fragment-level representation):
* TextureVisuals (UV + PBR baseColorTexture) -> UV interpolated, texture
sampled per fragment (native full-res, nothing baked down).
* ColorVisuals (per-vertex RGBA) -> vertex colour interpolated.
* flat / material-only meshes -> constant base colour.
Every mesh therefore reduces to "an unlit RGB per fragment", which is the
single common representation the spec asks for.
Camera conventions (must match metrics/evaluate_synth.py):
c2w_cv is an OpenCV camera-to-world matrix (x right, y down, z forward into
the scene). evaluate_synth back-projects depth with exactly this convention;
render_input_view() inverts it and composes an OpenGL projection so the two
agree. This is verified in selfcheck.py by overlapping the rendered alpha
with the stored depth>0 mask (mask IoU must be high).
"""
from __future__ import annotations
# FINAL: = metrics/appeval/render.py + render_input_view far-clip fix (see docstring).
import numpy as np
import torch
import trimesh
import nvdiffrast.torch as dr
# ----------------------------------------------------------------------------
# global context (one CUDA raster context per process)
# ----------------------------------------------------------------------------
_GLCTX = None
def get_ctx():
global _GLCTX
if _GLCTX is None:
_GLCTX = dr.RasterizeCudaContext()
return _GLCTX
# ----------------------------------------------------------------------------
# mesh preparation -> GPU tensors + a per-fragment colour source
# ----------------------------------------------------------------------------
class MeshGL:
"""A mesh prepared for nvdiffrast: verts, faces, and a colour source.
colour source is exactly one of:
mode == 'vertex' : self.vcol (V,3) float in [0,1]
mode == 'uv' : self.uv (V,2), self.tex (Ht,Wt,3) float in [0,1]
mode == 'flat' : self.flat (3,) float in [0,1]
"""
def __init__(self, verts, faces, device="cuda"):
self.device = device
self.verts = torch.as_tensor(verts, dtype=torch.float32, device=device)
self.faces = torch.as_tensor(faces, dtype=torch.int32, device=device)
self.mode = "flat"
self.flat = torch.tensor([0.6, 0.6, 0.6], dtype=torch.float32, device=device)
self.vcol = None
self.uv = None
self.tex = None
def _extract_texture_image(mat):
"""Return an (H,W,3) float[0,1] array from a trimesh material, or None."""
img = None
for attr in ("baseColorTexture", "image"):
cand = getattr(mat, attr, None)
if cand is not None:
img = cand
break
if img is None:
return None
arr = np.asarray(img)
if arr.ndim == 2: # grayscale
arr = np.stack([arr] * 3, -1)
if arr.shape[-1] == 4:
arr = arr[..., :3]
return arr.astype(np.float32) / 255.0
def prepare_mesh(mesh: trimesh.Trimesh, device="cuda") -> MeshGL:
"""Convert a trimesh mesh into a MeshGL with the right colour source."""
if not isinstance(mesh, trimesh.Trimesh):
mesh = mesh.dump(concatenate=True) if hasattr(mesh, "dump") else \
trimesh.util.concatenate(mesh)
g = MeshGL(np.asarray(mesh.vertices), np.asarray(mesh.faces), device)
vis = mesh.visual
# --- UV / textured path ---
uv = getattr(vis, "uv", None)
tex = None
if uv is not None:
mat = getattr(vis, "material", None)
if mat is not None:
tex = _extract_texture_image(mat)
if uv is not None and tex is not None and len(uv) == len(mesh.vertices):
g.mode = "uv"
# trimesh's glTF loader already flips V to bottom-left origin, but dr.texture
# indexes the (unflipped) texture array top-left -> flip V back so the texture
# is sampled with the correct orientation (was rendering UV meshes upside-down).
uv_arr = np.asarray(uv, dtype=np.float32).copy()
uv_arr[:, 1] = 1.0 - uv_arr[:, 1]
g.uv = torch.as_tensor(uv_arr, dtype=torch.float32, device=device)
g.tex = torch.as_tensor(tex, dtype=torch.float32, device=device)
return g
# --- textured but no image: use flat baseColorFactor if any ---
if uv is not None:
mat = getattr(vis, "material", None)
base = getattr(mat, "baseColorFactor", None) if mat is not None else None
if base is not None:
g.mode = "flat"
g.flat = torch.as_tensor(np.asarray(base)[:3] / (255.0 if np.max(base) > 1.5 else 1.0),
dtype=torch.float32, device=device)
return g
# --- vertex colour path ---
vc = getattr(vis, "vertex_colors", None)
if vc is not None and len(vc) == len(mesh.vertices):
vc = np.asarray(vc)[:, :3].astype(np.float32) / 255.0
g.mode = "vertex"
g.vcol = torch.as_tensor(vc, dtype=torch.float32, device=device)
return g
# --- fallback: convert whatever we have to per-vertex colour ---
try:
vc = np.asarray(vis.to_color().vertex_colors)[:, :3].astype(np.float32) / 255.0
g.mode = "vertex"
g.vcol = torch.as_tensor(vc, dtype=torch.float32, device=device)
except Exception:
pass # keep flat grey
return g
# ----------------------------------------------------------------------------
# camera matrices
# ----------------------------------------------------------------------------
def _normalize(v):
return v / (np.linalg.norm(v) + 1e-12)
def look_at(eye, at, up):
"""world->camera in OpenGL convention (camera looks down -Z, +Y up)."""
eye = np.asarray(eye, float)
at = np.asarray(at, float)
up = np.asarray(up, float)
f = _normalize(at - eye) # forward
s = _normalize(np.cross(f, up)) # right
u = np.cross(s, f) # true up
V = np.eye(4)
V[0, :3] = s
V[1, :3] = u
V[2, :3] = -f
V[0, 3] = -s @ eye
V[1, 3] = -u @ eye
V[2, 3] = f @ eye
return V
def gl_perspective(fovy_deg, aspect, near, far):
t = np.tan(np.radians(fovy_deg) / 2.0)
P = np.zeros((4, 4))
P[0, 0] = 1.0 / (aspect * t)
P[1, 1] = 1.0 / t
P[2, 2] = -(far + near) / (far - near)
P[2, 3] = -2.0 * far * near / (far - near)
P[3, 2] = -1.0
return P
def gl_proj_from_K(fx, fy, cx, cy, W, H, near, far):
"""OpenGL projection from OpenCV intrinsics.
Intended to be applied AFTER transforming vertices into an OpenGL camera
frame (see cv_extrinsic_to_gl). Principal-point offsets follow the OpenCV
top-left origin; the y sign is handled by the extrinsic flip.
"""
P = np.zeros((4, 4))
P[0, 0] = 2.0 * fx / W
P[1, 1] = 2.0 * fy / H
P[0, 2] = 1.0 - 2.0 * cx / W
P[1, 2] = 2.0 * cy / H - 1.0
P[2, 2] = -(far + near) / (far - near)
P[2, 3] = -2.0 * far * near / (far - near)
P[3, 2] = -1.0
return P
# OpenCV cam (x right, y down, z forward) -> OpenGL cam (x right, y up, z back)
_CV2GL = np.diag([1.0, -1.0, -1.0, 1.0])
def cv_extrinsic_to_gl(c2w_cv):
"""world->OpenGL-camera matrix from an OpenCV camera-to-world matrix."""
w2c_cv = np.linalg.inv(np.asarray(c2w_cv, float))
return _CV2GL @ w2c_cv
def orbit_cameras(radius=2.6, elevs=(-30, 0, 30),
azims=range(0, 360, 45), up=(0, 0, 1)):
"""Return list of dicts {name, eye, view} for the 24-view rig.
Azimuth is measured in the world XY plane; elevation lifts along +Z.
az=0 places the camera on -Y looking toward +Y (matches toys4k 'front').
"""
up = np.asarray(up, float)
cams = []
for el in elevs:
for az in azims:
ar = np.radians(az)
er = np.radians(el)
x = radius * np.cos(er) * np.sin(ar)
y = -radius * np.cos(er) * np.cos(ar)
z = radius * np.sin(er)
eye = np.array([x, y, z])
cams.append({
"name": f"az{az:03d}_el{el:+03d}",
"eye": eye,
"view": look_at(eye, (0, 0, 0), up),
})
return cams
# ----------------------------------------------------------------------------
# core render
# ----------------------------------------------------------------------------
def _render_mvp(g: MeshGL, mvp, H, W, ctx=None):
"""Rasterize mesh g under a 4x4 clip transform. Returns RGBA (H,W,4) [0,1].
The output is oriented conventionally (row 0 = top of image).
"""
ctx = ctx or get_ctx()
device = g.verts.device
mvp_t = torch.as_tensor(mvp, dtype=torch.float32, device=device)
vh = torch.cat([g.verts, torch.ones(len(g.verts), 1, device=device)], 1)
clip = (mvp_t @ vh.T).T.contiguous()[None] # (1,V,4)
rast, _ = dr.rasterize(ctx, clip, g.faces, (H, W))
alpha = (rast[..., 3:4] > 0).float() # (1,H,W,1)
if g.mode == "uv":
uv_i, _ = dr.interpolate(g.uv[None], rast, g.faces)
tex = g.tex[None] # (1,Ht,Wt,3)
col = dr.texture(tex, uv_i, filter_mode="linear") # (1,H,W,3)
elif g.mode == "vertex":
col, _ = dr.interpolate(g.vcol[None], rast, g.faces)
else:
col = g.flat[None, None, None, :].expand(1, H, W, 3)
col = col * alpha # zero the background
col = dr.antialias(col.contiguous(), rast, clip, g.faces)
alpha = dr.antialias(alpha.contiguous(), rast, clip, g.faces)
rgba = torch.cat([col, alpha], -1)[0].clamp(0, 1) # (H,W,4)
rgba = torch.flip(rgba, dims=[0]) # GL bottom-up -> top-down
return rgba
def render_orbit(g: MeshGL, cams, H=512, W=512, fovy=40.0, near=0.05, far=20.0,
ctx=None):
"""Render mesh g from a list of orbit cameras. Returns (N,H,W,4) tensor."""
P = gl_perspective(fovy, W / H, near, far)
out = []
for cam in cams:
mvp = P @ cam["view"]
out.append(_render_mvp(g, mvp, H, W, ctx=ctx))
return torch.stack(out, 0)
def render_input_view(g: MeshGL, K, c2w_cv, H, W, near=0.05, far=None, ctx=None):
"""Render mesh g from a SPECIFIC OpenCV camera (K dict + c2w_cv 4x4).
K = {'fx','fy','cx','cy'} in pixels at resolution (H,W). Returns (H,W,4).
FAR-CLIP FIX (2026-09-25): far was a fixed 20.0 canonical units; FB150
tiny objects have cameras 19.2-21.6 units from the origin, so the whole
object was clipped (empty render). far now defaults to
max(20, |camera centre| + 10): unchanged (20) for every camera within 10
units (all Toys/Omni, most FB150); only the depth mapping, never the
visible surface, changes otherwise.
"""
if far is None:
far = max(20.0, float(np.linalg.norm(np.asarray(c2w_cv, float)[:3, 3])) + 10.0)
P = gl_proj_from_K(K["fx"], K["fy"], K["cx"], K["cy"], W, H, near, far)
Vgl = cv_extrinsic_to_gl(c2w_cv)
mvp = P @ Vgl
return _render_mvp(g, mvp, H, W, ctx=ctx)