#!/usr/bin/env python3 """FINAL evaluator (FORGE3DBench-150 / OmniObject3D-300 / Toys4K-300). Toys4K-300 (--dataset toys300) keeps the ORIGINAL protocol unchanged: this script only calls appeval/evaluate_appforce.eval_object (IV = pred render vs input crop, unmasked; NV = 24-view orbit, pred render vs GT-mesh render; geometry) and writes the same summary.json/results.csv as below. Numbers are identical to evaluate_appforce.py (Toys cameras are 2.6 units away, so the far-clip fix is a no-op). FB150 + Omni300: ONE image-comparison path for input views (IV) and novel views (NV): the PREDICTED mesh (already in the GT canonical frame) is rendered with the unlit nvdiffrast renderer (appeval/render.py) at the dataset camera (K + c2w_cv from renders/_.npz), cropped with the stored bbox (y0,y1,x0,x1), and compared with the DATASET'S OWN GT IMAGE crop (inputs/_.png) inside the GT mask M = png alpha > 0.5: pred_alpha *= M ; ref_alpha := M ; both composited on white -> LPIPS(alex) / SSIM / PSNR / CLIP (appeval/appearance.py, unchanged). Pixels outside M are white in both (ignored); M-pixels not covered by the prediction are scored as white-vs-GT (penalised). M is the VISIBLE (modal) object mask on FORGE3DBench (occluders and occluded object parts are never scored) and the object alpha on Omni/Toys. This is eval_heldout.heldout_view() (.debug/eval_heldout_nv) verbatim, applied to both the input cameras and the held-out cameras; unit test `--selftest-equiv` checks bit-equality against it. IV = the input cameras (4v: front,side,back,oside ; 1v: front). NV = held-out cameras never fed to any model (same set for 1v and 4v): fb150 : the other scene cameras (of 24) with visibility (modal/amodal area, views_4v.json) >= --min-vis (0.20) and >=400 visible px; tags h, files from data/fb150_heldout.tar omni300: saved held-out renders top,top2,bottom,bottom2 (--nv-set saved, default) [+ optional official-render pool p (--nv-set all)] Geometry: appeval/geometry.py geometry_metrics(pred, GT canon) (CD-L1, F@tau, NC, vol-IoU; original keys unchanged, + extent-relative *_rel keys). No alignment is done here: meshes must already be in the GT frame (align_v2.py / cammap_*.py upstream, see README). Per-object means over views, then mean over objects. summary.json is written by THIS script (single run or --summarize over shard dirs), never by hand. Diagnostics (not the protocol): --pred-is-gt score the GT canonical mesh itself (= protocol ceiling; appearance is NOT perfect: lit scene images vs unlit render; geometry is exactly perfect). --ref gtrender replace the GT image by the GT-mesh render (same camera, same mask). With --pred-is-gt this must be exactly perfect (pipeline identity test). """ from __future__ import annotations import argparse import csv import json import os import sys import time from pathlib import Path import numpy as np HERE = Path(__file__).resolve().parent sys.path.insert(0, str(HERE / "appeval")) import render as R # noqa: E402 (far-clip fixed) import evaluate_appforce as EA # noqa: E402 (helpers only) from appearance import appearance_metrics, composite_white # noqa: E402 from geometry import geometry_metrics # noqa: E402 MKEYS = ("lpips", "ssim", "psnr", "clip") GKEYS = ("cd_l1", "f01", "f02", "f05", "normal_consistency", "vol_iou", "cd_l1_rel", "f01_rel") IV_TAGS = {"4v": ["front", "side", "back", "oside"], "1v": ["front"]} LP = "/lp-dev/jonghoon/mv-mesh" # defaults = layout on the original server; override with --exp / --heldout-root DEFAULT_EXP = { ("fb150", "4v"): f"{LP}/.debug/forgebench_eval/fb150/exp_4v", ("fb150", "1v"): f"{LP}/.debug/forgebench_eval/fb150/exp_1v", ("omni300", "4v"): f"{LP}/exp_faithfulness/omni3d300_rand", ("omni300", "1v"): f"{LP}/exp_faithfulness/omni3d300_rand", ("toys300", "4v"): f"{LP}/exp_faithfulness/toys4k300_rand", ("toys300", "1v"): f"{LP}/exp_faithfulness/toys4k300_rand", } DEFAULT_HELDOUT_ROOT = {"fb150": f"{LP}/.debug/eval_heldout_nv/fb150_heldout", "omni300": f"{LP}/.debug/eval_heldout_nv/omni300_heldout", "toys300": None} BAD_FRAME_FILE = HERE / "heldout_index" / "omni300_bad_frame_objects.txt" # --------------------------------------------------------------------------- views def masked_view(g, root: Path, obj: str, tag: str, ctx, gt_g=None): """== eval_heldout.heldout_view (verbatim logic). Returns (metrics, pred, ref).""" z = np.load(root / "renders" / f"{obj}_{tag}.npz") K = {k: float(z[k]) for k in ("fx", "fy", "cx", "cy")} res = int(z["res"]); y0, y1, x0, x1 = [int(v) for v in z["bbox"].tolist()] ref = EA.read_input_png(root / "inputs" / f"{obj}_{tag}.png") pred = R.render_input_view(g, K, z["c2w_cv"], res, res, ctx=ctx).cpu().numpy()[y0:y1, x0:x1] h, w = min(pred.shape[0], ref.shape[0]), min(pred.shape[1], ref.shape[1]) pred, ref = pred[:h, :w].copy(), ref[:h, :w] M = (ref[..., 3] > 0.5).astype(np.float32) if gt_g is not None: # diagnostic: GT-mesh render replaces the GT image (same mask) ref = R.render_input_view(gt_g, K, z["c2w_cv"], res, res, ctx=ctx).cpu().numpy()[y0:y1, x0:x1][:h, :w] pred[..., 3] *= M # restrict to GT (visible) mask ref = ref.copy(); ref[..., 3] = M if gt_g is None else ref[..., 3] * M m = appearance_metrics(pred, ref) m["coverage"] = float((pred[..., 3] > 0.5).sum() / max(M.sum(), 1)) # frac of mask hit by pred m.update(view=tag, mask_px=int(M.sum()), cam_dist=float(np.linalg.norm(z["c2w_cv"][:3, 3]))) return m, pred, ref def load_index(path: Path): return json.loads(Path(path).read_text()) def nv_views(a, recs): """Filter the held-out records of one object according to the protocol flags.""" out = [] for r in recs: if a.dataset == "fb150" and r["vis"] < a.min_vis: continue if a.dataset == "omni300" and a.nv_set != "all" and r["kind"] != a.nv_set: continue out.append(r) return out def view_root(a, r): return a.exp if r["src"] == "exp" else a.heldout_root def mean_views(per): d = {k: float(np.mean([v[k] for v in per])) for k in MKEYS + ("coverage",)} d["n_views"] = len(per) return d # --------------------------------------------------------------------------- object def run_object(a, obj, recs, ctx, save_debug): gt_path = EA.resolve_gt(a.exp / "renders", obj) if gt_path is None: return {"object": obj, "error": "missing gt mesh"} pred_path = gt_path if a.pred_is_gt else a.meshes / f"{obj}.glb" if not pred_path.exists(): return {"object": obj, "error": "missing pred glb"} if pred_path.stat().st_size < 1024: return {"object": obj, "error": f"pred glb too small ({pred_path.stat().st_size} B)"} pred_mesh = EA.load_mesh(pred_path) if len(getattr(pred_mesh, "faces", [])) == 0: return {"object": obj, "error": "pred mesh has no faces"} gt_mesh = EA.load_mesh(gt_path) g = R.prepare_mesh(pred_mesh) gt_g = R.prepare_mesh(gt_mesh) if a.ref == "gtrender" else None row = {"object": obj, "pred": str(pred_path), "bad_frame": obj in a.bad_frame} sheet = [] for kind, tags in (("iv", [dict(tag=t, src="exp") for t in IV_TAGS[a.setting]]), ("nv", nv_views(a, recs))): per = [] for r in tags: root = view_root(a, r) if not (root / "renders" / f"{obj}_{r['tag']}.npz").exists(): continue m, p, rf = masked_view(g, root, obj, r["tag"], ctx, gt_g) for k in ("cam", "vis", "kind"): if k in r: m[k] = r[k] per.append(m) if save_debug and len(sheet) < 10: sheet += [composite_white(rf), composite_white(p)] if per: row[kind] = mean_views(per) row[kind]["per_view"] = per else: row[f"{kind}_error"] = "no views" if not a.no_geometry: try: row["geometry"] = geometry_metrics(pred_mesh, gt_mesh, with_vol=not a.no_vol_iou) except Exception as e: # noqa: BLE001 row["geometry_error"] = f"{type(e).__name__}: {e}" if sheet: EA.save_sheet(sheet, a.out / "debug" / f"{obj}_iv_nv.png") return row def run_object_toys(a, obj, ctx, save_debug): """ORIGINAL Toys4K protocol: evaluate_appforce.eval_object, unchanged.""" meshes = (a.exp / "renders") if a.pred_is_gt else a.meshes if a.pred_is_gt: # eval_object reads /.glb -> link dir to the GT canon link = a.out / "_gt_links"; link.mkdir(parents=True, exist_ok=True) l = link / f"{obj}.glb" if not l.exists(): l.symlink_to(EA.resolve_gt(a.exp / "renders", obj)) meshes = link r = EA.eval_object(a.exp, meshes, a.exp / "renders", obj, IV_TAGS[a.setting], True, a.out, save_debug, ctx) if "error" in r: return r r = _legacy(r) if "nv" not in r and r.get("gt_textured") is False: r["nv_error"] = "GT untextured (orbit NV skipped by the original protocol)" return r # --------------------------------------------------------------------------- summary def complete(r, need_geom=True): """IV + geometry required; NV required unless the GT is untextured (Toys orbit NV skipped).""" nv_ok = "nv" in r or r.get("gt_textured") is False return ("error" not in r and "iv" in r and nv_ok and (not need_geom or "geometry" in r)) def _legacy(r): """Rows written by the ORIGINAL evaluate_appforce.py (Toys4K) -> iv/nv keys (values untouched).""" if "iv" not in r and r.get("input_view"): r = dict(r) r["iv"] = {k: r["input_view"][k] for k in MKEYS}; r["iv"]["coverage"] = float("nan") r["iv"]["n_views"] = len(r["input_view"].get("per_view", [])) or 1 if r.get("novel_view"): r["nv"] = {k: r["novel_view"][k] for k in MKEYS}; r["nv"]["coverage"] = float("nan"); r["nv"]["n_views"] = 24 return r def summarize(rows, meta, out: Path, expected=None, exclude=()): """Merge rows (dedup: the LAST complete row of an object wins, i.e. pass retry dirs after the main dirs; conflicting complete duplicates are listed), write results.json / results.csv / summary.json.""" by, conflicts = {}, {} for r in map(_legacy, rows): o = r["object"] c_new = complete(r, not meta.get("no_geometry")) if o in by and c_new and complete(by[o], not meta.get("no_geometry")): d = abs(by[o]["iv"]["lpips"] - r["iv"]["lpips"]) + abs(by[o].get("geometry", {}).get("cd_l1", 0) - r.get("geometry", {}).get("cd_l1", 0)) if d > 1e-6: conflicts[o] = conflicts.get(o, 0) + 1 if o not in by or c_new: by[o] = r rows = [by[o] for o in sorted(by)] need_g = not meta.get("no_geometry") ok = [r for r in rows if complete(r, need_g)] failed = {r["object"]: r.get("error") or r.get("iv_error") or r.get("nv_error") or r.get("geometry_error") for r in rows if not complete(r, need_g)} excl = sorted(o for o in exclude if o in by or (expected and o in expected)) inc = [r for r in ok if r["object"] not in set(exclude)] def means(rs): d = {"n_objects": len(rs)} if not rs: return d for kind in ("iv", "nv"): rk = [r for r in rs if kind in r] if not rk: continue d[kind] = {k: float(np.mean([r[kind][k] for r in rk])) for k in MKEYS + ("coverage",)} d[kind]["n_views_total"] = int(sum(r[kind]["n_views"] for r in rk)) d[kind]["n_objects"] = len(rk) gs = [r for r in rs if "geometry" in r] if gs: d["geometry"] = {k: float(np.nanmean([r["geometry"][k] for r in gs])) for k in GKEYS if k in gs[0]["geometry"]} return d summ = dict(meta) summ["means"] = means(inc) # HEADLINE (exclusions applied) if excl: summ["means_including_excluded"] = means(ok) summ["excluded_objects"] = excl summ["failed_objects"] = failed summ["duplicate_conflicts_last_wins"] = conflicts if expected is not None: summ["n_expected"] = len(expected) summ["missing_objects"] = sorted(set(expected) - set(by)) summ["n_scored"] = len(ok) summ["generated_by"] = f"{Path(__file__).name} (sha256 {_sha(Path(__file__))[:12]})" summ["time_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()) out.mkdir(parents=True, exist_ok=True) (out / "results.json").write_text(json.dumps(rows, indent=1)) with open(out / "results.csv", "w", newline="") as f: w = csv.writer(f) hdr = (["object", "status", "excluded"] + [f"iv_{k}" for k in MKEYS] + ["iv_n"] + [f"nv_{k}" for k in MKEYS] + ["nv_n"] + list(GKEYS)) w.writerow(hdr) for r in rows: st = "ok" if complete(r, need_g) else "failed" line = [r["object"], st, r["object"] in set(exclude)] for kind in ("iv", "nv"): d = r.get(kind, {}) line += [d.get(k, "") for k in MKEYS] + [d.get("n_views", "")] gg = r.get("geometry", {}) line += [gg.get(k, "") for k in GKEYS] w.writerow(line) (out / "summary.json").write_text(json.dumps(summ, indent=1)) return summ def _sha(p: Path): import hashlib return hashlib.sha256(p.read_bytes()).hexdigest() # --------------------------------------------------------------------------- main def parse(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("--dataset", choices=["fb150", "omni300", "toys300"]) ap.add_argument("--setting", choices=["4v", "1v"], default="4v", help="IV cameras (NV set is the same for both)") ap.add_argument("--meshes", type=Path, help="DIR with .glb already in the GT canonical frame") ap.add_argument("--method", default=None, help="label stored in summary.json (default: meshes dir name)") ap.add_argument("--out", type=Path) ap.add_argument("--exp", type=Path, help="dataset exp dir (inputs/, renders/, selection.json)") ap.add_argument("--heldout-root", type=Path, help="dir with held-out inputs/ renders/ (fb150 h**, omni pool p***)") ap.add_argument("--heldout-index", type=Path, help="default: heldout_index/.json next to this script") ap.add_argument("--min-vis", type=float, default=0.20, help="fb150 NV: min visible/full-area ratio (protocol 0.20)") ap.add_argument("--nv-set", choices=["saved", "pool", "all"], default="saved", help="omni300 NV views") ap.add_argument("--exclude-bad-frame", action="store_true", help="omni300: drop the 14 objects whose GT canon is not in the render-camera frame from the " "headline means (they are still scored and listed)") ap.add_argument("--ref", choices=["image", "gtrender"], default="image", help="image = PROTOCOL (dataset GT image); gtrender = diagnostic only") ap.add_argument("--pred-is-gt", action="store_true", help="diagnostic: score the GT canonical mesh itself") ap.add_argument("--no-geometry", action="store_true") ap.add_argument("--no-vol-iou", action="store_true", help="skip the (slow) volume IoU; other geometry unchanged") ap.add_argument("--objects", nargs="*"); ap.add_argument("--objects-file", type=Path) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--shard", type=int, default=0); ap.add_argument("--nshards", type=int, default=1) ap.add_argument("--debug-sheets", type=int, default=4, help="save IV/NV proof sheets for the first N objects") ap.add_argument("--summarize", nargs="+", type=Path, metavar="SHARD_DIR", help="merge shard dirs (their results.json) into --out and write the summary; no GPU") return ap.parse_args() def setup(a): a.exp = Path(a.exp or DEFAULT_EXP[(a.dataset, a.setting)]) hr = a.heldout_root or DEFAULT_HELDOUT_ROOT[a.dataset] a.heldout_root = Path(hr) if hr else a.exp a.heldout_index = Path(a.heldout_index or HERE / "heldout_index" / f"{a.dataset}.json") if a.dataset == "toys300" and (a.ref != "image" or a.nv_set != "saved" or a.no_geometry): sys.exit("toys300 = original evaluate_appforce protocol; --ref/--nv-set/--no-geometry do not apply") a.bad_frame = set() if a.dataset == "omni300" and BAD_FRAME_FILE.exists(): a.bad_frame = {l.split()[0] for l in BAD_FRAME_FILE.read_text().splitlines() if l.strip() and not l.startswith("#")} sel = json.loads((a.exp / "selection.json").read_text())["selections"] a.expected = [s["object"] for s in sel] if a.method is None: a.method = "GT" if a.pred_is_gt else (a.meshes.name if a.meshes else "?") proto = ("ORIGINAL evaluate_appforce: IV pred-render vs input crop (unmasked); NV 24-orbit pred vs GT-mesh render" if a.dataset == "toys300" else "masked GT-image: pred render vs dataset GT image inside the GT/visible mask, IV=input cams, NV=held-out cams") return dict(method=a.method, dataset=a.dataset, setting=a.setting, protocol=proto, ref=a.ref, pred_is_gt=a.pred_is_gt, min_vis=a.min_vis if a.dataset == "fb150" else None, nv_set={"fb150": "heldout_cams_vis>=min_vis", "omni300": a.nv_set, "toys300": "orbit24"}[a.dataset], exclude_bad_frame=a.exclude_bad_frame, no_geometry=a.no_geometry, no_vol_iou=a.no_vol_iou, meshes=str(a.meshes) if a.meshes else None, exp=str(a.exp)) def main(): a = parse() meta = setup(a) excl = sorted(a.bad_frame) if a.exclude_bad_frame else [] if a.summarize: rows = [] for d in a.summarize: rj = d / "results.json" if not rj.exists(): sys.exit(f"ABORT: {rj} missing (shard not finished?)") rows += json.loads(rj.read_text()) s = summarize(rows, meta, a.out, expected=a.expected, exclude=excl) print(json.dumps({k: s[k] for k in ("means", "n_scored", "n_expected", "excluded_objects")}, indent=1)) print(f"missing={len(s['missing_objects'])} failed={len(s['failed_objects'])}") return if not a.pred_is_gt and a.meshes is None: sys.exit("--meshes required (or --pred-is-gt)") idx = load_index(a.heldout_index) if a.dataset != "toys300" else {} objs = list(a.expected) want = set(a.objects or []) if a.objects_file: want |= {l.strip() for l in a.objects_file.read_text().split() if l.strip()} if want: objs = [o for o in objs if o in want] if a.limit: objs = objs[:a.limit] objs = objs[a.shard::a.nshards] a.out.mkdir(parents=True, exist_ok=True) ctx = R.get_ctx() rows, t0 = [], time.time() for i, obj in enumerate(objs): try: if a.dataset == "toys300": r = run_object_toys(a, obj, ctx, save_debug=i < a.debug_sheets) else: r = run_object(a, obj, idx.get(obj, []), ctx, save_debug=i < a.debug_sheets) except Exception as e: # noqa: BLE001 import traceback; traceback.print_exc() r = {"object": obj, "error": f"{type(e).__name__}: {e}"} rows.append(r) iv, nv, ge = r.get("iv", {}), r.get("nv", {}), r.get("geometry", {}) print(f"[{i+1}/{len(objs)} {time.time()-t0:.0f}s] {obj} {r.get('error','')} " f"IV lpips={iv.get('lpips', np.nan):.4f} psnr={iv.get('psnr', np.nan):.2f} | " f"NV n={nv.get('n_views')} lpips={nv.get('lpips', np.nan):.4f} psnr={nv.get('psnr', np.nan):.2f} | " f"CD={ge.get('cd_l1', np.nan):.4f} F01={ge.get('f01', np.nan):.3f}", flush=True) if (i + 1) % 10 == 0: # checkpoint partial results (a.out / "results.json").write_text(json.dumps(rows, indent=1)) exp_list = a.expected if (a.nshards == 1 and not want and not a.limit) else objs s = summarize(rows, meta, a.out, expected=exp_list, exclude=excl) print(json.dumps(s["means"], indent=1)) if __name__ == "__main__": main()