bert_simpson / forgebench /code /eval /eval_final.py
Ronaldo-GOAT's picture
FORGE3DBench: final eval protocol + eval_final.py, batched Ours inference, held-out view tars, missing-object lists, README
cca6827 verified
Raw History Blame Contribute Delete
20.6 kB
#!/usr/bin/env python3
"""FINAL evaluator (FORGE3DBench-150 / OmniObject3D-300 / Toys4K-300).
Toys4K-300 (--dataset toys300) keeps the ORIGINAL protocol unchanged: this script
only calls appeval/evaluate_appforce.eval_object (IV = pred render vs input crop,
unmasked; NV = 24-view orbit, pred render vs GT-mesh render; geometry) and
writes the same summary.json/results.csv as below. Numbers are identical to
evaluate_appforce.py (Toys cameras are 2.6 units away, so the far-clip fix is a no-op).
FB150 + Omni300: ONE image-comparison path for input views (IV) and novel views (NV):
the PREDICTED mesh (already in the GT canonical frame) is rendered with the
unlit nvdiffrast renderer (appeval/render.py) at the dataset camera
(K + c2w_cv from renders/<obj>_<tag>.npz), cropped with the stored bbox
(y0,y1,x0,x1), and compared with the DATASET'S OWN GT IMAGE crop
(inputs/<obj>_<tag>.png) inside the GT mask M = png alpha > 0.5:
pred_alpha *= M ; ref_alpha := M ; both composited on white
-> LPIPS(alex) / SSIM / PSNR / CLIP (appeval/appearance.py, unchanged).
Pixels outside M are white in both (ignored); M-pixels not covered by the
prediction are scored as white-vs-GT (penalised).
M is the VISIBLE (modal) object mask on FORGE3DBench (occluders and occluded
object parts are never scored) and the object alpha on Omni/Toys.
This is eval_heldout.heldout_view() (.debug/eval_heldout_nv) verbatim,
applied to both the input cameras and the held-out cameras; unit test
`--selftest-equiv` checks bit-equality against it.
IV = the input cameras (4v: front,side,back,oside ; 1v: front).
NV = held-out cameras never fed to any model (same set for 1v and 4v):
fb150 : the other scene cameras (of 24) with visibility
(modal/amodal area, views_4v.json) >= --min-vis (0.20) and >=400
visible px; tags h<cam>, files from data/fb150_heldout.tar
omni300: saved held-out renders top,top2,bottom,bottom2 (--nv-set saved,
default) [+ optional official-render pool p<idx> (--nv-set all)]
Geometry: appeval/geometry.py geometry_metrics(pred, GT canon) (CD-L1, F@tau,
NC, vol-IoU; original keys unchanged, + extent-relative *_rel keys). No
alignment is done here: meshes must already be in the GT frame
(align_v2.py / cammap_*.py upstream, see README).
Per-object means over views, then mean over objects. summary.json is written
by THIS script (single run or --summarize over shard dirs), never by hand.
Diagnostics (not the protocol):
--pred-is-gt score the GT canonical mesh itself (= protocol ceiling;
appearance is NOT perfect: lit scene images vs unlit
render; geometry is exactly perfect).
--ref gtrender replace the GT image by the GT-mesh render (same camera,
same mask). With --pred-is-gt this must be exactly
perfect (pipeline identity test).
"""
from __future__ import annotations
import argparse
import csv
import json
import os
import sys
import time
from pathlib import Path
import numpy as np
HERE = Path(__file__).resolve().parent
sys.path.insert(0, str(HERE / "appeval"))
import render as R # noqa: E402 (far-clip fixed)
import evaluate_appforce as EA # noqa: E402 (helpers only)
from appearance import appearance_metrics, composite_white # noqa: E402
from geometry import geometry_metrics # noqa: E402
MKEYS = ("lpips", "ssim", "psnr", "clip")
GKEYS = ("cd_l1", "f01", "f02", "f05", "normal_consistency", "vol_iou", "cd_l1_rel", "f01_rel")
IV_TAGS = {"4v": ["front", "side", "back", "oside"], "1v": ["front"]}
LP = "/lp-dev/jonghoon/mv-mesh"
# defaults = layout on the original server; override with --exp / --heldout-root
DEFAULT_EXP = {
("fb150", "4v"): f"{LP}/.debug/forgebench_eval/fb150/exp_4v",
("fb150", "1v"): f"{LP}/.debug/forgebench_eval/fb150/exp_1v",
("omni300", "4v"): f"{LP}/exp_faithfulness/omni3d300_rand",
("omni300", "1v"): f"{LP}/exp_faithfulness/omni3d300_rand",
("toys300", "4v"): f"{LP}/exp_faithfulness/toys4k300_rand",
("toys300", "1v"): f"{LP}/exp_faithfulness/toys4k300_rand",
}
DEFAULT_HELDOUT_ROOT = {"fb150": f"{LP}/.debug/eval_heldout_nv/fb150_heldout",
"omni300": f"{LP}/.debug/eval_heldout_nv/omni300_heldout",
"toys300": None}
BAD_FRAME_FILE = HERE / "heldout_index" / "omni300_bad_frame_objects.txt"
# --------------------------------------------------------------------------- views
def masked_view(g, root: Path, obj: str, tag: str, ctx, gt_g=None):
"""== eval_heldout.heldout_view (verbatim logic). Returns (metrics, pred, ref)."""
z = np.load(root / "renders" / f"{obj}_{tag}.npz")
K = {k: float(z[k]) for k in ("fx", "fy", "cx", "cy")}
res = int(z["res"]); y0, y1, x0, x1 = [int(v) for v in z["bbox"].tolist()]
ref = EA.read_input_png(root / "inputs" / f"{obj}_{tag}.png")
pred = R.render_input_view(g, K, z["c2w_cv"], res, res, ctx=ctx).cpu().numpy()[y0:y1, x0:x1]
h, w = min(pred.shape[0], ref.shape[0]), min(pred.shape[1], ref.shape[1])
pred, ref = pred[:h, :w].copy(), ref[:h, :w]
M = (ref[..., 3] > 0.5).astype(np.float32)
if gt_g is not None: # diagnostic: GT-mesh render replaces the GT image (same mask)
ref = R.render_input_view(gt_g, K, z["c2w_cv"], res, res, ctx=ctx).cpu().numpy()[y0:y1, x0:x1][:h, :w]
pred[..., 3] *= M # restrict to GT (visible) mask
ref = ref.copy(); ref[..., 3] = M if gt_g is None else ref[..., 3] * M
m = appearance_metrics(pred, ref)
m["coverage"] = float((pred[..., 3] > 0.5).sum() / max(M.sum(), 1)) # frac of mask hit by pred
m.update(view=tag, mask_px=int(M.sum()), cam_dist=float(np.linalg.norm(z["c2w_cv"][:3, 3])))
return m, pred, ref
def load_index(path: Path):
return json.loads(Path(path).read_text())
def nv_views(a, recs):
"""Filter the held-out records of one object according to the protocol flags."""
out = []
for r in recs:
if a.dataset == "fb150" and r["vis"] < a.min_vis:
continue
if a.dataset == "omni300" and a.nv_set != "all" and r["kind"] != a.nv_set:
continue
out.append(r)
return out
def view_root(a, r):
return a.exp if r["src"] == "exp" else a.heldout_root
def mean_views(per):
d = {k: float(np.mean([v[k] for v in per])) for k in MKEYS + ("coverage",)}
d["n_views"] = len(per)
return d
# --------------------------------------------------------------------------- object
def run_object(a, obj, recs, ctx, save_debug):
gt_path = EA.resolve_gt(a.exp / "renders", obj)
if gt_path is None:
return {"object": obj, "error": "missing gt mesh"}
pred_path = gt_path if a.pred_is_gt else a.meshes / f"{obj}.glb"
if not pred_path.exists():
return {"object": obj, "error": "missing pred glb"}
if pred_path.stat().st_size < 1024:
return {"object": obj, "error": f"pred glb too small ({pred_path.stat().st_size} B)"}
pred_mesh = EA.load_mesh(pred_path)
if len(getattr(pred_mesh, "faces", [])) == 0:
return {"object": obj, "error": "pred mesh has no faces"}
gt_mesh = EA.load_mesh(gt_path)
g = R.prepare_mesh(pred_mesh)
gt_g = R.prepare_mesh(gt_mesh) if a.ref == "gtrender" else None
row = {"object": obj, "pred": str(pred_path), "bad_frame": obj in a.bad_frame}
sheet = []
for kind, tags in (("iv", [dict(tag=t, src="exp") for t in IV_TAGS[a.setting]]), ("nv", nv_views(a, recs))):
per = []
for r in tags:
root = view_root(a, r)
if not (root / "renders" / f"{obj}_{r['tag']}.npz").exists():
continue
m, p, rf = masked_view(g, root, obj, r["tag"], ctx, gt_g)
for k in ("cam", "vis", "kind"):
if k in r:
m[k] = r[k]
per.append(m)
if save_debug and len(sheet) < 10:
sheet += [composite_white(rf), composite_white(p)]
if per:
row[kind] = mean_views(per)
row[kind]["per_view"] = per
else:
row[f"{kind}_error"] = "no views"
if not a.no_geometry:
try:
row["geometry"] = geometry_metrics(pred_mesh, gt_mesh, with_vol=not a.no_vol_iou)
except Exception as e: # noqa: BLE001
row["geometry_error"] = f"{type(e).__name__}: {e}"
if sheet:
EA.save_sheet(sheet, a.out / "debug" / f"{obj}_iv_nv.png")
return row
def run_object_toys(a, obj, ctx, save_debug):
"""ORIGINAL Toys4K protocol: evaluate_appforce.eval_object, unchanged."""
meshes = (a.exp / "renders") if a.pred_is_gt else a.meshes
if a.pred_is_gt: # eval_object reads <meshes>/<obj>.glb -> link dir to the GT canon
link = a.out / "_gt_links"; link.mkdir(parents=True, exist_ok=True)
l = link / f"{obj}.glb"
if not l.exists():
l.symlink_to(EA.resolve_gt(a.exp / "renders", obj))
meshes = link
r = EA.eval_object(a.exp, meshes, a.exp / "renders", obj, IV_TAGS[a.setting], True, a.out, save_debug, ctx)
if "error" in r:
return r
r = _legacy(r)
if "nv" not in r and r.get("gt_textured") is False:
r["nv_error"] = "GT untextured (orbit NV skipped by the original protocol)"
return r
# --------------------------------------------------------------------------- summary
def complete(r, need_geom=True):
"""IV + geometry required; NV required unless the GT is untextured (Toys orbit NV skipped)."""
nv_ok = "nv" in r or r.get("gt_textured") is False
return ("error" not in r and "iv" in r and nv_ok and (not need_geom or "geometry" in r))
def _legacy(r):
"""Rows written by the ORIGINAL evaluate_appforce.py (Toys4K) -> iv/nv keys (values untouched)."""
if "iv" not in r and r.get("input_view"):
r = dict(r)
r["iv"] = {k: r["input_view"][k] for k in MKEYS}; r["iv"]["coverage"] = float("nan")
r["iv"]["n_views"] = len(r["input_view"].get("per_view", [])) or 1
if r.get("novel_view"):
r["nv"] = {k: r["novel_view"][k] for k in MKEYS}; r["nv"]["coverage"] = float("nan"); r["nv"]["n_views"] = 24
return r
def summarize(rows, meta, out: Path, expected=None, exclude=()):
"""Merge rows (dedup: the LAST complete row of an object wins, i.e. pass retry dirs after the
main dirs; conflicting complete duplicates are listed), write results.json / results.csv / summary.json."""
by, conflicts = {}, {}
for r in map(_legacy, rows):
o = r["object"]
c_new = complete(r, not meta.get("no_geometry"))
if o in by and c_new and complete(by[o], not meta.get("no_geometry")):
d = abs(by[o]["iv"]["lpips"] - r["iv"]["lpips"]) + abs(by[o].get("geometry", {}).get("cd_l1", 0) - r.get("geometry", {}).get("cd_l1", 0))
if d > 1e-6:
conflicts[o] = conflicts.get(o, 0) + 1
if o not in by or c_new:
by[o] = r
rows = [by[o] for o in sorted(by)]
need_g = not meta.get("no_geometry")
ok = [r for r in rows if complete(r, need_g)]
failed = {r["object"]: r.get("error") or r.get("iv_error") or r.get("nv_error") or r.get("geometry_error")
for r in rows if not complete(r, need_g)}
excl = sorted(o for o in exclude if o in by or (expected and o in expected))
inc = [r for r in ok if r["object"] not in set(exclude)]
def means(rs):
d = {"n_objects": len(rs)}
if not rs:
return d
for kind in ("iv", "nv"):
rk = [r for r in rs if kind in r]
if not rk:
continue
d[kind] = {k: float(np.mean([r[kind][k] for r in rk])) for k in MKEYS + ("coverage",)}
d[kind]["n_views_total"] = int(sum(r[kind]["n_views"] for r in rk))
d[kind]["n_objects"] = len(rk)
gs = [r for r in rs if "geometry" in r]
if gs:
d["geometry"] = {k: float(np.nanmean([r["geometry"][k] for r in gs])) for k in GKEYS if k in gs[0]["geometry"]}
return d
summ = dict(meta)
summ["means"] = means(inc) # HEADLINE (exclusions applied)
if excl:
summ["means_including_excluded"] = means(ok)
summ["excluded_objects"] = excl
summ["failed_objects"] = failed
summ["duplicate_conflicts_last_wins"] = conflicts
if expected is not None:
summ["n_expected"] = len(expected)
summ["missing_objects"] = sorted(set(expected) - set(by))
summ["n_scored"] = len(ok)
summ["generated_by"] = f"{Path(__file__).name} (sha256 {_sha(Path(__file__))[:12]})"
summ["time_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
out.mkdir(parents=True, exist_ok=True)
(out / "results.json").write_text(json.dumps(rows, indent=1))
with open(out / "results.csv", "w", newline="") as f:
w = csv.writer(f)
hdr = (["object", "status", "excluded"] + [f"iv_{k}" for k in MKEYS] + ["iv_n"]
+ [f"nv_{k}" for k in MKEYS] + ["nv_n"] + list(GKEYS))
w.writerow(hdr)
for r in rows:
st = "ok" if complete(r, need_g) else "failed"
line = [r["object"], st, r["object"] in set(exclude)]
for kind in ("iv", "nv"):
d = r.get(kind, {})
line += [d.get(k, "") for k in MKEYS] + [d.get("n_views", "")]
gg = r.get("geometry", {})
line += [gg.get(k, "") for k in GKEYS]
w.writerow(line)
(out / "summary.json").write_text(json.dumps(summ, indent=1))
return summ
def _sha(p: Path):
import hashlib
return hashlib.sha256(p.read_bytes()).hexdigest()
# --------------------------------------------------------------------------- main
def parse():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--dataset", choices=["fb150", "omni300", "toys300"])
ap.add_argument("--setting", choices=["4v", "1v"], default="4v", help="IV cameras (NV set is the same for both)")
ap.add_argument("--meshes", type=Path, help="DIR with <object>.glb already in the GT canonical frame")
ap.add_argument("--method", default=None, help="label stored in summary.json (default: meshes dir name)")
ap.add_argument("--out", type=Path)
ap.add_argument("--exp", type=Path, help="dataset exp dir (inputs/, renders/, selection.json)")
ap.add_argument("--heldout-root", type=Path, help="dir with held-out inputs/ renders/ (fb150 h**, omni pool p***)")
ap.add_argument("--heldout-index", type=Path, help="default: heldout_index/<dataset>.json next to this script")
ap.add_argument("--min-vis", type=float, default=0.20, help="fb150 NV: min visible/full-area ratio (protocol 0.20)")
ap.add_argument("--nv-set", choices=["saved", "pool", "all"], default="saved", help="omni300 NV views")
ap.add_argument("--exclude-bad-frame", action="store_true",
help="omni300: drop the 14 objects whose GT canon is not in the render-camera frame from the "
"headline means (they are still scored and listed)")
ap.add_argument("--ref", choices=["image", "gtrender"], default="image",
help="image = PROTOCOL (dataset GT image); gtrender = diagnostic only")
ap.add_argument("--pred-is-gt", action="store_true", help="diagnostic: score the GT canonical mesh itself")
ap.add_argument("--no-geometry", action="store_true")
ap.add_argument("--no-vol-iou", action="store_true", help="skip the (slow) volume IoU; other geometry unchanged")
ap.add_argument("--objects", nargs="*"); ap.add_argument("--objects-file", type=Path)
ap.add_argument("--limit", type=int, default=0)
ap.add_argument("--shard", type=int, default=0); ap.add_argument("--nshards", type=int, default=1)
ap.add_argument("--debug-sheets", type=int, default=4, help="save IV/NV proof sheets for the first N objects")
ap.add_argument("--summarize", nargs="+", type=Path, metavar="SHARD_DIR",
help="merge shard dirs (their results.json) into --out and write the summary; no GPU")
return ap.parse_args()
def setup(a):
a.exp = Path(a.exp or DEFAULT_EXP[(a.dataset, a.setting)])
hr = a.heldout_root or DEFAULT_HELDOUT_ROOT[a.dataset]
a.heldout_root = Path(hr) if hr else a.exp
a.heldout_index = Path(a.heldout_index or HERE / "heldout_index" / f"{a.dataset}.json")
if a.dataset == "toys300" and (a.ref != "image" or a.nv_set != "saved" or a.no_geometry):
sys.exit("toys300 = original evaluate_appforce protocol; --ref/--nv-set/--no-geometry do not apply")
a.bad_frame = set()
if a.dataset == "omni300" and BAD_FRAME_FILE.exists():
a.bad_frame = {l.split()[0] for l in BAD_FRAME_FILE.read_text().splitlines() if l.strip() and not l.startswith("#")}
sel = json.loads((a.exp / "selection.json").read_text())["selections"]
a.expected = [s["object"] for s in sel]
if a.method is None:
a.method = "GT" if a.pred_is_gt else (a.meshes.name if a.meshes else "?")
proto = ("ORIGINAL evaluate_appforce: IV pred-render vs input crop (unmasked); NV 24-orbit pred vs GT-mesh render"
if a.dataset == "toys300" else
"masked GT-image: pred render vs dataset GT image inside the GT/visible mask, IV=input cams, NV=held-out cams")
return dict(method=a.method, dataset=a.dataset, setting=a.setting, protocol=proto, ref=a.ref, pred_is_gt=a.pred_is_gt,
min_vis=a.min_vis if a.dataset == "fb150" else None,
nv_set={"fb150": "heldout_cams_vis>=min_vis", "omni300": a.nv_set, "toys300": "orbit24"}[a.dataset],
exclude_bad_frame=a.exclude_bad_frame, no_geometry=a.no_geometry, no_vol_iou=a.no_vol_iou,
meshes=str(a.meshes) if a.meshes else None, exp=str(a.exp))
def main():
a = parse()
meta = setup(a)
excl = sorted(a.bad_frame) if a.exclude_bad_frame else []
if a.summarize:
rows = []
for d in a.summarize:
rj = d / "results.json"
if not rj.exists():
sys.exit(f"ABORT: {rj} missing (shard not finished?)")
rows += json.loads(rj.read_text())
s = summarize(rows, meta, a.out, expected=a.expected, exclude=excl)
print(json.dumps({k: s[k] for k in ("means", "n_scored", "n_expected", "excluded_objects")}, indent=1))
print(f"missing={len(s['missing_objects'])} failed={len(s['failed_objects'])}")
return
if not a.pred_is_gt and a.meshes is None:
sys.exit("--meshes required (or --pred-is-gt)")
idx = load_index(a.heldout_index) if a.dataset != "toys300" else {}
objs = list(a.expected)
want = set(a.objects or [])
if a.objects_file:
want |= {l.strip() for l in a.objects_file.read_text().split() if l.strip()}
if want:
objs = [o for o in objs if o in want]
if a.limit:
objs = objs[:a.limit]
objs = objs[a.shard::a.nshards]
a.out.mkdir(parents=True, exist_ok=True)
ctx = R.get_ctx()
rows, t0 = [], time.time()
for i, obj in enumerate(objs):
try:
if a.dataset == "toys300":
r = run_object_toys(a, obj, ctx, save_debug=i < a.debug_sheets)
else:
r = run_object(a, obj, idx.get(obj, []), ctx, save_debug=i < a.debug_sheets)
except Exception as e: # noqa: BLE001
import traceback; traceback.print_exc()
r = {"object": obj, "error": f"{type(e).__name__}: {e}"}
rows.append(r)
iv, nv, ge = r.get("iv", {}), r.get("nv", {}), r.get("geometry", {})
print(f"[{i+1}/{len(objs)} {time.time()-t0:.0f}s] {obj} {r.get('error','')} "
f"IV lpips={iv.get('lpips', np.nan):.4f} psnr={iv.get('psnr', np.nan):.2f} | "
f"NV n={nv.get('n_views')} lpips={nv.get('lpips', np.nan):.4f} psnr={nv.get('psnr', np.nan):.2f} | "
f"CD={ge.get('cd_l1', np.nan):.4f} F01={ge.get('f01', np.nan):.3f}", flush=True)
if (i + 1) % 10 == 0: # checkpoint partial results
(a.out / "results.json").write_text(json.dumps(rows, indent=1))
exp_list = a.expected if (a.nshards == 1 and not want and not a.limit) else objs
s = summarize(rows, meta, a.out, expected=exp_list, exclude=excl)
print(json.dumps(s["means"], indent=1))
if __name__ == "__main__":
main()