fishxinyu's picture
download
raw
6.8 kB
"""Go/no-go gate for a camera-trajectory backend, run before trusting any score it produces.
A camera metric that cannot tell the benchmark's own reference clips apart cannot possibly
measure whether a generation followed one of them. This scores every reference against every
reference and checks the obvious property: each clip must match itself better than it matches
any other clip, by a margin larger than the metric's own scatter.
python evaluation/eval_camera/validate.py --backend vggt --backend similarity2d
Reported per backend:
confusion the R x R matrix of `rot_err` / `trans_err` / `cam_mc`, diagonal = self-match
margin (min off-diagonal - diagonal) / off-diagonal spread, per row. > 1 is the gate.
dynamic each clip's error against the static control. A clip whose static-control error
is near zero has no motion the backend can see, and nothing measured on it means
anything -- this is the check the old 2D estimator silently failed on 2 of 4.
The self-match diagonal is not exactly zero even for a perfect backend: the trajectory is
recomputed from the same file, so it reflects estimator determinism, not error. It should be
numerically negligible.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
import numpy as np
from metric import SCORED_METRICS, compare, prepare, static_control
from traj import BACKENDS, extract_trajectory
DEFAULT_REFS = Path(
"/mnt/data/xinyuy/datasets/flexcombine-bench/v1/sources/camera"
)
# Two different camera motions must sit at least half as far apart as either sits from a
# locked-off camera. Below that, "followed reference A" and "followed reference B" are not
# separable statements and no downstream alignment number means anything.
MIN_SEPARATION = 0.5
def confusion(
refs: dict[str, Path],
backend: str,
max_frames: int,
cache_dir: Path | None,
) -> tuple[dict[str, np.ndarray], dict[str, dict], list[str]]:
names = sorted(refs)
trajs = {
name: extract_trajectory(refs[name], backend, max_frames, cache_dir)
for name in names
}
n = min(len(t["R"]) for t in trajs.values())
prepared = {name: prepare(t, n) for name, t in trajs.items()}
matrices = {m: np.zeros((len(names), len(names))) for m in SCORED_METRICS}
for i, a in enumerate(names):
for j, b in enumerate(names):
result = compare(prepared[a], prepared[b])
for m in SCORED_METRICS:
matrices[m][i, j] = result[m]
controls = {name: static_control(prepared[name]) for name in names}
return matrices, controls, names
def separations(matrix: np.ndarray, control: np.ndarray) -> np.ndarray:
"""Per-row distance to the nearest *other* reference, in units of that clip's own motion.
sep_i = min_{j != i} d(i, j) / static_control_i
The denominator is the clip's error against a locked-off camera, i.e. the full dynamic
range the metric has available on that clip. So sep = 1 means "the nearest other reference
is as far away as doing nothing at all", and sep << 1 means two genuinely different camera
motions land almost on top of each other and the metric cannot tell them apart.
Normalizing by the off-diagonal *spread* instead would be wrong: it punishes a backend for
the four references being unequally different from one another, which is a property of the
library, not a defect of the estimator. The diagonal is an exact zero here (the same file
re-scored against itself), so it carries no information and is not subtracted.
"""
out = np.zeros(len(matrix))
for i, row in enumerate(matrix):
off = np.delete(row, i)
denom = control[i] if control[i] > 1e-9 else 1e-9
out[i] = off.min() / denom
return out
def report(backend: str, matrices, controls, names) -> dict:
print(f"\n{'=' * 78}\nbackend: {backend}\n{'=' * 78}")
summary: dict[str, object] = {"backend": backend}
for m in SCORED_METRICS:
matrix = matrices[m]
control = np.array([controls[n][m] for n in names])
sep = separations(matrix, control)
print(f"\n{m} (rows/cols = reference clips; diagonal = self-match; "
f"'static' = error vs a locked-off camera)")
print(" " + " " * 12 + "".join(f"{n:>12}" for n in names)
+ f"{'static':>12}{'sep':>8}")
for i, name in enumerate(names):
cells = "".join(f"{matrix[i, j]:>12.4f}" for j in range(len(names)))
print(f" {name:>12}{cells}{control[i]:>12.4f}{sep[i]:>8.2f}")
summary[m] = {
"matrix": matrix.tolist(),
"separation": sep.tolist(),
"static_control": control.tolist(),
"diagonal_is_min": bool(all(matrix[i].argmin() == i for i in range(len(names)))),
"min_separation": float(sep.min()),
}
passed = all(
summary[m]["diagonal_is_min"] and summary[m]["min_separation"] > MIN_SEPARATION
for m in SCORED_METRICS
)
summary["passes_gate"] = passed
worst = {m: round(summary[m]["min_separation"], 2) for m in SCORED_METRICS}
print(f"\n GATE: {'PASS' if passed else 'FAIL'} worst separation per metric: {worst}"
f"\n (need diagonal minimal on every row AND separation > {MIN_SEPARATION})")
return summary
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument(
"--backend", action="append", choices=BACKENDS, default=None,
help="Repeatable. Default: all three, so they can be compared in one table.",
)
parser.add_argument("--refs-dir", default=str(DEFAULT_REFS))
parser.add_argument("--max-frames", type=int, default=41)
parser.add_argument("--cache-dir", default=None)
parser.add_argument("--output", default=None, help="Write the summary as JSON.")
args = parser.parse_args()
refs_dir = Path(args.refs_dir)
refs = {p.stem: p for p in sorted(refs_dir.glob("*.mp4"))}
if len(refs) < 2:
raise SystemExit(f"Need >=2 reference clips to build a confusion matrix, found {len(refs)} in {refs_dir}")
cache_dir = Path(args.cache_dir) if args.cache_dir else None
summaries = []
for backend in args.backend or list(BACKENDS):
try:
matrices, controls, names = confusion(refs, backend, args.max_frames, cache_dir)
except ImportError as exc:
print(f"\nbackend {backend}: SKIPPED ({exc})")
continue
summaries.append(report(backend, matrices, controls, names))
if args.output:
Path(args.output).write_text(json.dumps(summaries, indent=2))
print(f"\nSummary written to {args.output}")
if __name__ == "__main__":
main()

Xet Storage Details

Size:
6.8 kB
·
Xet hash:
5ddd8ff25008856763250dace55f924b409889a98d384fa393738266ac8e41b5

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.