Buckets:
| """Go/no-go gate for a camera-trajectory backend, run before trusting any score it produces. | |
| A camera metric that cannot tell the benchmark's own reference clips apart cannot possibly | |
| measure whether a generation followed one of them. This scores every reference against every | |
| reference and checks the obvious property: each clip must match itself better than it matches | |
| any other clip, by a margin larger than the metric's own scatter. | |
| python evaluation/eval_camera/validate.py --backend vggt --backend similarity2d | |
| Reported per backend: | |
| confusion the R x R matrix of `rot_err` / `trans_err` / `cam_mc`, diagonal = self-match | |
| margin (min off-diagonal - diagonal) / off-diagonal spread, per row. > 1 is the gate. | |
| dynamic each clip's error against the static control. A clip whose static-control error | |
| is near zero has no motion the backend can see, and nothing measured on it means | |
| anything -- this is the check the old 2D estimator silently failed on 2 of 4. | |
| The self-match diagonal is not exactly zero even for a perfect backend: the trajectory is | |
| recomputed from the same file, so it reflects estimator determinism, not error. It should be | |
| numerically negligible. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| import numpy as np | |
| from metric import SCORED_METRICS, compare, prepare, static_control | |
| from traj import BACKENDS, extract_trajectory | |
| DEFAULT_REFS = Path( | |
| "/mnt/data/xinyuy/datasets/flexcombine-bench/v1/sources/camera" | |
| ) | |
| # Two different camera motions must sit at least half as far apart as either sits from a | |
| # locked-off camera. Below that, "followed reference A" and "followed reference B" are not | |
| # separable statements and no downstream alignment number means anything. | |
| MIN_SEPARATION = 0.5 | |
| def confusion( | |
| refs: dict[str, Path], | |
| backend: str, | |
| max_frames: int, | |
| cache_dir: Path | None, | |
| ) -> tuple[dict[str, np.ndarray], dict[str, dict], list[str]]: | |
| names = sorted(refs) | |
| trajs = { | |
| name: extract_trajectory(refs[name], backend, max_frames, cache_dir) | |
| for name in names | |
| } | |
| n = min(len(t["R"]) for t in trajs.values()) | |
| prepared = {name: prepare(t, n) for name, t in trajs.items()} | |
| matrices = {m: np.zeros((len(names), len(names))) for m in SCORED_METRICS} | |
| for i, a in enumerate(names): | |
| for j, b in enumerate(names): | |
| result = compare(prepared[a], prepared[b]) | |
| for m in SCORED_METRICS: | |
| matrices[m][i, j] = result[m] | |
| controls = {name: static_control(prepared[name]) for name in names} | |
| return matrices, controls, names | |
| def separations(matrix: np.ndarray, control: np.ndarray) -> np.ndarray: | |
| """Per-row distance to the nearest *other* reference, in units of that clip's own motion. | |
| sep_i = min_{j != i} d(i, j) / static_control_i | |
| The denominator is the clip's error against a locked-off camera, i.e. the full dynamic | |
| range the metric has available on that clip. So sep = 1 means "the nearest other reference | |
| is as far away as doing nothing at all", and sep << 1 means two genuinely different camera | |
| motions land almost on top of each other and the metric cannot tell them apart. | |
| Normalizing by the off-diagonal *spread* instead would be wrong: it punishes a backend for | |
| the four references being unequally different from one another, which is a property of the | |
| library, not a defect of the estimator. The diagonal is an exact zero here (the same file | |
| re-scored against itself), so it carries no information and is not subtracted. | |
| """ | |
| out = np.zeros(len(matrix)) | |
| for i, row in enumerate(matrix): | |
| off = np.delete(row, i) | |
| denom = control[i] if control[i] > 1e-9 else 1e-9 | |
| out[i] = off.min() / denom | |
| return out | |
| def report(backend: str, matrices, controls, names) -> dict: | |
| print(f"\n{'=' * 78}\nbackend: {backend}\n{'=' * 78}") | |
| summary: dict[str, object] = {"backend": backend} | |
| for m in SCORED_METRICS: | |
| matrix = matrices[m] | |
| control = np.array([controls[n][m] for n in names]) | |
| sep = separations(matrix, control) | |
| print(f"\n{m} (rows/cols = reference clips; diagonal = self-match; " | |
| f"'static' = error vs a locked-off camera)") | |
| print(" " + " " * 12 + "".join(f"{n:>12}" for n in names) | |
| + f"{'static':>12}{'sep':>8}") | |
| for i, name in enumerate(names): | |
| cells = "".join(f"{matrix[i, j]:>12.4f}" for j in range(len(names))) | |
| print(f" {name:>12}{cells}{control[i]:>12.4f}{sep[i]:>8.2f}") | |
| summary[m] = { | |
| "matrix": matrix.tolist(), | |
| "separation": sep.tolist(), | |
| "static_control": control.tolist(), | |
| "diagonal_is_min": bool(all(matrix[i].argmin() == i for i in range(len(names)))), | |
| "min_separation": float(sep.min()), | |
| } | |
| passed = all( | |
| summary[m]["diagonal_is_min"] and summary[m]["min_separation"] > MIN_SEPARATION | |
| for m in SCORED_METRICS | |
| ) | |
| summary["passes_gate"] = passed | |
| worst = {m: round(summary[m]["min_separation"], 2) for m in SCORED_METRICS} | |
| print(f"\n GATE: {'PASS' if passed else 'FAIL'} worst separation per metric: {worst}" | |
| f"\n (need diagonal minimal on every row AND separation > {MIN_SEPARATION})") | |
| return summary | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) | |
| parser.add_argument( | |
| "--backend", action="append", choices=BACKENDS, default=None, | |
| help="Repeatable. Default: all three, so they can be compared in one table.", | |
| ) | |
| parser.add_argument("--refs-dir", default=str(DEFAULT_REFS)) | |
| parser.add_argument("--max-frames", type=int, default=41) | |
| parser.add_argument("--cache-dir", default=None) | |
| parser.add_argument("--output", default=None, help="Write the summary as JSON.") | |
| args = parser.parse_args() | |
| refs_dir = Path(args.refs_dir) | |
| refs = {p.stem: p for p in sorted(refs_dir.glob("*.mp4"))} | |
| if len(refs) < 2: | |
| raise SystemExit(f"Need >=2 reference clips to build a confusion matrix, found {len(refs)} in {refs_dir}") | |
| cache_dir = Path(args.cache_dir) if args.cache_dir else None | |
| summaries = [] | |
| for backend in args.backend or list(BACKENDS): | |
| try: | |
| matrices, controls, names = confusion(refs, backend, args.max_frames, cache_dir) | |
| except ImportError as exc: | |
| print(f"\nbackend {backend}: SKIPPED ({exc})") | |
| continue | |
| summaries.append(report(backend, matrices, controls, names)) | |
| if args.output: | |
| Path(args.output).write_text(json.dumps(summaries, indent=2)) | |
| print(f"\nSummary written to {args.output}") | |
| if __name__ == "__main__": | |
| main() | |
Xet Storage Details
- Size:
- 6.8 kB
- Xet hash:
- 5ddd8ff25008856763250dace55f924b409889a98d384fa393738266ac8e41b5
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.