#!/usr/bin/env python3 """compose_2x3.py -- 2x3 comparison video: row 1 = the ORIGINAL episode's three dataset camera views, row 2 = the augmented (perturbed + re-earned) rollout's same three views. The shorter row freezes on its last frame. compose_2x3.py [dataset_root] """ import sys from pathlib import Path import imageio.v2 as iio import numpy as np from actaug_core import DEFAULT_DATASET, write_mp4 VIEWS = ["robot0_agentview_left", "robot0_agentview_right", "robot0_eye_in_hand"] def read_video(path): return [f for f in iio.get_reader(str(path))] def main(ep_idx, dump_dir, out_path, dataset_root=str(DEFAULT_DATASET)): ep_idx = int(ep_idx) root = Path(dataset_root) chunk = ep_idx // 1000 orig = [read_video(root / "videos" / f"chunk-{chunk:03d}" / f"observation.images.{v}" / f"episode_{ep_idx:06d}.mp4") for v in VIEWS] dump = Path(dump_dir) aug = [read_video(dump / f"{n}.mp4") for n in ("left", "right", "wrist")] H, W = aug[0][0].shape[:2] orig = [[np.asarray(f)[..., :3] for f in v] for v in orig] aug = [[np.asarray(f)[..., :3] for f in v] for v in aug] # resize originals to the rollout render size if they differ if orig[0][0].shape[:2] != (H, W): import cv2 orig = [[cv2.resize(f, (W, H)) for f in v] for v in orig] T = max(len(orig[0]), len(aug[0])) pad = lambda v: v + [v[-1]] * (T - len(v)) orig, aug = [pad(v) for v in orig], [pad(v) for v in aug] frames = [np.concatenate([np.concatenate([o[t] for o in orig], axis=1), np.concatenate([a[t] for a in aug], axis=1)], axis=0) for t in range(T)] write_mp4(out_path, frames, fps=20) print(f"wrote {out_path} ({T} frames, {frames[0].shape[1]}x{frames[0].shape[0]})") if __name__ == "__main__": main(*sys.argv[1:])