| #!/usr/bin/env python3 | |
| """Randomly select videos present in captions.json, the original videos dir, | |
| pose/, and pose_no_face/, then materialize them into a flat pose-control | |
| dataset directory (mirroring the layout of Pose-Control-Dataset): each | |
| selected video becomes a {stem}.mp4 / {stem}_pose.mp4 pair plus a | |
| dataset.json describing media_path, reference_path, and caption. | |
| """ | |
| import argparse | |
| import json | |
| import os | |
| import random | |
| import shutil | |
| import sys | |
| from concurrent.futures import ThreadPoolExecutor, as_completed | |
| from pathlib import Path | |
| _HERE = Path(__file__).resolve().parent | |
| _HUMANVID_ROOT = _HERE.parent | |
| def find_eligible_stems(videos_dir: Path, captions: dict, pose_dir: Path, pose_no_face_dir: Path) -> list[str]: | |
| """Return stems present in videos_dir, captions, pose_dir, and pose_no_face_dir.""" | |
| eligible = [] | |
| for video_path in sorted(videos_dir.glob("*.mp4")): | |
| stem = video_path.stem | |
| if stem not in captions: | |
| continue | |
| if not (pose_dir / f"{stem}_pose.mp4").exists(): | |
| continue | |
| if not (pose_no_face_dir / f"{stem}_pose.mp4").exists(): | |
| continue | |
| eligible.append(stem) | |
| return eligible | |
| def link_or_copy(src: Path, dst: Path, mode: str) -> None: | |
| dst.unlink(missing_ok=True) | |
| if mode == "symlink": | |
| dst.symlink_to(src.resolve()) | |
| elif mode == "hardlink": | |
| try: | |
| os.link(src, dst) | |
| except OSError: | |
| shutil.copy2(src, dst) | |
| else: | |
| shutil.copy2(src, dst) | |
| def materialize_pair(stem: str, media_src: Path, reference_src: Path, output_dir: Path, link_mode: str): | |
| try: | |
| link_or_copy(media_src, output_dir / f"{stem}.mp4", link_mode) | |
| link_or_copy(reference_src, output_dir / f"{stem}_pose.mp4", link_mode) | |
| return stem, True, None | |
| except OSError as e: | |
| return stem, False, str(e) | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--videos-dir", type=Path, | |
| default=_HUMANVID_ROOT / "selected_videos", | |
| help="Directory containing the original videos and captions.json") | |
| parser.add_argument("--captions", type=Path, default=None, | |
| help="Path to captions.json (default: <videos-dir>/captions.json)") | |
| parser.add_argument("--pose-dir", type=Path, default=None, | |
| help="Directory of {stem}_pose.mp4 files with faces (default: <videos-dir>/pose)") | |
| parser.add_argument("--pose-no-face-dir", type=Path, default=None, | |
| help="Directory of {stem}_pose.mp4 files without faces " | |
| "(default: <videos-dir>/pose_no_face)") | |
| parser.add_argument("--reference-path", "--reference_path", dest="reference_path", | |
| help="Which pose folder to copy as each video's reference_path: " | |
| "'pose', 'pose_no_face', or an arbitrary directory containing " | |
| "{stem}_pose.mp4 files (default: pose_no_face)") | |
| parser.add_argument("--output", type=Path, required=True, | |
| help="Output dataset directory to create") | |
| parser.add_argument("--num-videos", type=int, default=2000, | |
| help="Number of videos to randomly select (default: 2000)") | |
| parser.add_argument("--seed", type=int, default=0, | |
| help="Random seed for reproducible selection") | |
| parser.add_argument("--link-mode", choices=["copy", "symlink", "hardlink"], default="hardlink", | |
| help="How to materialize files into --output (default: hardlink; " | |
| "falls back to a real copy if hardlinking fails, e.g. across filesystems)") | |
| parser.add_argument("--workers", type=int, default=16, | |
| help="Number of concurrent copy/link workers") | |
| args = parser.parse_args() | |
| captions_path = args.captions or args.videos_dir / "captions.json" | |
| pose_dir = args.pose_dir or args.videos_dir / "pose" | |
| pose_no_face_dir = args.pose_no_face_dir or args.videos_dir / "pose_no_face" | |
| if not args.videos_dir.is_dir(): | |
| sys.exit(f"videos dir not found: {args.videos_dir}") | |
| if not captions_path.is_file(): | |
| sys.exit(f"captions file not found: {captions_path}") | |
| if not pose_dir.is_dir(): | |
| sys.exit(f"pose dir not found: {pose_dir}") | |
| if not pose_no_face_dir.is_dir(): | |
| sys.exit(f"pose_no_face dir not found: {pose_no_face_dir}") | |
| captions = json.loads(captions_path.read_text(encoding="utf-8")) | |
| eligible = find_eligible_stems(args.videos_dir, captions, pose_dir, pose_no_face_dir) | |
| print(f"Found {len(eligible)} videos present in captions.json, {args.videos_dir.name}/, " | |
| f"pose/, and pose_no_face/", file=sys.stderr) | |
| if len(eligible) < args.num_videos: | |
| print(f"Warning: only {len(eligible)} eligible videos available, " | |
| f"fewer than requested {args.num_videos}. Using all of them.", file=sys.stderr) | |
| selected = eligible | |
| else: | |
| rng = random.Random(args.seed) | |
| selected = rng.sample(eligible, args.num_videos) | |
| selected.sort() | |
| if args.reference_path in (None, "pose_no_face"): | |
| reference_dir = pose_no_face_dir | |
| elif args.reference_path == "pose": | |
| reference_dir = pose_dir | |
| else: | |
| reference_dir = Path(args.reference_path) | |
| if not reference_dir.is_dir(): | |
| sys.exit(f"reference dir not found: {reference_dir}") | |
| args.output.mkdir(parents=True, exist_ok=True) | |
| results = {} | |
| with ThreadPoolExecutor(max_workers=args.workers) as pool: | |
| futures = [ | |
| pool.submit( | |
| materialize_pair, | |
| stem, | |
| args.videos_dir / f"{stem}.mp4", | |
| reference_dir / f"{stem}_pose.mp4", | |
| args.output, | |
| args.link_mode, | |
| ) | |
| for stem in selected | |
| ] | |
| done = 0 | |
| for fut in as_completed(futures): | |
| stem, ok, err = fut.result() | |
| results[stem] = (ok, err) | |
| done += 1 | |
| print(f"\r{done}/{len(selected)} materialized", end="", file=sys.stderr) | |
| print(file=sys.stderr) | |
| dataset = [] | |
| failed = [] | |
| for stem in selected: | |
| ok, err = results[stem] | |
| if not ok: | |
| failed.append((stem, err)) | |
| continue | |
| dataset.append({ | |
| "caption": captions[stem], | |
| "media_path": f"{stem}.mp4", | |
| "reference_path": f"{stem}_pose.mp4", | |
| }) | |
| if failed: | |
| print(f"Warning: {len(failed)} video(s) failed to materialize:", file=sys.stderr) | |
| for stem, err in failed: | |
| print(f" {stem}: {err}", file=sys.stderr) | |
| dataset_json_path = args.output / "dataset.json" | |
| dataset_json_path.write_text(json.dumps(dataset, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") | |
| print(f"Done. Wrote {len(dataset)} pairs to {args.output} " | |
| f"(reference source: {reference_dir}, link mode: {args.link_mode}), " | |
| f"manifest at {dataset_json_path}") | |
| if __name__ == "__main__": | |
| main() | |
Xet Storage Details
- Size:
- 7.19 kB
- Xet hash:
- 8a2a62ac90cde4d38838ba87b0460b67ed8bb7cb723d21cf50a05f1b85300ed1
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.