Buckets:
| """ | |
| For each row in local_add.csv and local_remove.csv where both the edited video | |
| and the original video exist in the videos folder, produce a side-by-side video | |
| (original left, edited right) and save it to display_video/. | |
| """ | |
| import csv | |
| import subprocess | |
| from pathlib import Path | |
| BASE_DIR = Path("/home/xinyuy/dataset_processing/OpenVE-3M/OpenVE-3M") | |
| VIDEOS_DIR = BASE_DIR / "videos" | |
| CSV_DIR = BASE_DIR / "csv_files" | |
| OUTPUT_DIR = Path("/home/xinyuy/dataset_processing/OpenVE-3M/display_video") | |
| PREFIX_MAP = { | |
| "local_add": "local_add", | |
| "local_remove": "local_remove", | |
| "global_style": "global_style_new", | |
| "background_change": "background_change", | |
| "local_change": "local_change", | |
| } | |
| CSV_FILES = [ | |
| CSV_DIR / "local_add.csv", | |
| CSV_DIR / "local_remove.csv", | |
| ] | |
| def resolve_path(csv_path: str) -> Path | None: | |
| parts = csv_path.strip().split("/", 1) | |
| if len(parts) != 2: | |
| return None | |
| folder_key, filename = parts | |
| actual_folder = PREFIX_MAP.get(folder_key) | |
| if actual_folder is None: | |
| return None | |
| return VIDEOS_DIR / actual_folder / filename | |
| def collect_pairs(): | |
| pairs = [] | |
| seen = set() | |
| for csv_file in CSV_FILES: | |
| with open(csv_file, newline="", encoding="utf-8") as f: | |
| reader = csv.DictReader(f, delimiter=";") | |
| for row in reader: | |
| edited_csv = row.get("video", "").strip() | |
| original_csv = row.get("original_video", "").strip() | |
| if not edited_csv or not original_csv: | |
| continue | |
| edited_path = resolve_path(edited_csv) | |
| original_path = resolve_path(original_csv) | |
| if edited_path is None or original_path is None: | |
| continue | |
| if not edited_path.exists() or not original_path.exists(): | |
| continue | |
| key = (str(edited_path), str(original_path)) | |
| if key in seen: | |
| continue | |
| seen.add(key) | |
| pairs.append((edited_path, original_path)) | |
| return pairs | |
| def make_side_by_side_video(original_path: Path, edited_path: Path, out_path: Path) -> bool: | |
| """ | |
| Use ffmpeg to scale both inputs to the same height, then hstack them. | |
| Stops at the shorter video. No audio output. | |
| """ | |
| cmd = [ | |
| "ffmpeg", "-y", | |
| "-i", str(original_path), | |
| "-i", str(edited_path), | |
| "-filter_complex", | |
| # Scale both to height=720 (width divisible by 2), then hstack | |
| "[0:v]scale=-2:720[v0];[1:v]scale=-2:720[v1];[v0][v1]hstack=inputs=2", | |
| "-c:v", "libx264", | |
| "-crf", "23", | |
| "-preset", "fast", | |
| "-an", | |
| "-shortest", | |
| str(out_path), | |
| ] | |
| result = subprocess.run(cmd, capture_output=True) | |
| return result.returncode == 0 | |
| def main(): | |
| OUTPUT_DIR.mkdir(parents=True, exist_ok=True) | |
| print("Scanning CSV files for valid pairs...", flush=True) | |
| pairs = collect_pairs() | |
| print(f"Found {len(pairs)} valid pairs where both videos exist.", flush=True) | |
| if not pairs: | |
| print("No pairs found. Exiting.") | |
| return | |
| ok_count = 0 | |
| skip_count = 0 | |
| for i, (edited_path, original_path) in enumerate(pairs): | |
| out_name = edited_path.stem + "__" + original_path.stem + ".mp4" | |
| out_path = OUTPUT_DIR / out_name | |
| if out_path.exists(): | |
| ok_count += 1 | |
| if (i + 1) % 100 == 0: | |
| print(f" Processed {i + 1}/{len(pairs)} pairs...", flush=True) | |
| continue | |
| success = make_side_by_side_video(original_path, edited_path, out_path) | |
| if success: | |
| ok_count += 1 | |
| else: | |
| print(f" [WARN] ffmpeg failed for pair {original_path.name} / {edited_path.name}", flush=True) | |
| skip_count += 1 | |
| if (i + 1) % 100 == 0: | |
| print(f" Processed {i + 1}/{len(pairs)} pairs...", flush=True) | |
| print(f"\nDone. Saved {ok_count} videos to {OUTPUT_DIR} (skipped {skip_count}).") | |
| if __name__ == "__main__": | |
| main() | |
Xet Storage Details
- Size:
- 4.06 kB
- Xet hash:
- 53f40ae44aa15d6bc1d153eeb55fd45368baef424d106d0b20479ebb3cb33c15
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.