fishxinyu/OpenVE-3M-process / make_display_video.py
fishxinyu's picture
download
raw
4.06 kB
"""
For each row in local_add.csv and local_remove.csv where both the edited video
and the original video exist in the videos folder, produce a side-by-side video
(original left, edited right) and save it to display_video/.
"""
import csv
import subprocess
from pathlib import Path
BASE_DIR = Path("/home/xinyuy/dataset_processing/OpenVE-3M/OpenVE-3M")
VIDEOS_DIR = BASE_DIR / "videos"
CSV_DIR = BASE_DIR / "csv_files"
OUTPUT_DIR = Path("/home/xinyuy/dataset_processing/OpenVE-3M/display_video")
PREFIX_MAP = {
"local_add": "local_add",
"local_remove": "local_remove",
"global_style": "global_style_new",
"background_change": "background_change",
"local_change": "local_change",
}
CSV_FILES = [
CSV_DIR / "local_add.csv",
CSV_DIR / "local_remove.csv",
]
def resolve_path(csv_path: str) -> Path | None:
parts = csv_path.strip().split("/", 1)
if len(parts) != 2:
return None
folder_key, filename = parts
actual_folder = PREFIX_MAP.get(folder_key)
if actual_folder is None:
return None
return VIDEOS_DIR / actual_folder / filename
def collect_pairs():
pairs = []
seen = set()
for csv_file in CSV_FILES:
with open(csv_file, newline="", encoding="utf-8") as f:
reader = csv.DictReader(f, delimiter=";")
for row in reader:
edited_csv = row.get("video", "").strip()
original_csv = row.get("original_video", "").strip()
if not edited_csv or not original_csv:
continue
edited_path = resolve_path(edited_csv)
original_path = resolve_path(original_csv)
if edited_path is None or original_path is None:
continue
if not edited_path.exists() or not original_path.exists():
continue
key = (str(edited_path), str(original_path))
if key in seen:
continue
seen.add(key)
pairs.append((edited_path, original_path))
return pairs
def make_side_by_side_video(original_path: Path, edited_path: Path, out_path: Path) -> bool:
"""
Use ffmpeg to scale both inputs to the same height, then hstack them.
Stops at the shorter video. No audio output.
"""
cmd = [
"ffmpeg", "-y",
"-i", str(original_path),
"-i", str(edited_path),
"-filter_complex",
# Scale both to height=720 (width divisible by 2), then hstack
"[0:v]scale=-2:720[v0];[1:v]scale=-2:720[v1];[v0][v1]hstack=inputs=2",
"-c:v", "libx264",
"-crf", "23",
"-preset", "fast",
"-an",
"-shortest",
str(out_path),
]
result = subprocess.run(cmd, capture_output=True)
return result.returncode == 0
def main():
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
print("Scanning CSV files for valid pairs...", flush=True)
pairs = collect_pairs()
print(f"Found {len(pairs)} valid pairs where both videos exist.", flush=True)
if not pairs:
print("No pairs found. Exiting.")
return
ok_count = 0
skip_count = 0
for i, (edited_path, original_path) in enumerate(pairs):
out_name = edited_path.stem + "__" + original_path.stem + ".mp4"
out_path = OUTPUT_DIR / out_name
if out_path.exists():
ok_count += 1
if (i + 1) % 100 == 0:
print(f" Processed {i + 1}/{len(pairs)} pairs...", flush=True)
continue
success = make_side_by_side_video(original_path, edited_path, out_path)
if success:
ok_count += 1
else:
print(f" [WARN] ffmpeg failed for pair {original_path.name} / {edited_path.name}", flush=True)
skip_count += 1
if (i + 1) % 100 == 0:
print(f" Processed {i + 1}/{len(pairs)} pairs...", flush=True)
print(f"\nDone. Saved {ok_count} videos to {OUTPUT_DIR} (skipped {skip_count}).")
if __name__ == "__main__":
main()

Xet Storage Details

Size:
4.06 kB
·
Xet hash:
53f40ae44aa15d6bc1d153eeb55fd45368baef424d106d0b20479ebb3cb33c15

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.