davanstrien/prelinger-sample / scripts /fetch_more_clips.py
davanstrien's picture
download
raw
2.65 kB
# /// script
# requires-python = ">=3.11"
# dependencies = ["requests"]
# ///
"""Fetch more Prelinger films from IA, cut 12x60s clips each into stress_clips/."""
import json
import subprocess
import sys
from pathlib import Path
import requests
BASE = Path(__file__).parent
OUT = BASE / "stress_clips"
search = requests.get(
"https://archive.org/advancedsearch.php",
params={"q": "collection:prelinger AND mediatype:movies",
"fl[]": "identifier", "sort[]": "downloads desc",
"rows": 20, "output": "json"},
timeout=60).json()
ids = [d["identifier"] for d in search["response"]["docs"]][6:]
made = 0
for ident in ids:
try:
meta = requests.get(f"https://archive.org/metadata/{ident}", timeout=60).json()
except (requests.RequestException, json.JSONDecodeError) as e:
print(f"SKIP {ident}: metadata failed ({e})", flush=True)
continue
mp4s = sorted((f["name"] for f in meta.get("files", []) if f["name"].endswith(".mp4")),
key=lambda n: ("512kb" not in n, n))
if not mp4s:
print(f"SKIP {ident}: no mp4 derivative", flush=True)
continue
raw = BASE / f"raw_{ident}.mp4"
if not raw.exists():
print(f"GET {ident} -> {mp4s[0]}", flush=True)
r = requests.get(f"https://archive.org/download/{ident}/{mp4s[0]}",
timeout=600, stream=True)
r.raise_for_status()
with open(raw, "wb") as f:
for chunk in r.iter_content(1 << 20):
f.write(chunk)
probe = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "csv=p=0", str(raw)], capture_output=True, text=True)
try:
dur = int(float(probe.stdout.strip()))
except ValueError:
print(f"SKIP {ident}: unprobeable", flush=True)
continue
if dur < 200:
print(f"SKIP {ident}: too short ({dur}s)", flush=True)
continue
step = max(10, (dur - 90) // 12)
for i in range(12):
off = 15 + i * step
if off + 60 > dur:
break
dest = OUT / f"{ident}_{off:03d}.mp4"
if dest.exists():
continue
subprocess.run(
["ffmpeg", "-hide_banner", "-loglevel", "error", "-ss", str(off),
"-i", str(raw), "-t", "60", "-c:v", "libx264", "-preset", "fast",
"-crf", "26", "-an", "-vf", "scale=640:-2", "-y", str(dest)],
check=True, stdin=subprocess.DEVNULL)
made += 1
print(f"OK {ident}: dur={dur}s", flush=True)
total = len(list(OUT.glob("*.mp4")))
print(f"DONE: +{made} new clips, {total} total", flush=True)

Xet Storage Details

Size:
2.65 kB
·
Xet hash:
70f468df1007ccf9bdb8b6f3ebf6a3702a18a387d3113e12ddb3033892b230e0

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.