junaid-simamdigital's picture
Add bounded FFmpeg video frame sampling
a0cbfcf verified
Raw History Blame Contribute Delete
5.3 kB
"""General media ingest and honest Gaussian-preview workbench."""
from __future__ import annotations
import hashlib
import json
import shutil
import subprocess
import sys
import tempfile
import time
from pathlib import Path
for _stream in (sys.stdout, sys.stderr):
if hasattr(_stream, "reconfigure"):
_stream.reconfigure(encoding="utf-8", errors="replace")
import gradio as gr
import numpy as np
from PIL import Image
from core import image_billboard_points, write_preview_ply
def _paths(items):
if not items:
return []
if isinstance(items, (str, Path)):
return [str(items)]
return [str(getattr(item, "path", None) or getattr(item, "name", item)) for item in items]
def _hash_file(path):
digest = hashlib.sha256()
size = 0
with Path(path).open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
size += len(chunk)
return {"name": Path(path).name, "bytes": size, "sha256": digest.hexdigest()}
def _sample_video(video_path: str, max_frames: int = 8) -> list[Path]:
"""Extract a bounded set of frames using the FFmpeg bundled by Spaces."""
frame_dir = Path(tempfile.mkdtemp(prefix="simamanything2gs_"))
pattern = frame_dir / "frame_%02d.jpg"
try:
subprocess.run(
["ffmpeg", "-hide_banner", "-loglevel", "error", "-i", str(video_path),
"-vf", "fps=10", "-frames:v", str(max_frames), "-q:v", "3", str(pattern)],
check=True,
capture_output=True,
text=True,
timeout=120,
)
except FileNotFoundError as exc:
raise gr.Error("FFmpeg is unavailable; this Space cannot sample video frames yet.") from exc
except (subprocess.CalledProcessError, subprocess.TimeoutExpired) as exc:
detail = getattr(exc, "stderr", "") or "invalid or unreadable video"
raise gr.Error(f"Video frame extraction failed: {str(detail).strip()[-300:]}") from exc
frames = sorted(frame_dir.glob("frame_*.jpg"))
if not frames:
raise gr.Error("No readable frames were found in the uploaded video.")
return frames
def ingest(images, videos, mode):
image_paths, video_paths = _paths(images), _paths(videos)
if not image_paths and not video_paths:
raise gr.Error("Add at least one image or video.")
sampled_frames = []
if image_paths:
first_path = image_paths[0]
else:
sampled_frames = _sample_video(video_paths[0])
first_path = sampled_frames[0]
first = Image.open(first_path).convert("RGB")
if sampled_frames:
shutil.rmtree(sampled_frames[0].parent, ignore_errors=True)
points, colors = image_billboard_points(first)
out_dir = Path("output")
out_dir.mkdir(exist_ok=True)
ply_path = write_preview_ply(out_dir / "simamanything2gs_preview.ply", points, colors)
manifest = {
"project": "SimamAnything2GS",
"version": "0.1.0",
"created_at_unix": int(time.time()),
"mode": mode,
"images": [_hash_file(path) for path in image_paths],
"videos": [_hash_file(path) for path in video_paths],
"sampled_frame_count": len(sampled_frames),
"sampling": "FFmpeg fps=10, max 8 frames" if sampled_frames else "not needed",
"preview_points": int(len(points)),
"preview_type": "2.5D image billboard; not a learned 3D reconstruction",
"limitations": [
"The preview places pixels from the first image or sampled video frame on a single plane at z=1.",
"No depth, camera pose, occlusion completion, or Gaussian optimization is claimed.",
"Real image/video-to-Gaussian backends must be evaluated separately on held-out views.",
],
}
manifest_path = out_dir / "simamanything2gs_manifest.json"
manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return str(ply_path), str(manifest_path), json.dumps(manifest, indent=2), "Preview exported; use it only as an ingest/export smoke test."
with gr.Blocks(title="SimamAnything2GS") as demo:
gr.Markdown("# SimamAnything2GS\nImage/video ingest toward Gaussian reconstruction")
gr.Markdown(
"This first slice accepts arbitrary media and exports a deterministic 2.5D Gaussian-style preview. "
"Depth, pose estimation, and learned Gaussian optimization are explicit future adapters."
)
with gr.Row():
with gr.Column():
images = gr.Files(label="Images", file_types=["image"], file_count="multiple", type="filepath")
videos = gr.Files(label="Optional videos", file_types=["video"], file_count="multiple", type="filepath")
mode = gr.Dropdown(["CPU ingest preview", "Future depth + pose backend"], value="CPU ingest preview", label="Pipeline mode")
run = gr.Button("Build Gaussian preview", variant="primary")
with gr.Column():
ply = gr.File(label="Preview PLY")
manifest = gr.File(label="Manifest JSON")
report = gr.Code(label="Run report", language="json")
status = gr.Markdown()
run.click(ingest, [images, videos, mode], [ply, manifest, report, status])
if __name__ == "__main__":
demo.launch(show_error=True)