import json import os import time import gradio as gr from typing import Dict, Any, Tuple # Sample prompt templates based on joeygambino/MiniMax-H3-Multishot-Workflow rules DEFAULT_SHOTS = """[Shot 1] A detective in a dark trenchcoat walks down a dimly lit, rain-slicked alleyway in Neo-Tokyo. Neon signs reflect off the wet asphalt. --- [Shot 2] The detective stops under a blue neon bar sign, reaches into his coat pocket, and pulls out a brass lighter. --- [Shot 3] He flicks the lighter open. The flame briefly illuminates his sharp jawline and worn facial expression before a loud thunder crack echoes in the background.""" def parse_shot_script(script_text: str) -> list[str]: """Splits multi-shot prompt scripts delimited by '---'.""" shots = [shot.strip() for shot in script_text.split("---") if shot.strip()] return shots if shots else ["A cinematic scene with continuous motion."] def generate_multishot_video( prompts: str, resolution: str = "1280x736", continuity_mode: str = "first_frame", take_seconds: int = 15, fps: int = 24, steps: int = 30, ) -> Tuple[str, str, Dict[str, Any]]: """ Main function for multi-shot video generation. Exposed as an agent-callable API via Gradio. """ shots = parse_shot_script(prompts) width, height = map(int, resolution.split("x")) # Payload structured according to MiniMax-H3-Multishot node spec pipeline_payload = { "workflow_version": "2.7.0", "shots_count": len(shots), "shots": shots, "master_controls": { "width": width, "height": height, "fps": fps, "steps": steps, "take_seconds": take_seconds, "continuity": continuity_mode, } } # Simulated execution pipeline (connects to backend GPU ComfyUI instance/Inference Endpoint) time.sleep(2) # Pipeline latency placeholder # Summary response execution_summary = { "status": "success", "processed_shots": len(shots), "total_duration_sec": take_seconds, "resolution": resolution, "continuity_applied": continuity_mode, "payload": pipeline_payload } # Dummy video path output for demo structure output_video = None # Replace with actual output file path when backend runner is attached audio_track = None return output_video, audio_track, execution_summary # Gradio UI Design with gr.Blocks(title="MiniMax-H3 Multi-Shot Studio", theme=gr.themes.Soft()) as demo: gr.Markdown( """ # 🎬 MiniMax-H3 Seamless Multi-Shot Studio Create continuous, multi-shot video takes with unified audio and character continuity using the **MiniMax-H3-Multishot** workflow. """ ) with gr.Row(): with gr.Column(scale=2): prompts_input = gr.Textbox( label="Multi-Shot Script (Separate shots with '---')", value=DEFAULT_SHOTS, lines=10, placeholder="Write Shot 1...\n---\nWrite Shot 2...", ) with gr.Accordion("⚙️ Master Controls & Parameters", open=True): with gr.Row(): resolution_dropdown = gr.Dropdown( choices=["1280x736", "1024x576", "768x512"], value="1280x736", label="Resolution" ) continuity_radio = gr.Radio( choices=["first_frame", "context_pin"], value="first_frame", label="Continuity Mode", info="first_frame = zero extra deps | context_pin = latent memory alignment" ) with gr.Row(): take_sec_slider = gr.Slider( minimum=5, maximum=60, value=15, step=5, label="Take Duration (Seconds)" ) steps_slider = gr.Slider( minimum=15, maximum=50, value=30, step=1, label="Sampling Steps" ) generate_btn = gr.Button("🚀 Generate Multi-Shot Take", variant="primary") with gr.Column(scale=2): video_output = gr.Video(label="Generated Master Video") audio_output = gr.Audio(label="Master Audio Track") status_json = gr.JSON(label="Pipeline Execution & Agent Metadata") # Wire up the Gradio button action & API endpoint generate_btn.click( fn=generate_multishot_video, inputs=[ prompts_input, resolution_dropdown, continuity_radio, take_sec_slider, steps_slider ], outputs=[video_output, audio_output, status_json], api_name="generate_multishot_video" # Exposed for AI Agents ) gr.Markdown( """ --- ### 🤖 For AI Agents Access the automated `/agents.md` endpoint of this Space or query `/info` to retrieve JSON signatures. """ ) if __name__ == "__main__": demo.launch()