Download Aitool from Zr7Ai/tool: direct link, hf CLI and curl.
- Browser
- Download file 6.02 kB
-
https://huggingface.co/Zr7Ai/tool/resolve/main/Aitool
- Command line
-
hf download hf://Zr7Ai/tool/Aitool
-
curl -L -o Aitool https://huggingface.co/Zr7Ai/tool/resolve/main/Aitool
6.02 kB
| # -*- coding: utf-8 -*- | |
| """AI Video Generator - Complete Text to Video + Audio System""" | |
| # Install dependencies | |
| print("π¦ Installing dependencies...") | |
| !pip install diffusers accelerate torch transformers coqui-tts opencv-python pillow ffmpeg-python xformers -q | |
| !apt update && apt install -y ffmpeg > /dev/null 2>&1 | |
| import torch | |
| import numpy as np | |
| from diffusers import StableVideoDiffusionPipeline | |
| from coqui_tts import TTS | |
| import subprocess | |
| import os | |
| from IPython.display import display, HTML | |
| import base64 | |
| import cv2 | |
| import ffmpeg | |
| from pathlib import Path | |
| class AIVideoGenerator: | |
| def __init__(self): | |
| self.video_pipe = None | |
| self.tts_model = None | |
| def load_models(self): | |
| """Load AI models""" | |
| print("π Loading models...") | |
| # Load video generation model | |
| self.video_pipe = StableVideoDiffusionPipeline.from_pretrained( | |
| "stabilityai/stable-video-diffusion-img2vid-xt", | |
| torch_dtype=torch.float16, | |
| variant="fp16" | |
| ) | |
| self.video_pipe.enable_model_cpu_offload() | |
| # Load TTS model | |
| self.tts_model = TTS(model_name="tts_models/en/ljspeech/tacotron2-DDC", progress_bar=False) | |
| print("β Models loaded successfully!") | |
| def create_video_from_frames(self, frames, output_path, fps=8): | |
| """Convert frames to video using OpenCV""" | |
| if not frames: | |
| return None | |
| height, width = frames[0].shape[:2] | |
| fourcc = cv2.VideoWriter_fourcc(*'mp4v') | |
| out = cv2.VideoWriter(output_path, fourcc, fps, (width, height)) | |
| for frame in frames: | |
| # Convert RGB to BGR for OpenCV | |
| frame_bgr = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR) | |
| out.write(frame_bgr) | |
| out.release() | |
| return output_path | |
| def generate_video(self, prompt, duration=4, seed=42): | |
| """Generate video from text prompt""" | |
| print(f"π¬ Generating video: {prompt}") | |
| if self.video_pipe is None: | |
| self.load_models() | |
| generator = torch.manual_seed(seed) | |
| frames = self.video_pipe( | |
| prompt, | |
| num_frames=duration * 8, | |
| generator=generator, | |
| decode_chunk_size=4 | |
| ).frames[0] | |
| video_path = "generated_video.mp4" | |
| self.create_video_from_frames(frames, video_path) | |
| return video_path | |
| def generate_audio(self, text, output_path="generated_audio.wav"): | |
| """Generate audio from text using TTS""" | |
| print(f"π Generating audio: {text}") | |
| if self.tts_model is None: | |
| self.load_models() | |
| try: | |
| self.tts_model.tts_to_file(text=text, file_path=output_path) | |
| return output_path | |
| except Exception as e: | |
| print(f"β TTS failed: {e}") | |
| return None | |
| def merge_audio_video(self, video_path, audio_path, output_path="final_video.mp4"): | |
| """Merge audio and video using FFmpeg""" | |
| print("π Merging audio and video...") | |
| try: | |
| input_video = ffmpeg.input(video_path) | |
| input_audio = ffmpeg.input(audio_path) | |
| ffmpeg.output( | |
| input_video, | |
| input_audio, | |
| output_path, | |
| vcodec='libx264', | |
| acodec='aac', | |
| strict='experimental' | |
| ).overwrite_output().run(quiet=True) | |
| return output_path | |
| except Exception as e: | |
| print(f"β Merge failed: {e}") | |
| return video_path | |
| def display_video(self, video_path): | |
| """Display video in Colab""" | |
| try: | |
| with open(video_path, 'rb') as f: | |
| video_bytes = f.read() | |
| video_b64 = base64.b64encode(video_bytes).decode() | |
| html = f''' | |
| <video width="640" height="480" controls autoplay> | |
| <source src="data:video/mp4;base64,{video_b64}" type="video/mp4"> | |
| </video> | |
| ''' | |
| display(HTML(html)) | |
| except Exception as e: | |
| print(f"β Display failed: {e}") | |
| # MAIN EXECUTION | |
| def main(): | |
| generator = AIVideoGenerator() | |
| # Configuration | |
| prompt = "A beautiful sunset over mountains, cinematic style, 4K, high quality" | |
| audio_text = "This is a beautiful sunset scene generated by artificial intelligence." | |
| duration = 4 | |
| seed = 42 | |
| print("π Starting AI Video Generation Pipeline...") | |
| try: | |
| # Generate video | |
| video_path = generator.generate_video(prompt, duration, seed) | |
| # Generate audio | |
| audio_path = generator.generate_audio(audio_text) | |
| # Merge if audio generated successfully | |
| if audio_path and os.path.exists(audio_path): | |
| final_path = generator.merge_audio_video(video_path, audio_path) | |
| else: | |
| final_path = video_path | |
| # Display result | |
| print(f"β Final video saved: {final_path}") | |
| generator.display_video(final_path) | |
| # Provide download link | |
| if os.path.exists(final_path): | |
| file_size = os.path.getsize(final_path) / (1024 * 1024) | |
| print(f"π File size: {file_size:.2f} MB") | |
| except Exception as e: | |
| print(f"β Generation failed: {e}") | |
| finally: | |
| # Cleanup | |
| print("π§Ή Cleaning up temporary files...") | |
| for path in ["generated_video.mp4", "generated_audio.wav", "final_video.mp4"]: | |
| if os.path.exists(path): | |
| os.remove(path) | |
| # Run the complete pipeline | |
| if __name__ == "__main__": | |
| main() | |
| fastapi==0.104.1 | |
| uvicorn==0.24.0 | |
| diffusers==0.24.0 | |
| accelerate==0.24.1 | |
| torch==2.1.0 | |
| transformers==4.35.0 | |
| coqui-tts==0.11.0 | |
| ffmpeg-python==0.2.0 | |
| python-multipart==0.0.6 | |
| aiofiles==23.2.1 | |
| pillow==10.0.1 | |
| numpy==1.24.3 | |
| opencv-python==4.8.1.78 | |
| python-dotenv==1.0.0 |