Happy commited on
Commit
6536113
Β·
verified Β·
1 Parent(s): a00dd91

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +80 -66
app.py CHANGED
@@ -6,112 +6,138 @@ import torch
6
  import numpy as np
7
  from PIL import Image
8
 
9
- # ── Load OVI Model ──────────────────────────────────────────
10
  print("Loading OVI model...")
11
 
12
- from transformers import AutoProcessor, AutoModel
 
13
 
14
- MODEL_ID = "chetwinlow1/Ovi"
 
15
 
16
  try:
17
- processor = AutoProcessor.from_pretrained(MODEL_ID)
18
- model = AutoModel.from_pretrained(MODEL_ID, torch_dtype=torch.float16)
19
- device = "cuda" if torch.cuda.is_available() else "cpu"
20
- model = model.to(device)
21
- model.eval()
22
- print(f"βœ… OVI Model loaded on {device}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
23
  except Exception as e:
24
- print(f"❌ Model load error: {e}")
25
- processor = None
26
  model = None
 
27
 
28
 
29
  def generate_video(image, prompt):
30
- """Generate talking avatar video - FREE, no auth needed."""
31
-
32
  if image is None:
33
  raise gr.Error("Please upload an image!")
34
-
35
  if not prompt or prompt.strip() == "":
36
  raise gr.Error("Please enter a text prompt!")
37
-
38
  if model is None:
39
- raise gr.Error("Model not loaded. Please restart the Space.")
40
-
41
  try:
42
- # Process inputs
43
  if isinstance(image, str):
44
  pil_image = Image.open(image).convert("RGB")
45
  else:
46
  pil_image = Image.fromarray(image).convert("RGB")
47
-
48
- # Run model
49
- inputs = processor(
50
- text=prompt,
51
- images=pil_image,
52
- return_tensors="pt"
53
- ).to(device)
54
-
55
- with torch.no_grad():
56
- outputs = model.generate(**inputs)
57
-
58
- # Save output video
59
  output_path = tempfile.NamedTemporaryFile(suffix=".mp4", delete=False).name
60
-
61
- # Write video frames
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  if hasattr(outputs, 'video'):
63
- video_data = outputs.video[0].cpu().numpy()
64
  import cv2
65
- h, w = video_data.shape[1:3]
66
  writer = cv2.VideoWriter(
67
  output_path,
68
  cv2.VideoWriter_fourcc(*'mp4v'),
69
  24, (w, h)
70
  )
71
- for frame in video_data:
72
  frame_bgr = cv2.cvtColor(
73
  (frame * 255).astype(np.uint8),
74
  cv2.COLOR_RGB2BGR
75
  )
76
  writer.write(frame_bgr)
77
  writer.release()
78
-
 
 
 
 
 
 
 
 
 
 
79
  return output_path
80
-
81
  except Exception as e:
82
- raise gr.Error(f"Generation failed: {str(e)}")
83
 
84
 
85
- # ── Gradio Interface ────────────────────────────────────────
86
  with gr.Blocks(theme=gr.themes.Soft()) as demo:
87
-
88
  gr.Markdown("""
89
  # 🎬 OVI β€” Talking Avatar Generator
90
- **Free & Open Source** | No login required | Powered by OVI AI Model
91
  """)
92
-
93
  with gr.Row():
94
- with gr.Column(scale=1):
95
  image_input = gr.Image(
96
  label="πŸ“Έ Upload Image",
97
  type="filepath",
98
- sources=["upload", "clipboard"],
99
  height=300,
100
  )
101
  prompt_input = gr.Textbox(
102
  label="πŸ’¬ Text Prompt",
103
  lines=3,
104
- placeholder=(
105
- "Describe what you want. Example:\n"
106
- "A person looks at camera and says "
107
- "<S>Hello everyone, welcome!<E> "
108
- "<AUDCAP>Clear friendly voice<ENDAUDCAP>"
109
- ),
110
  )
111
  generate_btn = gr.Button("🎬 Generate Video", variant="primary", size="lg")
112
  clear_btn = gr.Button("πŸ—‘οΈ Clear", variant="secondary")
113
 
114
- with gr.Column(scale=1):
115
  video_output = gr.Video(
116
  label="πŸŽ₯ Generated Video",
117
  height=300,
@@ -119,30 +145,18 @@ with gr.Blocks(theme=gr.themes.Soft()) as demo:
119
  )
120
 
121
  gr.Markdown("""
122
- ---
123
- ### πŸ’‘ Prompt Tips:
124
- - `<S>Your speech here<E>` β€” what avatar says
125
- - `<AUDCAP>voice description<ENDAUDCAP>` β€” voice style
126
- - Example: `A man speaks to camera. <S>Hello world!<E> <AUDCAP>Deep calm voice<ENDAUDCAP>`
127
  """)
128
 
129
- gr.Examples(
130
- examples=[
131
- [None, "A person smiles at the camera. <S>Hello! Welcome to my channel.<E> <AUDCAP>Friendly energetic voice<ENDAUDCAP>"],
132
- [None, "A woman speaks confidently. <S>Today I want to share something amazing with you.<E> <AUDCAP>Clear professional voice<ENDAUDCAP>"],
133
- ],
134
- inputs=[image_input, prompt_input],
135
- )
136
-
137
  generate_btn.click(
138
  fn=generate_video,
139
  inputs=[image_input, prompt_input],
140
  outputs=[video_output],
141
  )
142
-
143
  clear_btn.click(
144
  fn=lambda: (None, "", None),
145
- inputs=None,
146
  outputs=[image_input, prompt_input, video_output],
147
  )
148
 
 
6
  import numpy as np
7
  from PIL import Image
8
 
 
9
  print("Loading OVI model...")
10
 
11
+ device = "cuda" if torch.cuda.is_available() else "cpu"
12
+ print(f"Device: {device}")
13
 
14
+ model = None
15
+ tokenizer = None
16
 
17
  try:
18
+ from transformers import AutoTokenizer
19
+ from huggingface_hub import hf_hub_download, snapshot_download
20
+ import sys
21
+
22
+ # Download full repo
23
+ repo_path = snapshot_download("chetwinlow1/Ovi")
24
+ sys.path.insert(0, repo_path)
25
+
26
+ # Try importing model directly from repo
27
+ try:
28
+ from modeling_ovi import OviModel
29
+ from processing_ovi import OviProcessor
30
+
31
+ processor = OviProcessor.from_pretrained("chetwinlow1/Ovi")
32
+ model = OviModel.from_pretrained(
33
+ "chetwinlow1/Ovi",
34
+ torch_dtype=torch.float16 if device == "cuda" else torch.float32,
35
+ ).to(device)
36
+ model.eval()
37
+ print("βœ… OVI loaded via custom classes!")
38
+
39
+ except ImportError:
40
+ # Fallback - try pipeline
41
+ from transformers import pipeline
42
+ pipe = pipeline(
43
+ "image-to-video",
44
+ model="chetwinlow1/Ovi",
45
+ device=0 if device == "cuda" else -1,
46
+ )
47
+ model = pipe
48
+ processor = None
49
+ print("βœ… OVI loaded via pipeline!")
50
+
51
  except Exception as e:
52
+ print(f"❌ Model error: {e}")
 
53
  model = None
54
+ processor = None
55
 
56
 
57
  def generate_video(image, prompt):
 
 
58
  if image is None:
59
  raise gr.Error("Please upload an image!")
 
60
  if not prompt or prompt.strip() == "":
61
  raise gr.Error("Please enter a text prompt!")
 
62
  if model is None:
63
+ raise gr.Error("Model not loaded!")
64
+
65
  try:
 
66
  if isinstance(image, str):
67
  pil_image = Image.open(image).convert("RGB")
68
  else:
69
  pil_image = Image.fromarray(image).convert("RGB")
70
+
 
 
 
 
 
 
 
 
 
 
 
71
  output_path = tempfile.NamedTemporaryFile(suffix=".mp4", delete=False).name
72
+
73
+ if processor is not None:
74
+ # Custom processor path
75
+ inputs = processor(
76
+ text=prompt,
77
+ images=pil_image,
78
+ return_tensors="pt"
79
+ ).to(device)
80
+ with torch.no_grad():
81
+ outputs = model.generate(**inputs)
82
+ else:
83
+ # Pipeline path
84
+ outputs = model(pil_image, prompt)
85
+
86
+ # Save video
87
  if hasattr(outputs, 'video'):
88
+ video_frames = outputs.video[0].cpu().numpy()
89
  import cv2
90
+ h, w = video_frames.shape[1:3]
91
  writer = cv2.VideoWriter(
92
  output_path,
93
  cv2.VideoWriter_fourcc(*'mp4v'),
94
  24, (w, h)
95
  )
96
+ for frame in video_frames:
97
  frame_bgr = cv2.cvtColor(
98
  (frame * 255).astype(np.uint8),
99
  cv2.COLOR_RGB2BGR
100
  )
101
  writer.write(frame_bgr)
102
  writer.release()
103
+ elif isinstance(outputs, str) and os.path.exists(outputs):
104
+ shutil.copy(outputs, output_path)
105
+ elif isinstance(outputs, list) and len(outputs) > 0:
106
+ out = outputs[0]
107
+ if isinstance(out, str) and os.path.exists(out):
108
+ shutil.copy(out, output_path)
109
+ elif hasattr(out, 'get'):
110
+ v = out.get('video') or out.get('path')
111
+ if v:
112
+ shutil.copy(v, output_path)
113
+
114
  return output_path
115
+
116
  except Exception as e:
117
+ raise gr.Error(f"Error: {str(e)}")
118
 
119
 
 
120
  with gr.Blocks(theme=gr.themes.Soft()) as demo:
 
121
  gr.Markdown("""
122
  # 🎬 OVI β€” Talking Avatar Generator
123
+ **Free & Open Source** | No login required
124
  """)
 
125
  with gr.Row():
126
+ with gr.Column():
127
  image_input = gr.Image(
128
  label="πŸ“Έ Upload Image",
129
  type="filepath",
 
130
  height=300,
131
  )
132
  prompt_input = gr.Textbox(
133
  label="πŸ’¬ Text Prompt",
134
  lines=3,
135
+ placeholder="A person speaks. <S>Hello world!<E> <AUDCAP>Clear voice<ENDAUDCAP>",
 
 
 
 
 
136
  )
137
  generate_btn = gr.Button("🎬 Generate Video", variant="primary", size="lg")
138
  clear_btn = gr.Button("πŸ—‘οΈ Clear", variant="secondary")
139
 
140
+ with gr.Column():
141
  video_output = gr.Video(
142
  label="πŸŽ₯ Generated Video",
143
  height=300,
 
145
  )
146
 
147
  gr.Markdown("""
148
+ ### πŸ’‘ Tips:
149
+ - `<S>speech here<E>` β€” what avatar says
150
+ - `<AUDCAP>voice style<ENDAUDCAP>` β€” voice description
 
 
151
  """)
152
 
 
 
 
 
 
 
 
 
153
  generate_btn.click(
154
  fn=generate_video,
155
  inputs=[image_input, prompt_input],
156
  outputs=[video_output],
157
  )
 
158
  clear_btn.click(
159
  fn=lambda: (None, "", None),
 
160
  outputs=[image_input, prompt_input, video_output],
161
  )
162