Abdullahcoder54 commited on
Commit
986d8aa
·
1 Parent(s): 659dad6

Thread HF_TOKEN through gated Lightricks loads; local GGUF first; fail-fast fingerprint

Browse files
Files changed (1) hide show
  1. app.py +95 -17
app.py CHANGED
@@ -20,9 +20,11 @@ is a hard no-op here so no rewrite happens.
20
 
21
  from __future__ import annotations
22
 
 
23
  import os
24
  import subprocess
25
  import threading
 
26
 
27
  # ZeroGPU rule #1: `import spaces` must precede any CUDA-touching import
28
  # (torch etc.) — it monkey-patches torch.cuda at import time.
@@ -31,15 +33,22 @@ import spaces # noqa: E402
31
  import numpy as np
32
  import torch
33
 
34
- HF_TOKEN = os.environ.get("HF_TOKEN", "")
 
 
 
 
35
 
36
  # Direct download URL of the distilled GGUF transformer (~15GB). The backend
37
- # only needs LTX_SPACE_URL; this key is configured in the Space.
 
38
  LTX_MODEL_URL = os.environ.get(
39
  "LTX_MODEL_URL",
40
  "https://huggingface.co/realrebelai/LTX-2.5_GGUFs/resolve/main/LTX-2.5-Distilled-Q4_K_M.gguf",
41
  )
42
  PIPELINE_ID = "Lightricks/LTX-2.5-Diffusers"
 
 
43
  SEED = int(os.environ.get("LTX_SEED", "0"))
44
  FPS = int(os.environ.get("LTX_FPS", "24"))
45
 
@@ -60,6 +69,49 @@ _model = None
60
  # ----------------------------------------------------------------- model load
61
 
62
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
63
  def _download_gguf(dest: str) -> str:
64
  import httpx
65
 
@@ -76,7 +128,17 @@ def _download_gguf(dest: str) -> str:
76
 
77
  # Support "repo_id:filename" shorthand.
78
  repo, _, filename = LTX_MODEL_URL.partition(":")
79
- return hf_hub_download(repo, filename or "LTX-2.5-Distilled-Q4_K_M.gguf", token=HF_TOKEN)
 
 
 
 
 
 
 
 
 
 
80
 
81
 
82
  def _build_transformer(gguf_path: str):
@@ -97,11 +159,14 @@ def _build_transformer(gguf_path: str):
97
  quantization_config=GGUFQuantizationConfig(compute_dtype=torch.bfloat16),
98
  dtype=torch.bfloat16,
99
  )
 
 
100
  try:
101
- return TransformerCls.from_single_file(gguf_path, config=PIPELINE_ID, **kwargs)
 
102
  except Exception as exc: # config may live elsewhere; trust the GGUF KV metadata
103
  print(f"[ltx] from_single_file with config failed ({exc}), retrying without")
104
- return TransformerCls.from_single_file(gguf_path, **kwargs)
105
 
106
 
107
  def _load_model() -> dict:
@@ -113,9 +178,16 @@ def _load_model() -> dict:
113
  if _model is not None:
114
  return _model
115
 
 
 
 
 
 
 
 
 
116
  print("[ltx] downloading GGUF transformer…")
117
- gguf_path = os.environ.get("LTX_GGUF_CACHE", "/tmp/ltx-model.gguf")
118
- gguf_path = _download_gguf(gguf_path)
119
 
120
  print("[ltx] building quantized transformer…")
121
  transformer = _build_transformer(gguf_path)
@@ -125,17 +197,22 @@ def _load_model() -> dict:
125
  built = {
126
  "transformer": transformer,
127
  }
 
128
  try:
129
  built["t2v"] = LTX2Pipeline.from_pretrained(
130
- PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16
131
  )
132
  built["i2v"] = LTX2ImageToVideoPipeline.from_pretrained(
133
- PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16
134
  )
135
  except Exception as exc: # pragma: no cover - component layout differences
 
 
 
 
136
  raise RuntimeError(
137
- "[ltx] pipeline build failed - check Lightricks/LTX-2.5-Diffusers "
138
- f"component layout. {exc}"
139
  ) from exc
140
 
141
  for pipe in (built["t2v"], built["i2v"]):
@@ -182,11 +259,11 @@ def _extract_audio(output):
182
  if torch.is_tensor(arr):
183
  arr = arr.detach().float().cpu().numpy()
184
  arr = np.asarray(arr)
185
- if arr.ndim == 3: # (batch, samples, channels)
186
  arr = arr[0]
187
- if arr.ndim == 2: # downmix to mono for a clean narration bed
188
- arr = arr.mean(axis=1)
189
- return arr.astype(np.float32), int(sr)
190
 
191
 
192
  def _mux_audio(video_path, frames, fps, audio) -> str:
@@ -200,10 +277,11 @@ def _mux_audio(video_path, frames, fps, audio) -> str:
200
  from scipy.io import wavfile
201
 
202
  arr, sr = audio
203
- wav_path = video_path.with_suffix(".wav")
 
204
  wavfile.write(str(wav_path), sr, arr)
205
  ffmpeg = imageio_ffmpeg.get_ffmpeg_exe()
206
- muxed = str(video_path.with_suffix("_mux.mp4"))
207
  subprocess.run(
208
  [
209
  ffmpeg, "-y",
 
20
 
21
  from __future__ import annotations
22
 
23
+ import glob
24
  import os
25
  import subprocess
26
  import threading
27
+ from pathlib import Path
28
 
29
  # ZeroGPU rule #1: `import spaces` must precede any CUDA-touching import
30
  # (torch etc.) — it monkey-patches torch.cuda at import time.
 
33
  import numpy as np
34
  import torch
35
 
36
+ # Requires the Space secret HF_TOKEN after accepting the LTX-2.x Community
37
+ # License (free for entities under $10M annual revenue). Needed for every file
38
+ # served from the gated Lightricks/LTX-2.5-Diffusers repo (config, Gemma-4-12B
39
+ # text encoder, VAEs, connectors, audio_vae + vocoder).
40
+ HF_TOKEN = os.environ.get("HF_TOKEN", "").strip()
41
 
42
  # Direct download URL of the distilled GGUF transformer (~15GB). The backend
43
+ # only needs LTX_SPACE_URL; this key is configured in the Space. A GGUF the
44
+ # user uploads into the repo `model/` folder is preferred over this URL.
45
  LTX_MODEL_URL = os.environ.get(
46
  "LTX_MODEL_URL",
47
  "https://huggingface.co/realrebelai/LTX-2.5_GGUFs/resolve/main/LTX-2.5-Distilled-Q4_K_M.gguf",
48
  )
49
  PIPELINE_ID = "Lightricks/LTX-2.5-Diffusers"
50
+ # Overridable repo id (e.g. a locally mirrored copy: LTX_PIPELINE_ID=/app/ltx25).
51
+ LTX_PIPELINE_ID = os.environ.get("LTX_PIPELINE_ID", PIPELINE_ID)
52
  SEED = int(os.environ.get("LTX_SEED", "0"))
53
  FPS = int(os.environ.get("LTX_FPS", "24"))
54
 
 
69
  # ----------------------------------------------------------------- model load
70
 
71
 
72
+ def _find_local_gguf() -> str | None:
73
+ """Pick a GGUF the user uploaded into the repo (web UI ``model/`` folder)."""
74
+ for folder in ("model", "models", "weights"):
75
+ hits = [p for p in glob.glob(os.path.join(folder, "**", "*.gguf"), recursive=True) if os.path.isfile(p)]
76
+ if hits:
77
+ return sorted(hits)[0]
78
+ return None
79
+
80
+
81
+ def _gguf_fingerprint(path: str) -> None:
82
+ """Fail fast if the GGUF is not an LTX-2 video transformer.
83
+
84
+ diffusers 0.40 converts GGUFs that keep the ComfyUI/native
85
+ ``model.diffusion_model.*`` tensor names (llama.cpp-style ``blk.*`` renames
86
+ are not supported by the LTX-2 GGUF converter).
87
+ """
88
+ from gguf import GGUFReader
89
+
90
+ reader = GGUFReader(path)
91
+ names = [tensor.name for tensor in reader.tensors]
92
+ print(f"[ltx] gguf tensors={len(names)}")
93
+
94
+ def count(marker: str) -> int:
95
+ return sum(1 for name in names if marker in name)
96
+
97
+ blocks = count("transformer_blocks.")
98
+ ltx2_av_gate = count("model.diffusion_model.av_ca_a2v_gate_adaln_single")
99
+ llama_blk = count("blk.") + count(".attn_qkv.")
100
+ print(
101
+ f"[ltx] gguf fingerprint blocks={blocks} ltx2_av_gate={ltx2_av_gate} llama_blk={llama_blk}"
102
+ )
103
+ if blocks == 0 and llama_blk == 0:
104
+ raise RuntimeError(
105
+ "GGUF does not look like an LTX-2 video transformer (no transformer_blocks found)"
106
+ )
107
+ if blocks == 0 and llama_blk > 0:
108
+ raise RuntimeError(
109
+ "GGUF uses llama.cpp-renamed tensors (blk.*); diffusers 0.40's LTX-2 GGUF "
110
+ "converter expects the native model.diffusion_model.* names. Use a GGUF "
111
+ "converted from the ComfyUI/native checkpoint instead."
112
+ )
113
+
114
+
115
  def _download_gguf(dest: str) -> str:
116
  import httpx
117
 
 
128
 
129
  # Support "repo_id:filename" shorthand.
130
  repo, _, filename = LTX_MODEL_URL.partition(":")
131
+ return hf_hub_download(repo, filename or "LTX-2.5-Distilled-Q4_K_M.gguf", token=HF_TOKEN or None)
132
+
133
+
134
+ def _ensure_gguf() -> str:
135
+ """Local vendored GGUF first, then network download."""
136
+ local = _find_local_gguf()
137
+ if local is not None:
138
+ print(f"[ltx] using vendored GGUF: {local}")
139
+ return local
140
+ print("[ltx] no local GGUF found; downloading…")
141
+ return _download_gguf(os.environ.get("LTX_GGUF_CACHE", "/tmp/ltx-model.gguf"))
142
 
143
 
144
  def _build_transformer(gguf_path: str):
 
159
  quantization_config=GGUFQuantizationConfig(compute_dtype=torch.bfloat16),
160
  dtype=torch.bfloat16,
161
  )
162
+ _gguf_fingerprint(gguf_path)
163
+ token = HF_TOKEN or None
164
  try:
165
+ # `config=` pulls transformer/config.json from the gated repo (token).
166
+ return TransformerCls.from_single_file(gguf_path, config=LTX_PIPELINE_ID, token=token, **kwargs)
167
  except Exception as exc: # config may live elsewhere; trust the GGUF KV metadata
168
  print(f"[ltx] from_single_file with config failed ({exc}), retrying without")
169
+ return TransformerCls.from_single_file(gguf_path, token=token, **kwargs)
170
 
171
 
172
  def _load_model() -> dict:
 
178
  if _model is not None:
179
  return _model
180
 
181
+ if not HF_TOKEN and not os.path.isdir(LTX_PIPELINE_ID):
182
+ raise RuntimeError(
183
+ "HF_TOKEN secret is not set. Open the Space → Settings → Variables and "
184
+ "secrets and create a secret named HF_TOKEN: a fine-grained token with "
185
+ "read access to Lightricks/LTX-2.5-Diffusers (after accepting its "
186
+ "LTX-2.x Community License on that repo)."
187
+ )
188
+
189
  print("[ltx] downloading GGUF transformer…")
190
+ gguf_path = _ensure_gguf()
 
191
 
192
  print("[ltx] building quantized transformer…")
193
  transformer = _build_transformer(gguf_path)
 
197
  built = {
198
  "transformer": transformer,
199
  }
200
+ token = HF_TOKEN or None
201
  try:
202
  built["t2v"] = LTX2Pipeline.from_pretrained(
203
+ LTX_PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16, token=token
204
  )
205
  built["i2v"] = LTX2ImageToVideoPipeline.from_pretrained(
206
+ LTX_PIPELINE_ID, transformer=transformer, torch_dtype=torch.bfloat16, token=token
207
  )
208
  except Exception as exc: # pragma: no cover - component layout differences
209
+ msg = str(exc)
210
+ hint = ""
211
+ if "401" in msg or "is not a valid model identifier" in msg or "gated" in msg.lower():
212
+ hint = " (hint: accept the LTX-2.x Community License on the repo + set the HF_TOKEN secret)"
213
  raise RuntimeError(
214
+ f"[ltx] pipeline build failed - check Lightricks/LTX-2.5-Diffusers "
215
+ f"component layout. {msg}{hint}"
216
  ) from exc
217
 
218
  for pipe in (built["t2v"], built["i2v"]):
 
259
  if torch.is_tensor(arr):
260
  arr = arr.detach().float().cpu().numpy()
261
  arr = np.asarray(arr)
262
+ if arr.ndim == 3: # (batch, channels, samples)
263
  arr = arr[0]
264
+ if arr.ndim == 2: # collapse to mono; channel axis is the smaller dimension
265
+ arr = arr.mean(axis=int(arr.shape[0] < arr.shape[1]))
266
+ return np.ascontiguousarray(arr.astype(np.float32)), int(sr)
267
 
268
 
269
  def _mux_audio(video_path, frames, fps, audio) -> str:
 
277
  from scipy.io import wavfile
278
 
279
  arr, sr = audio
280
+ vp = Path(video_path)
281
+ wav_path = vp.with_suffix(".wav")
282
  wavfile.write(str(wav_path), sr, arr)
283
  ffmpeg = imageio_ffmpeg.get_ffmpeg_exe()
284
+ muxed = str(vp.with_name(vp.stem + "_mux.mp4"))
285
  subprocess.run(
286
  [
287
  ffmpeg, "-y",