devxyasir commited on
Commit
8a9903b
Β·
verified Β·
1 Parent(s): fd518c5

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +20 -8
app.py CHANGED
@@ -33,6 +33,23 @@ from fastapi.responses import StreamingResponse
33
  from pydantic import BaseModel
34
  from huggingface_hub import InferenceClient
35
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36
 
37
  # ═══════════════════════════════════════════════════════════════════════
38
  # Configuration
@@ -471,18 +488,13 @@ async def health():
471
 
472
 
473
  # ═══════════════════════════════════════════════════════════════════════
474
- # Gradio Chat UI β€” @spaces.GPU satisfies ZeroGPU startup requirement
475
  # ═══════════════════════════════════════════════════════════════════════
476
 
 
477
  @spaces.GPU(duration=120)
478
  def _gpu_inference(message: str, history: list, system_prompt: str) -> str:
479
- """
480
- ZeroGPU requires at least one @spaces.GPU function.
481
- Actual inference happens via InferenceClient (HF servers),
482
- so the GPU allocation is minimal. This wrapper satisfies the
483
- ZeroGPU startup scanner.
484
- """
485
- # Update global system prompt from UI
486
  with _lock:
487
  _state["system_prompt"] = system_prompt
488
 
 
33
  from pydantic import BaseModel
34
  from huggingface_hub import InferenceClient
35
 
36
+ # ═══════════════════════════════════════════════════════════════════════
37
+ # ZeroGPU β€” decorated function MUST be defined early for startup scanner
38
+ # ═══════════════════════════════════════════════════════════════════════
39
+
40
+ # Force torch detection (ZeroGPU scans for torch usage)
41
+ _ = torch.cuda.is_available()
42
+
43
+ @spaces.GPU(duration=120)
44
+ def _gpu_inference(message: str, history: list, system_prompt: str) -> str:
45
+ """
46
+ ZeroGPU requires at least one @spaces.GPU function.
47
+ Actual inference happens via InferenceClient (HF servers),
48
+ so the GPU allocation is minimal.
49
+ """
50
+ # This will be replaced after client is initialized below
51
+ return "Model loading..."
52
+
53
 
54
  # ═══════════════════════════════════════════════════════════════════════
55
  # Configuration
 
488
 
489
 
490
  # ═══════════════════════════════════════════════════════════════════════
491
+ # Gradio Chat UI
492
  # ═══════════════════════════════════════════════════════════════════════
493
 
494
+ # Re-define _gpu_inference now that client exists
495
  @spaces.GPU(duration=120)
496
  def _gpu_inference(message: str, history: list, system_prompt: str) -> str:
497
+ """Inference via HF InferenceClient, wrapped in @spaces.GPU for ZeroGPU."""
 
 
 
 
 
 
498
  with _lock:
499
  _state["system_prompt"] = system_prompt
500