Spaces:
Running on Zero
Running on Zero
Bot commited on
Commit ·
4b556cf
1
Parent(s): 7c55e05
Fix: load base model in plain fp32 + PeftModel, avoiding the 4-bit quantization auto-detection that required a GPU at import time (unavailable under ZeroGPU)
Browse files
app.py
CHANGED
|
@@ -17,7 +17,7 @@ from fastapi import FastAPI, HTTPException
|
|
| 17 |
from pydantic import BaseModel
|
| 18 |
import gradio as gr
|
| 19 |
from transformers import AutoTokenizer, AutoModelForCausalLM
|
| 20 |
-
from peft import
|
| 21 |
import torch
|
| 22 |
|
| 23 |
# Model is loaded once at IMPORT TIME, not inside a FastAPI lifespan hook.
|
|
@@ -29,21 +29,31 @@ import torch
|
|
| 29 |
# Also: under HF's free ZeroGPU tier, no GPU is visible at import time --
|
| 30 |
# it's only allocated for the duration of an @spaces.GPU-decorated call.
|
| 31 |
# So load on CPU here, then move to CUDA inside gradio_interface() below.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
print("Loading fine-tuned model...")
|
| 33 |
ADAPTER_PATH = "vishnuadupa/qwen-sql-lora"
|
|
|
|
| 34 |
|
| 35 |
try:
|
| 36 |
print(f"Loading adapter from HF Hub: {ADAPTER_PATH}")
|
| 37 |
tokenizer = AutoTokenizer.from_pretrained(ADAPTER_PATH)
|
| 38 |
-
|
|
|
|
| 39 |
model = model.merge_and_unload()
|
| 40 |
print("Fine-tuned model loaded.")
|
| 41 |
except Exception as e:
|
| 42 |
print(f"Could not load fine-tuned model ({e}); falling back to base model.")
|
| 43 |
-
tokenizer = AutoTokenizer.from_pretrained(
|
| 44 |
-
model = AutoModelForCausalLM.from_pretrained(
|
| 45 |
-
"Qwen/Qwen2.5-Coder-1.5B-Instruct", torch_dtype=torch.float16
|
| 46 |
-
)
|
| 47 |
print("Base model loaded.")
|
| 48 |
|
| 49 |
model = model.to("cpu")
|
|
|
|
| 17 |
from pydantic import BaseModel
|
| 18 |
import gradio as gr
|
| 19 |
from transformers import AutoTokenizer, AutoModelForCausalLM
|
| 20 |
+
from peft import PeftModel
|
| 21 |
import torch
|
| 22 |
|
| 23 |
# Model is loaded once at IMPORT TIME, not inside a FastAPI lifespan hook.
|
|
|
|
| 29 |
# Also: under HF's free ZeroGPU tier, no GPU is visible at import time --
|
| 30 |
# it's only allocated for the duration of an @spaces.GPU-decorated call.
|
| 31 |
# So load on CPU here, then move to CUDA inside gradio_interface() below.
|
| 32 |
+
#
|
| 33 |
+
# IMPORTANT: load the base model in plain fp32 and apply the adapter via
|
| 34 |
+
# plain PeftModel, NOT AutoPeftModelForCausalLM. The adapter was trained
|
| 35 |
+
# with load_in_4bit=True, so its saved config carries a 4-bit quantization
|
| 36 |
+
# config; AutoPeftModelForCausalLM auto-detects and applies that at load
|
| 37 |
+
# time, which requires an actual CUDA device to instantiate -- confirmed
|
| 38 |
+
# via a hard failure: "Could not load fine-tuned model (No CUDA GPUs are
|
| 39 |
+
# available)" at import time, since ZeroGPU grants no GPU until a
|
| 40 |
+
# @spaces.GPU-decorated call actually runs. Loading in plain fp32 needs no
|
| 41 |
+
# quantization step at all, so it works with zero GPU present.
|
| 42 |
print("Loading fine-tuned model...")
|
| 43 |
ADAPTER_PATH = "vishnuadupa/qwen-sql-lora"
|
| 44 |
+
BASE_MODEL_NAME = "Qwen/Qwen2.5-Coder-1.5B-Instruct"
|
| 45 |
|
| 46 |
try:
|
| 47 |
print(f"Loading adapter from HF Hub: {ADAPTER_PATH}")
|
| 48 |
tokenizer = AutoTokenizer.from_pretrained(ADAPTER_PATH)
|
| 49 |
+
base_model = AutoModelForCausalLM.from_pretrained(BASE_MODEL_NAME, torch_dtype=torch.float32)
|
| 50 |
+
model = PeftModel.from_pretrained(base_model, ADAPTER_PATH)
|
| 51 |
model = model.merge_and_unload()
|
| 52 |
print("Fine-tuned model loaded.")
|
| 53 |
except Exception as e:
|
| 54 |
print(f"Could not load fine-tuned model ({e}); falling back to base model.")
|
| 55 |
+
tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL_NAME)
|
| 56 |
+
model = AutoModelForCausalLM.from_pretrained(BASE_MODEL_NAME, torch_dtype=torch.float32)
|
|
|
|
|
|
|
| 57 |
print("Base model loaded.")
|
| 58 |
|
| 59 |
model = model.to("cpu")
|