Bot commited on
Commit
4b556cf
·
1 Parent(s): 7c55e05

Fix: load base model in plain fp32 + PeftModel, avoiding the 4-bit quantization auto-detection that required a GPU at import time (unavailable under ZeroGPU)

Browse files
Files changed (1) hide show
  1. app.py +16 -6
app.py CHANGED
@@ -17,7 +17,7 @@ from fastapi import FastAPI, HTTPException
17
  from pydantic import BaseModel
18
  import gradio as gr
19
  from transformers import AutoTokenizer, AutoModelForCausalLM
20
- from peft import AutoPeftModelForCausalLM
21
  import torch
22
 
23
  # Model is loaded once at IMPORT TIME, not inside a FastAPI lifespan hook.
@@ -29,21 +29,31 @@ import torch
29
  # Also: under HF's free ZeroGPU tier, no GPU is visible at import time --
30
  # it's only allocated for the duration of an @spaces.GPU-decorated call.
31
  # So load on CPU here, then move to CUDA inside gradio_interface() below.
 
 
 
 
 
 
 
 
 
 
32
  print("Loading fine-tuned model...")
33
  ADAPTER_PATH = "vishnuadupa/qwen-sql-lora"
 
34
 
35
  try:
36
  print(f"Loading adapter from HF Hub: {ADAPTER_PATH}")
37
  tokenizer = AutoTokenizer.from_pretrained(ADAPTER_PATH)
38
- model = AutoPeftModelForCausalLM.from_pretrained(ADAPTER_PATH)
 
39
  model = model.merge_and_unload()
40
  print("Fine-tuned model loaded.")
41
  except Exception as e:
42
  print(f"Could not load fine-tuned model ({e}); falling back to base model.")
43
- tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-Coder-1.5B-Instruct")
44
- model = AutoModelForCausalLM.from_pretrained(
45
- "Qwen/Qwen2.5-Coder-1.5B-Instruct", torch_dtype=torch.float16
46
- )
47
  print("Base model loaded.")
48
 
49
  model = model.to("cpu")
 
17
  from pydantic import BaseModel
18
  import gradio as gr
19
  from transformers import AutoTokenizer, AutoModelForCausalLM
20
+ from peft import PeftModel
21
  import torch
22
 
23
  # Model is loaded once at IMPORT TIME, not inside a FastAPI lifespan hook.
 
29
  # Also: under HF's free ZeroGPU tier, no GPU is visible at import time --
30
  # it's only allocated for the duration of an @spaces.GPU-decorated call.
31
  # So load on CPU here, then move to CUDA inside gradio_interface() below.
32
+ #
33
+ # IMPORTANT: load the base model in plain fp32 and apply the adapter via
34
+ # plain PeftModel, NOT AutoPeftModelForCausalLM. The adapter was trained
35
+ # with load_in_4bit=True, so its saved config carries a 4-bit quantization
36
+ # config; AutoPeftModelForCausalLM auto-detects and applies that at load
37
+ # time, which requires an actual CUDA device to instantiate -- confirmed
38
+ # via a hard failure: "Could not load fine-tuned model (No CUDA GPUs are
39
+ # available)" at import time, since ZeroGPU grants no GPU until a
40
+ # @spaces.GPU-decorated call actually runs. Loading in plain fp32 needs no
41
+ # quantization step at all, so it works with zero GPU present.
42
  print("Loading fine-tuned model...")
43
  ADAPTER_PATH = "vishnuadupa/qwen-sql-lora"
44
+ BASE_MODEL_NAME = "Qwen/Qwen2.5-Coder-1.5B-Instruct"
45
 
46
  try:
47
  print(f"Loading adapter from HF Hub: {ADAPTER_PATH}")
48
  tokenizer = AutoTokenizer.from_pretrained(ADAPTER_PATH)
49
+ base_model = AutoModelForCausalLM.from_pretrained(BASE_MODEL_NAME, torch_dtype=torch.float32)
50
+ model = PeftModel.from_pretrained(base_model, ADAPTER_PATH)
51
  model = model.merge_and_unload()
52
  print("Fine-tuned model loaded.")
53
  except Exception as e:
54
  print(f"Could not load fine-tuned model ({e}); falling back to base model.")
55
+ tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL_NAME)
56
+ model = AutoModelForCausalLM.from_pretrained(BASE_MODEL_NAME, torch_dtype=torch.float32)
 
 
57
  print("Base model loaded.")
58
 
59
  model = model.to("cpu")