from fastapi import FastAPI, UploadFile, File from llama_cpp import Llama from llama_cpp.llama_chat_format import Llava15ChatHandler from pdf2image import convert_from_bytes import io from PIL import Image app = FastAPI() print("⏳ Loading Llava 1.6 Model...") # 1. Initialize Vision Handler # The Dockerfile (which ran successfully!) saved the file here: chat_handler = Llava15ChatHandler(clip_model_path="/app/model/mmproj.gguf") # 2. Initialize Model llm = Llama( model_path="/app/model/model.gguf", chat_handler=chat_handler, n_ctx=2048, n_gpu_layers=0, # Force CPU verbose=True ) print("✅ Model Loaded Successfully!") @app.post("/extract") async def extract_text(file: UploadFile = File(...)): # --- Image Processing --- if file.filename.endswith('.pdf'): pdf_bytes = await file.read() images = convert_from_bytes(pdf_bytes) image = images[0] else: image_data = await file.read() image = Image.open(io.BytesIO(image_data)) temp_path = "/tmp/temp_doc.jpg" image.save(temp_path) # --- Prompt --- messages = [ {"role": "system", "content": "You are an AI that extracts text from images."}, { "role": "user", "content": [ {"type": "image_url", "image_url": {"url": f"file://{temp_path}"}}, {"type": "text", "text": "Extract all text from this image. Output in Markdown format."} ] } ] response = llm.create_chat_completion(messages=messages, max_tokens=1500) return {"filename": file.filename, "content": response["choices"][0]["message"]["content"]}