File size: 1,658 Bytes
44db53f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
from fastapi import FastAPI, UploadFile, File
from llama_cpp import Llama
from llama_cpp.llama_chat_format import Llava15ChatHandler
from pdf2image import convert_from_bytes
import io
from PIL import Image

app = FastAPI()

print("⏳ Loading Llava 1.6 Model...")

# 1. Initialize Vision Handler
# The Dockerfile (which ran successfully!) saved the file here:
chat_handler = Llava15ChatHandler(clip_model_path="/app/model/mmproj.gguf")

# 2. Initialize Model
llm = Llama(
    model_path="/app/model/model.gguf",
    chat_handler=chat_handler,
    n_ctx=2048,
    n_gpu_layers=0, # Force CPU
    verbose=True
)
print("✅ Model Loaded Successfully!")

@app.post("/extract")
async def extract_text(file: UploadFile = File(...)):
    # --- Image Processing ---
    if file.filename.endswith('.pdf'):
        pdf_bytes = await file.read()
        images = convert_from_bytes(pdf_bytes)
        image = images[0]
    else:
        image_data = await file.read()
        image = Image.open(io.BytesIO(image_data))

    temp_path = "/tmp/temp_doc.jpg"
    image.save(temp_path)

    # --- Prompt ---
    messages = [
        {"role": "system", "content": "You are an AI that extracts text from images."},
        {
            "role": "user",
            "content": [
                {"type": "image_url", "image_url": {"url": f"file://{temp_path}"}},
                {"type": "text", "text": "Extract all text from this image. Output in Markdown format."}
            ]
        }
    ]
    
    response = llm.create_chat_completion(messages=messages, max_tokens=1500)
    return {"filename": file.filename, "content": response["choices"][0]["message"]["content"]}