Spaces:
Sleeping
Sleeping
gene commited on
Commit ·
e076dcd
1
Parent(s): d074dd4
model
Browse files- app.py +77 -27
- requirements.txt +5 -3
app.py
CHANGED
|
@@ -1,33 +1,35 @@
|
|
|
|
|
|
|
|
| 1 |
import gradio as gr
|
| 2 |
import fitz # PyMuPDF
|
| 3 |
from PIL import Image
|
| 4 |
import io
|
|
|
|
| 5 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
def convert_pdf_to_image(pdf_file):
|
| 7 |
"""
|
| 8 |
-
Convert PDF to
|
| 9 |
"""
|
| 10 |
if pdf_file is None:
|
| 11 |
return "Please upload a PDF file first.", None
|
| 12 |
|
| 13 |
try:
|
| 14 |
-
# Open the PDF file
|
| 15 |
pdf_document = fitz.open(pdf_file.name)
|
| 16 |
-
|
| 17 |
-
# Get PDF info before closing
|
| 18 |
total_pages = len(pdf_document)
|
| 19 |
-
|
| 20 |
-
# Get the first page
|
| 21 |
first_page = pdf_document[0]
|
| 22 |
-
|
| 23 |
-
# Convert page to image (pixmap)
|
| 24 |
pix = first_page.get_pixmap(matrix=fitz.Matrix(2, 2)) # 2x zoom for better quality
|
| 25 |
-
|
| 26 |
-
# Convert pixmap to PIL Image
|
| 27 |
img_data = pix.tobytes("png")
|
| 28 |
img = Image.open(io.BytesIO(img_data))
|
| 29 |
-
|
| 30 |
-
# Get PDF info
|
| 31 |
pdf_info = f"""
|
| 32 |
**PDF Conversion Results:**
|
| 33 |
- Total pages: {total_pages}
|
|
@@ -35,38 +37,86 @@ def convert_pdf_to_image(pdf_file):
|
|
| 35 |
- Image size: {img.width} x {img.height} pixels
|
| 36 |
- Format: PNG
|
| 37 |
"""
|
| 38 |
-
|
| 39 |
-
# Close the PDF document
|
| 40 |
pdf_document.close()
|
| 41 |
-
|
| 42 |
return pdf_info, img
|
| 43 |
-
|
| 44 |
except Exception as e:
|
| 45 |
return f"Error processing PDF: {str(e)}", None
|
| 46 |
|
| 47 |
-
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 52 |
with gr.Row():
|
| 53 |
with gr.Column():
|
| 54 |
pdf_input = gr.File(
|
| 55 |
-
label="Upload
|
| 56 |
file_types=[".pdf"],
|
| 57 |
-
file_count="single"
|
| 58 |
)
|
| 59 |
-
convert_btn = gr.Button("Convert to Image", variant="primary")
|
|
|
|
| 60 |
|
| 61 |
with gr.Column():
|
| 62 |
conversion_output = gr.Markdown(label="Conversion Results")
|
| 63 |
converted_image = gr.Image(label="Converted Image (First Page)")
|
| 64 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
convert_btn.click(
|
| 66 |
fn=convert_pdf_to_image,
|
| 67 |
inputs=[pdf_input],
|
| 68 |
-
outputs=[conversion_output, converted_image]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 69 |
)
|
| 70 |
|
| 71 |
if __name__ == "__main__":
|
| 72 |
-
demo.launch(share=True)
|
|
|
|
| 1 |
+
from transformers import DonutProcessor, VisionEncoderDecoderModel
|
| 2 |
+
import torch
|
| 3 |
import gradio as gr
|
| 4 |
import fitz # PyMuPDF
|
| 5 |
from PIL import Image
|
| 6 |
import io
|
| 7 |
+
import json
|
| 8 |
|
| 9 |
+
# ====== Load Donut model (no fine-tuning needed) ======
|
| 10 |
+
MODEL_ID = "naver-clova-ix/donut-base-finetuned-docvqa"
|
| 11 |
+
device = "cuda" if torch.cuda.is_available() else "cpu"
|
| 12 |
+
|
| 13 |
+
processor = DonutProcessor.from_pretrained(MODEL_ID)
|
| 14 |
+
model = VisionEncoderDecoderModel.from_pretrained(MODEL_ID).to(device)
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
# ====== Convert PDF to Image ======
|
| 18 |
def convert_pdf_to_image(pdf_file):
|
| 19 |
"""
|
| 20 |
+
Convert first page of PDF to PIL Image and return both image and info.
|
| 21 |
"""
|
| 22 |
if pdf_file is None:
|
| 23 |
return "Please upload a PDF file first.", None
|
| 24 |
|
| 25 |
try:
|
|
|
|
| 26 |
pdf_document = fitz.open(pdf_file.name)
|
|
|
|
|
|
|
| 27 |
total_pages = len(pdf_document)
|
|
|
|
|
|
|
| 28 |
first_page = pdf_document[0]
|
|
|
|
|
|
|
| 29 |
pix = first_page.get_pixmap(matrix=fitz.Matrix(2, 2)) # 2x zoom for better quality
|
|
|
|
|
|
|
| 30 |
img_data = pix.tobytes("png")
|
| 31 |
img = Image.open(io.BytesIO(img_data))
|
| 32 |
+
|
|
|
|
| 33 |
pdf_info = f"""
|
| 34 |
**PDF Conversion Results:**
|
| 35 |
- Total pages: {total_pages}
|
|
|
|
| 37 |
- Image size: {img.width} x {img.height} pixels
|
| 38 |
- Format: PNG
|
| 39 |
"""
|
| 40 |
+
|
|
|
|
| 41 |
pdf_document.close()
|
|
|
|
| 42 |
return pdf_info, img
|
| 43 |
+
|
| 44 |
except Exception as e:
|
| 45 |
return f"Error processing PDF: {str(e)}", None
|
| 46 |
|
| 47 |
+
|
| 48 |
+
# ====== Process Image with Donut ======
|
| 49 |
+
def process_image(image):
|
| 50 |
+
if image is None:
|
| 51 |
+
return "No image found."
|
| 52 |
+
|
| 53 |
+
try:
|
| 54 |
+
# Preprocess image
|
| 55 |
+
pixel_values = processor(image, return_tensors="pt").pixel_values.to(device)
|
| 56 |
+
task_prompt = "<s_docvqa><s_question>Extract all driver’s license fields.<s_answer>"
|
| 57 |
+
decoder_input_ids = processor.tokenizer(
|
| 58 |
+
task_prompt, add_special_tokens=False, return_tensors="pt"
|
| 59 |
+
).input_ids.to(device)
|
| 60 |
+
|
| 61 |
+
# Generate text output
|
| 62 |
+
outputs = model.generate(
|
| 63 |
+
pixel_values,
|
| 64 |
+
decoder_input_ids=decoder_input_ids,
|
| 65 |
+
max_length=512,
|
| 66 |
+
num_beams=4,
|
| 67 |
+
)
|
| 68 |
+
result = processor.batch_decode(outputs, skip_special_tokens=True)[0]
|
| 69 |
+
|
| 70 |
+
# Try parsing as JSON
|
| 71 |
+
try:
|
| 72 |
+
data = json.loads(result)
|
| 73 |
+
formatted = json.dumps(data, indent=2)
|
| 74 |
+
except Exception:
|
| 75 |
+
formatted = result # fallback to raw text
|
| 76 |
+
|
| 77 |
+
return formatted
|
| 78 |
+
|
| 79 |
+
except Exception as e:
|
| 80 |
+
return f"Error during extraction: {str(e)}"
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
# ====== Gradio Interface ======
|
| 84 |
+
with gr.Blocks(title="Driver’s License Info Extractor") as demo:
|
| 85 |
+
gr.Markdown("# 🪪 Driver’s License Info Extractor (Prototype)")
|
| 86 |
+
gr.Markdown(
|
| 87 |
+
"Upload a **PDF of a driver’s license**. The app converts it to an image, "
|
| 88 |
+
"then uses a **Donut document understanding model** to extract key details."
|
| 89 |
+
)
|
| 90 |
+
|
| 91 |
with gr.Row():
|
| 92 |
with gr.Column():
|
| 93 |
pdf_input = gr.File(
|
| 94 |
+
label="Upload Driver’s License PDF",
|
| 95 |
file_types=[".pdf"],
|
| 96 |
+
file_count="single",
|
| 97 |
)
|
| 98 |
+
convert_btn = gr.Button("1️⃣ Convert PDF to Image", variant="primary")
|
| 99 |
+
extract_btn = gr.Button("2️⃣ Extract Information", variant="primary")
|
| 100 |
|
| 101 |
with gr.Column():
|
| 102 |
conversion_output = gr.Markdown(label="Conversion Results")
|
| 103 |
converted_image = gr.Image(label="Converted Image (First Page)")
|
| 104 |
+
extraction_output = gr.Textbox(
|
| 105 |
+
label="Extracted License Info (JSON/Text)", lines=10
|
| 106 |
+
)
|
| 107 |
+
|
| 108 |
+
# Link buttons
|
| 109 |
convert_btn.click(
|
| 110 |
fn=convert_pdf_to_image,
|
| 111 |
inputs=[pdf_input],
|
| 112 |
+
outputs=[conversion_output, converted_image],
|
| 113 |
+
)
|
| 114 |
+
|
| 115 |
+
extract_btn.click(
|
| 116 |
+
fn=process_image,
|
| 117 |
+
inputs=[converted_image],
|
| 118 |
+
outputs=[extraction_output],
|
| 119 |
)
|
| 120 |
|
| 121 |
if __name__ == "__main__":
|
| 122 |
+
demo.launch(share=True)
|
requirements.txt
CHANGED
|
@@ -1,3 +1,5 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
|
|
|
|
|
|
|
|
|
| 1 |
+
torch
|
| 2 |
+
transformers
|
| 3 |
+
gradio
|
| 4 |
+
pymupdf
|
| 5 |
+
pillow
|