gene commited on
Commit
e076dcd
·
1 Parent(s): d074dd4
Files changed (2) hide show
  1. app.py +77 -27
  2. requirements.txt +5 -3
app.py CHANGED
@@ -1,33 +1,35 @@
 
 
1
  import gradio as gr
2
  import fitz # PyMuPDF
3
  from PIL import Image
4
  import io
 
5
 
 
 
 
 
 
 
 
 
 
6
  def convert_pdf_to_image(pdf_file):
7
  """
8
- Convert PDF to image(s) and return the first page as PIL Image
9
  """
10
  if pdf_file is None:
11
  return "Please upload a PDF file first.", None
12
 
13
  try:
14
- # Open the PDF file
15
  pdf_document = fitz.open(pdf_file.name)
16
-
17
- # Get PDF info before closing
18
  total_pages = len(pdf_document)
19
-
20
- # Get the first page
21
  first_page = pdf_document[0]
22
-
23
- # Convert page to image (pixmap)
24
  pix = first_page.get_pixmap(matrix=fitz.Matrix(2, 2)) # 2x zoom for better quality
25
-
26
- # Convert pixmap to PIL Image
27
  img_data = pix.tobytes("png")
28
  img = Image.open(io.BytesIO(img_data))
29
-
30
- # Get PDF info
31
  pdf_info = f"""
32
  **PDF Conversion Results:**
33
  - Total pages: {total_pages}
@@ -35,38 +37,86 @@ def convert_pdf_to_image(pdf_file):
35
  - Image size: {img.width} x {img.height} pixels
36
  - Format: PNG
37
  """
38
-
39
- # Close the PDF document
40
  pdf_document.close()
41
-
42
  return pdf_info, img
43
-
44
  except Exception as e:
45
  return f"Error processing PDF: {str(e)}", None
46
 
47
- # Create the Gradio interface
48
- with gr.Blocks(title="PDF to Image Converter") as demo:
49
- gr.Markdown("# 📄 PDF to Image Converter")
50
- gr.Markdown("Upload a PDF file to convert the first page to an image!")
51
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
52
  with gr.Row():
53
  with gr.Column():
54
  pdf_input = gr.File(
55
- label="Upload PDF File",
56
  file_types=[".pdf"],
57
- file_count="single"
58
  )
59
- convert_btn = gr.Button("Convert to Image", variant="primary")
 
60
 
61
  with gr.Column():
62
  conversion_output = gr.Markdown(label="Conversion Results")
63
  converted_image = gr.Image(label="Converted Image (First Page)")
64
-
 
 
 
 
65
  convert_btn.click(
66
  fn=convert_pdf_to_image,
67
  inputs=[pdf_input],
68
- outputs=[conversion_output, converted_image]
 
 
 
 
 
 
69
  )
70
 
71
  if __name__ == "__main__":
72
- demo.launch(share=True)
 
1
+ from transformers import DonutProcessor, VisionEncoderDecoderModel
2
+ import torch
3
  import gradio as gr
4
  import fitz # PyMuPDF
5
  from PIL import Image
6
  import io
7
+ import json
8
 
9
+ # ====== Load Donut model (no fine-tuning needed) ======
10
+ MODEL_ID = "naver-clova-ix/donut-base-finetuned-docvqa"
11
+ device = "cuda" if torch.cuda.is_available() else "cpu"
12
+
13
+ processor = DonutProcessor.from_pretrained(MODEL_ID)
14
+ model = VisionEncoderDecoderModel.from_pretrained(MODEL_ID).to(device)
15
+
16
+
17
+ # ====== Convert PDF to Image ======
18
  def convert_pdf_to_image(pdf_file):
19
  """
20
+ Convert first page of PDF to PIL Image and return both image and info.
21
  """
22
  if pdf_file is None:
23
  return "Please upload a PDF file first.", None
24
 
25
  try:
 
26
  pdf_document = fitz.open(pdf_file.name)
 
 
27
  total_pages = len(pdf_document)
 
 
28
  first_page = pdf_document[0]
 
 
29
  pix = first_page.get_pixmap(matrix=fitz.Matrix(2, 2)) # 2x zoom for better quality
 
 
30
  img_data = pix.tobytes("png")
31
  img = Image.open(io.BytesIO(img_data))
32
+
 
33
  pdf_info = f"""
34
  **PDF Conversion Results:**
35
  - Total pages: {total_pages}
 
37
  - Image size: {img.width} x {img.height} pixels
38
  - Format: PNG
39
  """
40
+
 
41
  pdf_document.close()
 
42
  return pdf_info, img
43
+
44
  except Exception as e:
45
  return f"Error processing PDF: {str(e)}", None
46
 
47
+
48
+ # ====== Process Image with Donut ======
49
+ def process_image(image):
50
+ if image is None:
51
+ return "No image found."
52
+
53
+ try:
54
+ # Preprocess image
55
+ pixel_values = processor(image, return_tensors="pt").pixel_values.to(device)
56
+ task_prompt = "<s_docvqa><s_question>Extract all driver’s license fields.<s_answer>"
57
+ decoder_input_ids = processor.tokenizer(
58
+ task_prompt, add_special_tokens=False, return_tensors="pt"
59
+ ).input_ids.to(device)
60
+
61
+ # Generate text output
62
+ outputs = model.generate(
63
+ pixel_values,
64
+ decoder_input_ids=decoder_input_ids,
65
+ max_length=512,
66
+ num_beams=4,
67
+ )
68
+ result = processor.batch_decode(outputs, skip_special_tokens=True)[0]
69
+
70
+ # Try parsing as JSON
71
+ try:
72
+ data = json.loads(result)
73
+ formatted = json.dumps(data, indent=2)
74
+ except Exception:
75
+ formatted = result # fallback to raw text
76
+
77
+ return formatted
78
+
79
+ except Exception as e:
80
+ return f"Error during extraction: {str(e)}"
81
+
82
+
83
+ # ====== Gradio Interface ======
84
+ with gr.Blocks(title="Driver’s License Info Extractor") as demo:
85
+ gr.Markdown("# 🪪 Driver’s License Info Extractor (Prototype)")
86
+ gr.Markdown(
87
+ "Upload a **PDF of a driver’s license**. The app converts it to an image, "
88
+ "then uses a **Donut document understanding model** to extract key details."
89
+ )
90
+
91
  with gr.Row():
92
  with gr.Column():
93
  pdf_input = gr.File(
94
+ label="Upload Driver’s License PDF",
95
  file_types=[".pdf"],
96
+ file_count="single",
97
  )
98
+ convert_btn = gr.Button("1️⃣ Convert PDF to Image", variant="primary")
99
+ extract_btn = gr.Button("2️⃣ Extract Information", variant="primary")
100
 
101
  with gr.Column():
102
  conversion_output = gr.Markdown(label="Conversion Results")
103
  converted_image = gr.Image(label="Converted Image (First Page)")
104
+ extraction_output = gr.Textbox(
105
+ label="Extracted License Info (JSON/Text)", lines=10
106
+ )
107
+
108
+ # Link buttons
109
  convert_btn.click(
110
  fn=convert_pdf_to_image,
111
  inputs=[pdf_input],
112
+ outputs=[conversion_output, converted_image],
113
+ )
114
+
115
+ extract_btn.click(
116
+ fn=process_image,
117
+ inputs=[converted_image],
118
+ outputs=[extraction_output],
119
  )
120
 
121
  if __name__ == "__main__":
122
+ demo.launch(share=True)
requirements.txt CHANGED
@@ -1,3 +1,5 @@
1
- gradio==5.49.0
2
- PyMuPDF==1.23.8
3
- Pillow==10.0.1
 
 
 
1
+ torch
2
+ transformers
3
+ gradio
4
+ pymupdf
5
+ pillow