ocr_reader / app.py
Noorzzz's picture
Create app.py
4d1ffd6 verified
Raw History Blame Contribute Delete
1.94 kB
import gradio as gr
import torch
from transformers import AutoProcessor, AutoModelForImageTextToText
MODEL_ID = "XingChen-AGI/TeleOCR"
processor = AutoProcessor.from_pretrained(
MODEL_ID,
trust_remote_code=True
)
model = AutoModelForImageTextToText.from_pretrained(
MODEL_ID,
trust_remote_code=True,
torch_dtype="auto",
device_map="auto"
)
def extract_text(image):
if image is None:
return "Please upload an image."
messages = [
{
"role": "user",
"content": [
{"type": "image", "image": image},
{
"type": "text",
"text": "Read all the text in this image."
}
]
}
]
text = processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
inputs = processor(
text=[text],
images=[image],
padding=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
with torch.no_grad():
generated_ids = model.generate(
**inputs,
max_new_tokens=512
)
generated_ids_trimmed = [
out_ids[len(in_ids):]
for in_ids, out_ids in zip(
inputs.input_ids,
generated_ids
)
]
result = processor.batch_decode(
generated_ids_trimmed,
skip_special_tokens=True,
clean_up_tokenization_spaces=False
)
return result[0]
demo = gr.Interface(
fn=extract_text,
inputs=gr.Image(
type="pil",
label="Upload Image"
),
outputs=gr.Textbox(
label="Extracted Text",
lines=12
),
title="TeleOCR – Image Text Reader",
description="Upload an image and AI will extract the printed text from it.",
submit_btn="Extract Text",
clear_btn="Clear"
)
if __name__ == "__main__":
demo.launch()