import gradio as gr import torch from transformers import AutoProcessor, AutoModelForImageTextToText MODEL_ID = "XingChen-AGI/TeleOCR" processor = AutoProcessor.from_pretrained( MODEL_ID, trust_remote_code=True ) model = AutoModelForImageTextToText.from_pretrained( MODEL_ID, trust_remote_code=True, torch_dtype="auto", device_map="auto" ) def extract_text(image): if image is None: return "Please upload an image." messages = [ { "role": "user", "content": [ {"type": "image", "image": image}, { "type": "text", "text": "Read all the text in this image." } ] } ] text = processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True ) inputs = processor( text=[text], images=[image], padding=True, return_tensors="pt" ) inputs = inputs.to(model.device) with torch.no_grad(): generated_ids = model.generate( **inputs, max_new_tokens=512 ) generated_ids_trimmed = [ out_ids[len(in_ids):] for in_ids, out_ids in zip( inputs.input_ids, generated_ids ) ] result = processor.batch_decode( generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False ) return result[0] demo = gr.Interface( fn=extract_text, inputs=gr.Image( type="pil", label="Upload Image" ), outputs=gr.Textbox( label="Extracted Text", lines=12 ), title="TeleOCR – Image Text Reader", description="Upload an image and AI will extract the printed text from it.", submit_btn="Extract Text", clear_btn="Clear" ) if __name__ == "__main__": demo.launch()