import gradio as gr import spaces import torch from transformers import AutoProcessor, Florence2ForConditionalGeneration MODEL_ID = "florence-community/Florence-2-large" model = Florence2ForConditionalGeneration.from_pretrained( MODEL_ID, dtype=torch.float16 ).to("cuda") processor = AutoProcessor.from_pretrained(MODEL_ID) @spaces.GPU def read_text(image): if image is None: return "Please upload an image first." image = image.convert("RGB") inputs = processor(text="", images=image, return_tensors="pt").to( "cuda", torch.float16 ) generated_ids = model.generate( **inputs, max_new_tokens=1024, num_beams=3, do_sample=False ) raw = processor.batch_decode(generated_ids, skip_special_tokens=False)[0] parsed = processor.post_process_generation( raw, task="", image_size=image.size ) text = parsed[""].strip() return text or "No text was found in this image." demo = gr.Interface( fn=read_text, inputs=gr.Image(type="pil", label="Upload an image with printed text"), outputs=gr.Textbox(label="Text the model read", lines=10), title="OCR Reader", description="Upload an image of printed text and get back the text the model reads.", ) demo.launch()