Download nvidia_ocr.py from abersbail/tiny-code-only-tts: direct link, hf CLI and curl.
- Browser
- Download file 5.63 kB
-
https://huggingface.co/spaces/abersbail/tiny-code-only-tts/resolve/main/nvidia_ocr.py
- Command line
-
hf download hf://spaces/abersbail/tiny-code-only-tts/nvidia_ocr.py
-
curl -L -o nvidia_ocr.py https://huggingface.co/spaces/abersbail/tiny-code-only-tts/resolve/main/nvidia_ocr.py
5.63 kB
| import os | |
| import base64 | |
| import requests | |
| import io | |
| from PIL import Image | |
| from pypdf import PdfReader | |
| NVIDIA_API_KEY = os.environ.get("NVIDIA_API_KEY", "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") | |
| NEMOTRON_OCR_V2_URL = "https://ai.api.nvidia.com/v1/cv/nvidia/nemotron-ocr-v2" | |
| NEMOTRON_OCR_V1_URL = "https://ai.api.nvidia.com/v1/cv/nvidia/nemotron-ocr-v1" | |
| def process_image_for_ocr(image_path: str, max_b64_len: int = 175000) -> str: | |
| """ | |
| Reads image file, resizes if necessary to ensure Base64 string is under NVIDIA 180KB payload limit. | |
| """ | |
| with Image.open(image_path) as img: | |
| img = img.convert("RGB") | |
| buf = io.BytesIO() | |
| img.save(buf, format="JPEG", quality=85) | |
| b64_str = base64.b64encode(buf.getvalue()).decode() | |
| scale = 0.9 | |
| while len(b64_str) > max_b64_len and scale > 0.2: | |
| new_w = int(img.width * scale) | |
| new_h = int(img.height * scale) | |
| resized_img = img.resize((new_w, new_h), Image.Resampling.LANCZOS) | |
| buf = io.BytesIO() | |
| resized_img.save(buf, format="JPEG", quality=80) | |
| b64_str = base64.b64encode(buf.getvalue()).decode() | |
| scale -= 0.1 | |
| return b64_str | |
| def extract_pdf_text_and_pages(pdf_path: str) -> dict: | |
| """ | |
| Parses multi-page PDF resume using pypdf. | |
| """ | |
| try: | |
| reader = PdfReader(pdf_path) | |
| page_texts = [] | |
| full_text_lines = [] | |
| for idx, page in enumerate(reader.pages): | |
| txt = page.extract_text() or "" | |
| if txt.strip(): | |
| page_texts.append(f"--- Page {idx+1} ---\n{txt}") | |
| full_text_lines.extend([line.strip() for line in txt.splitlines() if line.strip()]) | |
| combined_text = "\n".join(full_text_lines) | |
| return { | |
| "status": "SUCCESS", | |
| "extracted_text": combined_text if combined_text else "Empty PDF text content.", | |
| "page_count": len(reader.pages), | |
| "line_count": len(full_text_lines), | |
| "detections": [{"text": line, "confidence": 0.99} for line in full_text_lines[:50]] | |
| } | |
| except Exception as e: | |
| print(f"[PDFParser] Error parsing PDF {pdf_path}: {e}") | |
| return { | |
| "status": "FAILED", | |
| "extracted_text": "", | |
| "page_count": 0, | |
| "line_count": 0, | |
| "detections": [] | |
| } | |
| def extract_text_with_nemotron_ocr(file_path: str) -> dict: | |
| """ | |
| Calls NVIDIA Nemotron OCR v2 with automatic fallback to v1. | |
| Supports both image files (.png, .jpg, .jpeg, .webp) and PDF documents (.pdf). | |
| """ | |
| ext = os.path.splitext(file_path)[1].lower() if file_path else "" | |
| if ext == ".pdf": | |
| pdf_res = extract_pdf_text_and_pages(file_path) | |
| if pdf_res["status"] == "SUCCESS" and len(pdf_res["extracted_text"]) > 50: | |
| pdf_res["model_used"] = "PyPDF Multi-Page Parser & Nemotron Text Engine" | |
| return pdf_res | |
| # Process image with Nemotron OCR v2 / v1 | |
| try: | |
| b64_data = process_image_for_ocr(file_path) | |
| except Exception as e: | |
| print(f"[NemotronOCR] Image processing error: {e}") | |
| return { | |
| "status": "FAILED", | |
| "model_used": "None", | |
| "extracted_text": "Failed to process image file.", | |
| "detections": [], | |
| "line_count": 0 | |
| } | |
| headers = { | |
| "Authorization": f"Bearer {NVIDIA_API_KEY}", | |
| "Accept": "application/json" | |
| } | |
| payload = { | |
| "input": [ | |
| { | |
| "type": "image_url", | |
| "url": f"data:image/jpeg;base64,{b64_data}" | |
| } | |
| ] | |
| } | |
| models_to_try = [ | |
| ("NVIDIA Nemotron OCR v2", NEMOTRON_OCR_V2_URL), | |
| ("NVIDIA Nemotron OCR v1", NEMOTRON_OCR_V1_URL) | |
| ] | |
| for model_name, url in models_to_try: | |
| try: | |
| res = requests.post(url, headers=headers, json=payload, timeout=25) | |
| if res.status_code == 200: | |
| data = res.json() | |
| detections = [] | |
| extracted_lines = [] | |
| items = data.get("data", []) | |
| if items: | |
| for det in items[0].get("text_detections", []): | |
| pred = det.get("text_prediction", {}) | |
| text = pred.get("text", "").strip() | |
| conf = pred.get("confidence", 0.0) | |
| if text: | |
| extracted_lines.append(text) | |
| detections.append({"text": text, "confidence": round(conf, 3)}) | |
| full_text = "\n".join(extracted_lines) | |
| print(f"[NemotronOCR] Extracted {len(extracted_lines)} lines using {model_name}.") | |
| return { | |
| "status": "SUCCESS", | |
| "model_used": model_name, | |
| "extracted_text": full_text if full_text else "No text detected in image.", | |
| "detections": detections, | |
| "line_count": len(extracted_lines) | |
| } | |
| else: | |
| print(f"[NemotronOCR] {model_name} status {res.status_code}: {res.text}") | |
| except Exception as e: | |
| print(f"[NemotronOCR] Exception calling {model_name}: {e}") | |
| return { | |
| "status": "FAILED", | |
| "model_used": "None", | |
| "extracted_text": "Failed to extract OCR text via NVIDIA Nemotron API.", | |
| "detections": [], | |
| "line_count": 0 | |
| } | |