File size: 4,326 Bytes
98689fd | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 | """
OCR utilities for extracting text from images
"""
import os
import io
from typing import Tuple
from PIL import Image, ImageEnhance, ImageFilter
import pytesseract
# Configure Tesseract path for different OS
if os.name == "nt": # Windows
tesseract_path = os.getenv("TESSERACT_PATH", r"C:\Program Files\Tesseract-OCR\tesseract.exe")
if os.path.exists(tesseract_path):
pytesseract.pytesseract.tesseract_cmd = tesseract_path
elif os.name == "posix": # Linux/Mac
# Tesseract should be in PATH or at common locations
common_paths = [
"/usr/bin/tesseract",
"/usr/local/bin/tesseract",
"/opt/homebrew/bin/tesseract"
]
for path in common_paths:
if os.path.exists(path):
pytesseract.pytesseract.tesseract_cmd = path
break
def preprocess_image(image: Image.Image) -> Image.Image:
"""
Preprocess image for better OCR results
Args:
image: PIL Image object
Returns:
Preprocessed PIL Image
"""
# Convert to grayscale
image = image.convert('L')
# Increase contrast
enhancer = ImageEnhance.Contrast(image)
image = enhancer.enhance(2.0)
# Increase sharpness
enhancer = ImageEnhance.Sharpness(image)
image = enhancer.enhance(1.5)
# Apply slight blur to reduce noise
image = image.filter(ImageFilter.MedianFilter(size=3))
# Resize if too small (OCR works better on larger images)
width, height = image.size
if width < 1000 or height < 1000:
scale_factor = max(1000 / width, 1000 / height)
new_size = (int(width * scale_factor), int(height * scale_factor))
image = image.resize(new_size, Image.Resampling.LANCZOS)
return image
def extract_text_from_image(image_bytes: bytes) -> Tuple[str, float]:
"""
Extract text from image using OCR
Args:
image_bytes: Image file bytes
Returns:
Tuple of (extracted_text, confidence_score)
"""
try:
# Load image
image = Image.open(io.BytesIO(image_bytes))
# Preprocess image
processed_image = preprocess_image(image)
# Perform OCR with detailed output
ocr_data = pytesseract.image_to_data(
processed_image,
lang='eng',
output_type=pytesseract.Output.DICT
)
# Extract text and calculate confidence
text_parts = []
confidences = []
for i, conf in enumerate(ocr_data['conf']):
if int(conf) > 0: # Valid confidence
text = ocr_data['text'][i].strip()
if text:
text_parts.append(text)
confidences.append(int(conf))
# Combine text
extracted_text = ' '.join(text_parts)
# Calculate average confidence
avg_confidence = sum(confidences) / len(confidences) if confidences else 0.0
normalized_confidence = avg_confidence / 100.0 # Convert to 0-1 scale
return extracted_text, normalized_confidence
except Exception as e:
print(f"OCR extraction error: {e}")
# Fallback to simple OCR
try:
image = Image.open(io.BytesIO(image_bytes))
text = pytesseract.image_to_string(image, lang='eng')
return text.strip(), 0.5 # Default confidence
except Exception as fallback_error:
print(f"Fallback OCR also failed: {fallback_error}")
raise Exception(f"OCR failed: {str(e)}")
def extract_text_simple(image_bytes: bytes) -> str:
"""
Simple text extraction without preprocessing
Args:
image_bytes: Image file bytes
Returns:
Extracted text string
"""
try:
image = Image.open(io.BytesIO(image_bytes))
text = pytesseract.image_to_string(image, lang='eng')
return text.strip()
except Exception as e:
raise Exception(f"Simple OCR failed: {str(e)}")
def is_tesseract_available() -> bool:
"""
Check if Tesseract OCR is available
Returns:
True if Tesseract is available, False otherwise
"""
try:
pytesseract.get_tesseract_version()
return True
except:
return False
|