File size: 13,297 Bytes
cd964f2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
import os
import cv2
import numpy as np
import easyocr
import warnings
import ssl
import re
from passporteye import read_mrz
from pdf2image import convert_from_path
from PIL import Image
import string as st

from src.utils import (
    clean_string,
    clean_mrz_line,
    parse_date,
    get_country_name,
    get_sex,
    setup_logger,
    clean_name_field
)

from src.fallback_mrz import FallbackMRZ
from config.settings import USE_GPU, OCR_LANGUAGES, TEMP_DIR

warnings.filterwarnings("ignore")
logger = setup_logger(__name__)


# Fix SSL issue (Mac EasyOCR model download fix)
try:
    _create_unverified_https_context = ssl._create_unverified_context
except AttributeError:
    pass
else:
    ssl._create_default_https_context = _create_unverified_https_context


class PassportExtractor:

    def __init__(self, use_gpu=USE_GPU, languages=None):
        self.languages = languages if languages else OCR_LANGUAGES

        base_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
        model_dir = os.path.join(base_dir, "data", "models")
        os.makedirs(model_dir, exist_ok=True)

        logger.debug(f"Initializing EasyOCR Reader (GPU={use_gpu})...")
        
        # Create custom user network directory to avoid permission issues
        user_network_dir = os.path.join(base_dir, "data", "easyocr_user_network")
        os.makedirs(user_network_dir, exist_ok=True)
        
        self.reader = easyocr.Reader(
            self.languages,
            gpu=use_gpu,
            model_storage_directory=model_dir,
            user_network_directory=user_network_dir
        )
        logger.debug("EasyOCR initialized.")

    # ---------------------------------------------------
    # VISUAL GIVEN NAME EXTRACTION (PRIMARY SOURCE)
    # ---------------------------------------------------
    def extract_given_names_from_visual(self, img_path):
        try:
            results = self.reader.readtext(img_path, detail=0)
            lines = [r.strip() for r in results if r.strip()]

            for i, line in enumerate(lines):
                upper_line = line.upper()

                if "GIVEN" in upper_line and "NAME" in upper_line:

                    if ":" in line:
                        candidate = line.split(":")[1].strip()
                    else:
                        if i + 1 < len(lines):
                            candidate = lines[i + 1].strip()
                        else:
                            return ""

                    # Keep only letters and spaces, but preserve spaces between names
                    candidate = re.sub(r'[^A-Za-z\s]', '', candidate)
                    
                    # Clean up extra spaces but keep single spaces between names
                    candidate = re.sub(r'\s+', ' ', candidate).strip()

                    # Remove trailing single letter only if it's clearly an OCR artifact (not part of a name)
                    # This is more conservative - only removes single letters that are likely OCR errors
                    candidate = re.sub(r'([A-Z]{2,})[K]$', r'\1', candidate)  # K is common OCR error for <

                    return candidate.strip()

            return ""

        except Exception as e:
            logger.error(f"Given Names extraction failed: {e}")
            return ""

    # ---------------------------------------------------
    # MRZ EXTRACTION
    # ---------------------------------------------------
    def extract_mrz_from_roi(self, img_path):
        try:
            mrz = read_mrz(img_path, save_roi=True)

            if not mrz:
                return None, None, None

            roi = mrz.aux["roi"]

            if roi.dtype != np.uint8:
                roi = (roi * 255).astype(np.uint8)

            img_resized = cv2.resize(roi, (1110, 140))
            allow = st.ascii_uppercase + st.digits + "<"

            code = self.reader.readtext(
                img_resized,
                detail=0,
                allowlist=allow,
                batch_size=1,  # Single image processing for speed
                workers=0,     # Use main thread for stability
                decoder='greedy'  # Faster decoding
            )

            if len(code) < 2:
                return None, None, mrz

            line1 = clean_mrz_line(code[0])
            line2 = clean_mrz_line(code[1])

            return line1, line2, mrz

        except Exception as e:
            logger.error(f"MRZ extraction failed: {e}")
            return None, None, None

    # ---------------------------------------------------
    # MAIN DATA FUNCTION
    # ---------------------------------------------------
    def get_data(self, img_path, airline="iraqi"):

        if not os.path.exists(img_path):
            logger.error(f"File not found: {img_path}")
            return None

        line1, line2, mrz = self.extract_mrz_from_roi(img_path)

        if line1 and line2:
            mrz = FallbackMRZ(line1, line2)

        if mrz is None:
            logger.warning("MRZ not detected.")
            # Return data with placeholder values instead of None
            return {
                "surname": "•••",
                "name": "•••",
                "country": "•••",
                "nationality": "•••",
                "passport_number": "•••",
                "sex": "•••",
                "date_of_birth": "•••",
                "expiration_date": "•••",
                "personal_number": "•••",
                "mrz_full_string": "",
                "valid_score": 0,
                "mrz_found": False
            }

        surname = clean_name_field(getattr(mrz, "surname", ""))

        # 🔥 ALWAYS prefer visual name
        visual_name = self.extract_given_names_from_visual(img_path)

        if visual_name:
            name = visual_name
        else:
            name = clean_name_field(
                getattr(mrz, "names", getattr(mrz, "name", ""))
            )

            # Final defensive cleanup - only remove trailing K which is a common OCR artifact
            name = re.sub(r'([A-Z]{2,})[K]$', r'\1', name)

        data = {
            "surname": surname,
            "name": name,
            "country": get_country_name(getattr(mrz, "country", "")),
            "nationality": get_country_name(getattr(mrz, "nationality", "")),
            "passport_number": clean_string(getattr(mrz, "number", "")),
            "sex": get_sex(getattr(mrz, "sex", "")),
            "date_of_birth": parse_date(getattr(mrz, "date_of_birth", ""), airline=airline),
            "expiration_date": parse_date(getattr(mrz, "expiration_date", ""), airline=airline),
            "mrz_full_string": (line1 or "") + (line2 or ""),
            "valid_score": getattr(mrz, "valid_score", 0),
            "mrz_found": True,
        }

        return data

    # ---------------------------------------------------
    # PDF PROCESSING
    # ---------------------------------------------------
    def process_pdf(self, pdf_path, progress_callback=None, airline="iraqi"):
        """

        Memory-safe PDF processing for Streamlit free tier with fallback support.

        Converts PDF pages to images and extracts passport data from each page.

        

        Args:

            pdf_path (str): Path to the PDF file.

            progress_callback (function, optional): Progress callback function.

            airline (str): Airline format for date formatting ("iraqi", "default", "fly dubai", "fly baghdad").

            

        Returns:

            list: List of dictionaries with extracted passport data per page.

        """
        # Ensure temp directory exists before processing
        os.makedirs(TEMP_DIR, exist_ok=True)
        
        # Check file size for Streamlit free tier (max 10 MB)
        try:
            file_size = os.path.getsize(pdf_path)
            if file_size > 10 * 1024 * 1024:  # 10 MB limit
                logger.error(f"PDF too large for free tier: {file_size / (1024*1024):.1f} MB")
                return []
        except Exception as e:
            logger.error(f"Could not check file size: {e}")
            return []
        
        results = []
        
        try:
            # Try pdf2image first (primary method)
            from pdf2image import pdfinfo_from_path, convert_from_path
            
            info = pdfinfo_from_path(pdf_path)
            total_pages = info["Pages"]
            
            logger.debug(f"Processing PDF with {total_pages} pages (size: {file_size / (1024*1024):.1f} MB) using pdf2image")
            
            for page in range(1, total_pages + 1):
                try:
                    # Update progress if callback provided
                    if progress_callback:
                        progress_callback(page / total_pages)
                    
                    # Convert ONE page at a time, optimize for speed
                    images = convert_from_path(
                        pdf_path,
                        dpi=150,  # Further reduced DPI for faster processing
                        first_page=page,
                        last_page=page,
                        thread_count=1,  # Single thread for stability
                        use_pdftocairo=True  # Faster backend
                    )
                    
                    image = images[0]
                    
                    # Save temporary image to TEMP_DIR with lower quality for speed
                    temp_image_path = os.path.join(TEMP_DIR, f"temp_page_{page}.jpg")
                    image.save(temp_image_path, "JPEG", quality=70, optimize=True)
                    
                    # Extract passport data with airline-specific formatting
                    result = self.get_data(temp_image_path, airline=airline)
                    
                    if result:
                        result["page_number"] = page
                        results.append(result)
                        logger.debug(f"Successfully extracted data from page {page}")
                    
                    # Delete temp image immediately to free memory
                    if os.path.exists(temp_image_path):
                        os.remove(temp_image_path)
                        
                except Exception as e:
                    logger.error(f"Error on page {page} with pdf2image: {e}")
                    continue
            
            logger.debug(f"PDF processing finished with pdf2image. Valid pages: {len(results)}")
            
        except Exception as pdf2image_error:
            logger.warning(f"pdf2image failed: {pdf2image_error}. Trying fallback with PyMuPDF...")
            
            # Fallback to PyMuPDF (fitz)
            try:
                import fitz
                
                doc = fitz.open(pdf_path)
                total_pages = len(doc)
                
                logger.debug(f"Processing PDF with {total_pages} pages using PyMuPDF fallback")
                
                for i in range(total_pages):
                    try:
                        # Update progress if callback provided
                        if progress_callback:
                            progress_callback((i + 1) / total_pages)
                        
                        page = doc.load_page(i)
                        pix = page.get_pixmap(dpi=200)
                        
                        # Convert to PIL Image
                        img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples)
                        
                        # Save temporary image to TEMP_DIR
                        temp_image_path = os.path.join(TEMP_DIR, f"temp_page_{i+1}.png")
                        img.save(temp_image_path, "PNG")
                        
                        # Extract passport data with airline-specific formatting
                        result = self.get_data(temp_image_path, airline=airline)
                        
                        if result:
                            result["page_number"] = i + 1
                            results.append(result)
                            logger.debug(f"Successfully extracted data from page {i+1}")
                        
                        # Cleanup
                        if os.path.exists(temp_image_path):
                            os.remove(temp_image_path)
                            
                    except Exception as e:
                        logger.error(f"Error on page {i+1} with PyMuPDF: {e}")
                        continue
                
                doc.close()
                logger.debug(f"PDF processing finished with PyMuPDF fallback. Valid pages: {len(results)}")
                
            except Exception as fitz_error:
                logger.error(f"PyMuPDF fallback also failed: {fitz_error}")
                return []
        
        return results