File size: 5,228 Bytes
890c0e1
45e3486
890c0e1
 
45e3486
 
890c0e1
 
45e3486
890c0e1
 
 
 
 
 
 
 
 
 
 
4229e0c
1a056b8
890c0e1
45e3486
890c0e1
1a056b8
4229e0c
 
1a056b8
 
4229e0c
1a056b8
 
45e3486
 
890c0e1
 
45e3486
 
 
 
 
 
0fd983e
45e3486
644dd41
 
 
45e3486
0fd983e
 
45e3486
644dd41
45e3486
644dd41
 
45e3486
644dd41
 
 
 
 
 
 
 
 
 
 
 
 
45e3486
644dd41
 
 
 
 
 
45e3486
5694c9f
 
 
644dd41
0fd983e
890c0e1
644dd41
890c0e1
 
45e3486
890c0e1
45e3486
644dd41
890c0e1
 
644dd41
 
 
 
 
 
 
45e3486
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e6fcce2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cc09051
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
import string as st
from dateutil import parser
import logging
import sys
import re
from config.settings import COUNTRY_CODES

def setup_logger(name=__name__):
    """Sets up a logger with standard formatting."""
    logger = logging.getLogger(name)
    if not logger.handlers:
        handler = logging.StreamHandler(sys.stdout)
        formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s')
        handler.setFormatter(formatter)
        logger.addHandler(handler)
        logger.setLevel(logging.INFO)
    return logger

logger = setup_logger(__name__)

def parse_date(date_obj, airline="iraqi"):
    """Parses a date object or string based on airline format requirements."""
    try:
        date_str = date_obj.isoformat() if hasattr(date_obj, 'isoformat') else str(date_obj)
        date = parser.parse(date_str, yearfirst=True).date()
        
        if airline.lower() == "iraqi":
            # Iraqi format (previously Flydubai): DDMMMYY (e.g., 13NOV84)
            return date.strftime('%d%b%y').upper()
        else:
            # Default, Fly Dubai and Fly Baghdad format: DD/MM/YYYY (e.g., 12/1/2023)
            return date.strftime('%d/%m/%Y')
            
    except (ValueError, TypeError) as e:
        logger.debug(f"Date parsing failed for {date_obj}: {e}")
        return str(date_obj)

def clean_string(text):
    """Removes non-alphanumeric characters and converts to uppercase."""
    if not text:
        return ""
    return ''.join(i for i in text if i.isalnum()).upper()

def clean_name_field(text):
    """
    Cleans name/surname fields from MRZ.
    Handles separators (<<, <) and fixes OCR errors where filler '<' are read as 'K'.
    This version is safer and targets only trailing junk 'K's.
    """
    if not text:
        return ""
    
    text = text.upper()
    
    # Standard MRZ separator between surname and names
    text = text.replace("<<", " ")
    
    # Find the last non-'K' character's index
    last_good_char_idx = -1
    for i in range(len(text) - 1, -1, -1):
        if text[i] != 'K':
            last_good_char_idx = i
            break
            
    # If the string was all 'K's, it's empty.
    if last_good_char_idx == -1:
        return ""
        
    # Calculate how many 'K's are at the end
    trailing_k_count = len(text) - 1 - last_good_char_idx
    
    # If there are 2 or more trailing 'K's, they are junk fillers. Trim them.
    if trailing_k_count >= 2:
        text = text[:last_good_char_idx + 1]
        
    # Now, any remaining single '<' characters are separators.
    text = text.replace("<", " ")
    
    # Remove any numbers from the name
    text = ''.join(char for char in text if not char.isdigit())
    
    return text.strip()

def clean_mrz_line(line: str) -> str:
    """Fix bad spacing or bad OCR for MRZ lines."""
    if not line:
        return ""
    
    line = line.upper().replace(" ", "")
    
    # Remove accidental characters except allowed
    allowed = set(st.ascii_uppercase + st.digits + "<")
    line = "".join([c for c in line if c in allowed])

    # Ensure 44 length (standard TD3 MRZ length)
    # Note: TD1/TD2 might be different lengths (30 or 36), but this logic enforces 44.
    # We will keep existing logic for consistency but be aware of other formats.
    if len(line) < 44:
        line += "<" * (44 - len(line))
    return line[:44]

def get_country_name(country_code):
    """Resolves 3-letter country code to full name."""
    country_code = str(country_code).upper()
    for c in COUNTRY_CODES:
        if c['alpha-3'] == country_code:
            return c['name'].upper()
    return country_code

def get_sex(code):
    """Standardizes sex code."""
    code = str(code).upper() if code else ''
    if code in ['M', 'F']:
        return code
    if code == '0':
        return 'M' # Fallback based on existing logic
    return code

def parse_barcode_data(barcode_data):
    """
    Parse PDF417 barcode data from passport.
    Handles various formats from different countries.
    """
    try:
        lines = barcode_data.split('\n')
        if len(lines) < 2:
            return None
        
        # Extract names from first line
        name_line = lines[0].replace('@','').replace('<',' ').strip()
        surname, given_names = name_line.split(' ', 1) if ' ' in name_line else (name_line, '')
        
        # Parse remaining fields (simplified format)
        if len(lines[1]) >= 25:
            passport_number = lines[1][:9].strip()
            nationality = lines[1][9:12].strip()
            dob = parse_date(lines[1][12:18])
            sex = lines[1][18]
            expiry = parse_date(lines[1][19:25])
            
            return {
                "surname": clean_name_field(surname),
                "given_names": clean_name_field(given_names),
                "passport_number": passport_number,
                "nationality": nationality,
                "sex": sex,
                "date_of_birth": dob,
                "expiration_date": expiry,
                "barcode_data": barcode_data
            }
        return None
    except Exception as e:
        logger.error(f"Error parsing barcode data: {e}")
        return None