Spaces:
Build error
Build error
File size: 5,383 Bytes
cd964f2 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 | import string as st
from dateutil import parser
import logging
import sys
import re
from config.settings import COUNTRY_CODES
def setup_logger(name=__name__):
"""Sets up a logger with standard formatting."""
logger = logging.getLogger(name)
if not logger.handlers:
handler = logging.StreamHandler(sys.stdout)
formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s')
handler.setFormatter(formatter)
logger.addHandler(handler)
logger.setLevel(logging.INFO)
return logger
logger = setup_logger(__name__)
def parse_date(date_obj, airline="iraqi"):
"""Parses a date object or string based on airline format requirements."""
try:
date_str = date_obj.isoformat() if hasattr(date_obj, 'isoformat') else str(date_obj)
date = parser.parse(date_str, yearfirst=True).date()
if airline.lower() == "iraqi":
# Iraqi format (previously Flydubai): DDMMMYY (e.g., 13NOV84)
return date.strftime('%d%b%y').upper()
else:
# Default, Fly Dubai and Fly Baghdad format: DD/MM/YYYY (e.g., 12/1/2023)
return date.strftime('%d/%m/%Y')
except (ValueError, TypeError) as e:
logger.debug(f"Date parsing failed for {date_obj}: {e}")
return str(date_obj)
def clean_string(text):
"""Removes non-alphanumeric characters and converts to uppercase."""
if not text:
return ""
return ''.join(i for i in text if i.isalnum()).upper()
def clean_name_field(text):
"""
Cleans name/surname fields from MRZ.
Handles separators (<<, <) and fixes OCR errors where filler '<' are read as 'K'.
This version is safer and targets only trailing junk 'K's.
"""
if not text:
return ""
text = text.upper()
# Standard MRZ separator between surname and names
text = text.replace("<<", " ")
# Find the last non-'K' character's index
last_good_char_idx = -1
for i in range(len(text) - 1, -1, -1):
if text[i] != 'K':
last_good_char_idx = i
break
# If the string was all 'K's, it's empty.
if last_good_char_idx == -1:
return ""
# Calculate how many 'K's are at the end
trailing_k_count = len(text) - 1 - last_good_char_idx
# If there are 2 or more trailing 'K's, they are junk fillers. Trim them.
if trailing_k_count >= 2:
text = text[:last_good_char_idx + 1]
# Now, any remaining single '<' characters are separators.
text = text.replace("<", " ")
# Remove any numbers from the name
text = ''.join(char for char in text if not char.isdigit())
return text.strip()
def clean_mrz_line(line: str) -> str:
"""Fix bad spacing or bad OCR for MRZ lines."""
if not line:
return ""
line = line.upper().replace(" ", "")
# Remove accidental characters except allowed
allowed = set(st.ascii_uppercase + st.digits + "<")
line = "".join([c for c in line if c in allowed])
# Ensure 44 length (standard TD3 MRZ length)
# Note: TD1/TD2 might be different lengths (30 or 36), but this logic enforces 44.
# We will keep existing logic for consistency but be aware of other formats.
if len(line) < 44:
line += "<" * (44 - len(line))
return line[:44]
def get_country_name(country_code):
"""Resolves 3-letter country code to full name."""
country_code = str(country_code).upper()
for c in COUNTRY_CODES:
if c['alpha-3'] == country_code:
return c['name'].upper()
return country_code
def get_sex(code):
"""Standardizes sex code."""
code = str(code).upper() if code else ''
if code in ['M', 'F']:
return code
if code == '0':
return 'M' # Fallback based on existing logic
return code
def parse_barcode_data(barcode_data):
"""
Parse PDF417 barcode data from passport.
Handles various formats from different countries.
"""
try:
lines = barcode_data.split('\n')
if len(lines) < 2:
return None
# Extract names from first line
name_line = lines[0].replace('@','').replace('<',' ').strip()
surname, given_names = name_line.split(' ', 1) if ' ' in name_line else (name_line, '')
# Parse remaining fields (simplified format)
if len(lines[1]) >= 25:
passport_number = lines[1][:9].strip()
nationality = lines[1][9:12].strip()
dob = parse_date(lines[1][12:18])
sex = lines[1][18]
expiry = parse_date(lines[1][19:25])
return {
"surname": clean_name_field(surname),
"given_names": clean_name_field(given_names),
"passport_number": passport_number,
"nationality": nationality,
"sex": sex,
"date_of_birth": dob,
"expiration_date": expiry,
"barcode_data": barcode_data
}
return None
except Exception as e:
logger.error(f"Error parsing barcode data: {e}")
return None
|