"""Rule-based field validators for document fraud detection. Pure standard library. Every validator returns (ok: bool, reason: str). Rules encode publicly documented validity constraints: - SSN structure per SSA randomization (area 001-899 excluding 666, group 01-99, serial 0001-9999; 000/666/9xx areas, 00 groups and 0000 serials have never been issued) - ICAO Doc 9303 TD3 (passport) MRZ check digits, weights 7-3-1 """ import datetime as _dt import re _SSN_RE = re.compile(r"^(\d{3})-(\d{2})-(\d{4})$") def validate_ssn(value): if value is None: return False, "missing" m = _SSN_RE.match(value.strip()) if not m: return False, "malformed" area, group, serial = (int(x) for x in m.groups()) if area == 0 or area == 666 or 900 <= area <= 999: return False, "invalid_area" if group == 0: return False, "invalid_group" if serial == 0: return False, "invalid_serial" return True, "ok" _DATE_FORMATS = ("%m/%d/%Y", "%Y-%m-%d", "%d.%m.%Y", "%Y%m%d") def parse_date(value): for fmt in _DATE_FORMATS: try: return _dt.datetime.strptime(value.strip(), fmt).date() except (ValueError, AttributeError, TypeError): continue return None def validate_date(value, must_be_past=False, reference=None): d = parse_date(value) if isinstance(value, str) else None if d is None: return False, "unparseable" ref = reference or _dt.date.today() if must_be_past and d > ref: return False, "future_date" return True, "ok" _NAME_RE = re.compile(r"^[A-Z][A-Z' -]{0,63}$", re.IGNORECASE) def validate_name(value): if not value or not value.strip(): return False, "missing" if not _NAME_RE.match(value.strip()): return False, "bad_characters" return True, "ok" _MRZ_ALPHABET = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ" def mrz_check_digit(field): total = 0 for i, ch in enumerate(field): v = 0 if ch == "<" else _MRZ_ALPHABET.index(ch) total += v * (7, 3, 1)[i % 3] return str(total % 10) def validate_mrz_line2(line): line = (line or "").strip() if len(line) != 44: return False, "bad_length" if not re.match(r"^[A-Z0-9<]{44}$", line): return False, "bad_characters" sex = line[20] if not (sex == "<" or sex in "MF"): return False, "bad_sex" for part, label in ((line[13:19], "dob"), (line[21:27], "exp")): if not re.match(r"^\d{6}$", part): return False, "bad_" + label + "_format" mm, dd = int(part[2:4]), int(part[4:6]) if not (1 <= mm <= 12 and 1 <= dd <= 31): return False, "bad_" + label + "_value" composite = line[0:10] + line[13:20] + line[21:43] checks = ( (9, line[0:9], "pno"), (19, line[13:19], "dob"), (27, line[21:27], "exp"), (42, line[28:42], "personal"), (43, composite, "composite"), ) for pos, material, name in checks: if line[pos] != mrz_check_digit(material): return False, "check_digit_" + name return True, "ok" def parse_date_any(value): return parse_date(value) def validate_document(fields, reference=None): findings = [] if "document_number" in fields: ok, why = validate_ssn(fields["document_number"]) if not ok: findings.append("document_number:" + why) if "name" in fields: ok, why = validate_name(fields["name"]) if not ok: findings.append("name:" + why) for key in ("date_of_birth", "issue_date", "expiry_date"): if key in fields: ok, why = validate_date(fields[key], reference=reference) if not ok: findings.append(key + ":" + why) ref = reference or _dt.date.today() dob = parse_date(fields.get("date_of_birth", "")) if dob and dob > ref: findings.append("dob_in_future") if dob and (ref - dob).days > 150 * 365: findings.append("implausible_age") issued = parse_date(fields.get("issue_date", "")) expiry = parse_date(fields.get("expiry_date", "")) if issued and expiry and expiry <= issued: findings.append("expiry_before_issue") if "mrz_line2" in fields: ok, why = validate_mrz_line2(fields["mrz_line2"]) if not ok: findings.append("mrz:" + why) return findings