Instructions to use SlayerLab/NERGAL with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SlayerLab/NERGAL with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("token-classification", model="SlayerLab/NERGAL")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForTokenClassification tokenizer = AutoTokenizer.from_pretrained("SlayerLab/NERGAL") model = AutoModelForTokenClassification.from_pretrained("SlayerLab/NERGAL", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Add hybrid PII scripts: nergal.py and frozen scrub_pii.py
Browse files- scrub_pii.py +746 -0
scrub_pii.py
ADDED
|
@@ -0,0 +1,746 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Regex PII scrub for Polish web/official text.
|
| 2 |
+
|
| 3 |
+
Replaces emails, phones, PESEL/NIP/REGON/KRS, land-register (KW) numbers, electronic contact addresses and
|
| 4 |
+
account numbers in place so sentence structure survives. Names of public officials are left untouched —
|
| 5 |
+
that is intentional, not a gap. Phones map to [Telefon]; everything else
|
| 6 |
+
to [PII]. Bare PESEL requires checksum and date validation; explicit identifier
|
| 7 |
+
labels also redact damaged numbers (including common OCR O/I/l substitutions).
|
| 8 |
+
NIP/REGON/KRS and identity documents require explicit nearby labels. KW numbers
|
| 9 |
+
need a nearby KW/księga wieczysta label, or the full XX0X/00000000/0 form with a
|
| 10 |
+
valid check digit. Passport
|
| 11 |
+
variants include one-letter and diplomatic IDs. Foreign IBANs use country
|
| 12 |
+
lengths and MOD-97. ePUAP paths require a label; e-Doreczenia uses its AE:PL form.
|
| 13 |
+
Wrapped email domains, local parts hyphenated across one line break and small
|
| 14 |
+
extraction gaps around @/hyphens are supported.
|
| 15 |
+
Explicit phone extensions and terminal suffix ranges are included; room numbers are not. Labelled numeric
|
| 16 |
+
PINs (including URL pin= values) map to [PII] before phone detection.
|
| 17 |
+
Contact/helpline headings cover consecutive descriptive phone-list entries;
|
| 18 |
+
unrelated lines end the list. Bounded staff/address-directory evidence also covers
|
| 19 |
+
formatted phone fields. Labelled full-number ranges retain shared prefixes.
|
| 20 |
+
Phones require a nearby contact cue, a Polish +48/0048 prefix, or explicit
|
| 21 |
+
international country/trunk notation such as +CC (0). With a cue, the EUR-Lex
|
| 22 |
+
"(32-2) 299 11 11" country-area form counts as international. Short service numbers
|
| 23 |
+
need strong labels; 116xxx numbers also accept nearby telephone prose. Unlabelled
|
| 24 |
+
domestic numbers are left for audit because table cells have the same shapes.
|
| 25 |
+
Flattened tables can glue labels to both neighbours ("Mödlingtel.: … 38112faks:");
|
| 26 |
+
such glued labels count as labels and end the preceding number.
|
| 27 |
+
|
| 28 |
+
Call after HTML-to-text, before the parquet is written.
|
| 29 |
+
"""
|
| 30 |
+
from __future__ import annotations
|
| 31 |
+
import datetime as dt
|
| 32 |
+
import re
|
| 33 |
+
|
| 34 |
+
PHONE_TAG = "[Telefon]"
|
| 35 |
+
PII_TAG = "[PII]"
|
| 36 |
+
COUNTS = ("email", "phone", "pesel", "nip", "regon", "account", "document", "krs", "electronic_address", "pin",
|
| 37 |
+
"land_register")
|
| 38 |
+
|
| 39 |
+
# Mobile + geographic area codes (2-digit national prefix after trunk 0 / +48).
|
| 40 |
+
_PL_PREFIX = {
|
| 41 |
+
"12", "13", "14", "15", "16", "17", "18", "22", "23", "24", "25", "26", "29",
|
| 42 |
+
"32", "33", "34", "39", "41", "42", "43", "44", "45", "46", "47", "48",
|
| 43 |
+
"50", "51", "52", "53", "54", "55", "56", "57", "58", "59",
|
| 44 |
+
"60", "61", "62", "63", "65", "66", "67", "68", "69",
|
| 45 |
+
"70", "71", "72", "73", "74", "75", "76", "77", "78", "79",
|
| 46 |
+
"80", "81", "82", "83", "84", "85", "86", "87", "88", "89",
|
| 47 |
+
"91", "94", "95",
|
| 48 |
+
}
|
| 49 |
+
|
| 50 |
+
_EMAIL_DOMAIN = r"[^\W_](?:[^\W_]|-[ \t]{0,3}(?=[^\W_]))*"
|
| 51 |
+
# PDF extraction can hyphenate a local part across one line break: "jan-\nna.k@example.com".
|
| 52 |
+
# The fragment is capped at the 64-character local-part limit so long tokens stay linear.
|
| 53 |
+
_EMAIL_START = r"\b(?:[\w.%+&-]{0,63}[^\W_]-[ \t]*\r?\n[ \t]*)?[\w.%+&-]+[ \t]{0,3}@"
|
| 54 |
+
_EMAIL_RE = re.compile(_EMAIL_START + r"[ \t]{0,3}(?:" + _EMAIL_DOMAIN + r"\.)+[^\W\d_]{2,}\b")
|
| 55 |
+
_EMAIL_WRAP_RE = re.compile(_EMAIL_START + r"[ \t]{0,3}(?:" + _EMAIL_DOMAIN
|
| 56 |
+
+ r"\.)+[ \t]*\r?\n[ \t]*(?:" + _EMAIL_DOMAIN + r"\.)*[^\W\d_]{2,}\b")
|
| 57 |
+
# Damaged contact fields can retain only the local part and @.
|
| 58 |
+
_EMAIL_FRAGMENT_RE = re.compile(_EMAIL_START + r"(?=[ \t]*\r?$)", re.M)
|
| 59 |
+
_EMAIL_LABEL_RE = re.compile(r"\be[ -]?mail[ \t]*:[ \t]*\Z", re.I)
|
| 60 |
+
_EDELIVERY_RE = re.compile(r"\bAE:PL-\d{5}-\d{5}-[A-Z0-9]{5}-\d{2}(?![\w-])", re.I)
|
| 61 |
+
_EPUAP_RE = re.compile(r"(?<![\w/])/[\w-]+/[\w-]+(?![\w/-])")
|
| 62 |
+
_EPUAP_LABEL_RE = re.compile(
|
| 63 |
+
r"\b(?:e[ -]?puap|elektroniczna[ \t]+skrzynka[ \t]+podawcza)\b[^\n/\[\]]{0,50}\Z", re.I)
|
| 64 |
+
_PIN_RE = re.compile(
|
| 65 |
+
r"(?P<label>\bPIN[ \t]*[:=][ \t]*)"
|
| 66 |
+
r"(?P<number>[\u202a-\u202e\u2066-\u2069]*\d(?:[ \t]?\d){3,15}#?)"
|
| 67 |
+
r"(?!\w|[ \t]*\d)", re.I)
|
| 68 |
+
# Horizontal Unicode spaces from HTML/PDF extraction; preserve paragraph breaks.
|
| 69 |
+
_SPACES = str.maketrans({c: " " for c in "\u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u202f\u205f\u3000"})
|
| 70 |
+
# Country lengths checked against Apache Commons Validator's IBAN registry table:
|
| 71 |
+
# https://commons.apache.org/proper/commons-validator/xref/org/apache/commons/validator/routines/IBANValidator.html
|
| 72 |
+
# ponytail: country/length/MOD-97 only; add national BBAN rules if false positives appear.
|
| 73 |
+
_IBAN_LENGTHS = {
|
| 74 |
+
country: length for length, countries in {
|
| 75 |
+
15: "NO", 16: "BE", 18: "DK FI AX FK FO GL NL SD",
|
| 76 |
+
19: "MK SI", 20: "AT BA EE KZ LT LU MN XK",
|
| 77 |
+
21: "CH HR LI LV", 22: "BG BH CR DE GB IM JE GG GE IE ME RS VA",
|
| 78 |
+
23: "AE GI IL IQ OM SO TL", 24: "AD CZ ES MD PK RO SA SE SK TN VG",
|
| 79 |
+
25: "LY PT ST", 26: "IS TR",
|
| 80 |
+
27: "BI DJ FR GF GP MQ RE PF TF YT NC BL MF PM WF GR IT MC MR SM",
|
| 81 |
+
28: "AL AZ BY CY DO GT HN HU LB NI PL SV", 29: "BR EG PS QA UA",
|
| 82 |
+
30: "JO KW MU YE", 31: "MT SC", 32: "LC", 33: "RU",
|
| 83 |
+
}.items() for country in countries.split()
|
| 84 |
+
}
|
| 85 |
+
_IBAN_RE = re.compile(
|
| 86 |
+
r"\b(?:" + "|".join(
|
| 87 |
+
country + r"[ \t-]*[0-9]{2}(?:[ \t-]*[A-Z0-9]){" + str(length - 4) + "}"
|
| 88 |
+
for country, length in _IBAN_LENGTHS.items()
|
| 89 |
+
) + r")\b", re.I,
|
| 90 |
+
)
|
| 91 |
+
# Optional PL, then 26 digits with short space/tab/hyphen gaps (invoice style).
|
| 92 |
+
_ACCOUNT_RE = re.compile(r"\b(?:PL[ \t-]*)?(?:\d[ \t-]*){25}\d\b", re.I)
|
| 93 |
+
_REGON14_RE = re.compile(r"\b\d{14}\b")
|
| 94 |
+
_PESEL_RE = re.compile(r"\b\d{11}\b")
|
| 95 |
+
_NIP_DASH_RE = re.compile(r"\b(?:PL[ \t]*)?\d{3}(?:[- \t]\d{3}[- \t]\d{2}[- \t]\d{2}|[- \t]\d{2}[- \t]\d{2}[- \t]\d{3})\b", re.I)
|
| 96 |
+
_NIP_RE = re.compile(r"\b(?:PL[ \t]*)?\d{10}\b", re.I)
|
| 97 |
+
_DOCUMENT_RE = re.compile(
|
| 98 |
+
r"\b(?P<label>(?:dow[oó]d(?:u|em)?[ \t]+osobist(?:y|ego|ym)|"
|
| 99 |
+
r"(?:nr|numer)[ \t]+dowodu|paszport(?:u|em)?(?:[ \t]+dyplomatyczn(?:y|ego|ym))?)"
|
| 100 |
+
r"[ \t]*(?:(?:seria[ \t]+i[ \t]+numer|serii|seria|numer|nr)\.?[ \t]*)?"
|
| 101 |
+
r"[:=-]?[ \t]*)"
|
| 102 |
+
r"(?P<number>[A-Z]{3}[ \t-]?[0-9OIl]{6}|[A-Z]{2}[ \t-]?[0-9OIl]{7}|[A-Z][ \t-]?[0-9]{6,9}|[0-9]{4,12})\b", re.I,
|
| 103 |
+
)
|
| 104 |
+
_LABELLED_ID_RE = re.compile(
|
| 105 |
+
r"\b(?P<label>(?P<kind>PESEL|NIP|REGON)\b"
|
| 106 |
+
r"(?:[ \t]+(?:Gminy|Powiatu|Miasta|firmy)(?:[ \t]+[^\W\d_][^\W\d_-]*){0,4})?[ \t]{0,8}"
|
| 107 |
+
r"(?:(?:nr\.?|numer)[ \t]{0,8})?[:=.\-]?[ \t]{0,8}(?:\r?\n[ \t]{0,8})?)"
|
| 108 |
+
r"(?P<number>(?:PL[ \t]*)?(?:[0-9OIl]{9,14}|\d(?:[ \t-]?\d){8,13}))"
|
| 109 |
+
r"(?!\w|[ \t-]*\d)", re.I,
|
| 110 |
+
)
|
| 111 |
+
# Land-register (KW) number: court code / number / check digit, e.g. WA1M/00123456/3.
|
| 112 |
+
_LAND_REGISTER_RE = re.compile(
|
| 113 |
+
r"\b(?P<court>[A-Z]{2}\d[A-Z])[ \t]*[/ \t][ \t]*(?P<number>\d{1,8})[ \t]*/[ \t]*(?P<check>\d)(?![\w/])")
|
| 114 |
+
_LAND_REGISTER_LABEL_RE = re.compile(
|
| 115 |
+
r"(?:(?-i:\bKW\b)|\bks(?:\.|i[eęą]\w*)[ \t]+wieczyst\w*)[^\n]{0,30}\Z", re.I)
|
| 116 |
+
# Check-digit values; court codes skip Q and V.
|
| 117 |
+
_LAND_REGISTER_VALUES = {c: i for i, c in enumerate("0123456789XABCDEFGHIJKLMNOPRSTUWYZ")}
|
| 118 |
+
_REGON9_RE = re.compile(r"\b\d{9}\b")
|
| 119 |
+
_ID_LABEL_END = r"[ \t]{0,8}(?:(?:nr\.?|numer)[ \t]{0,8})?[:=.\-]?[ \t]{0,8}(?:\r?\n[ \t]{0,8})?\Z"
|
| 120 |
+
_NIP_LABEL_RE = re.compile(r"\bNIP\b" + _ID_LABEL_END, re.I)
|
| 121 |
+
_REGON_LABEL_RE = re.compile(r"\bREGON\b" + _ID_LABEL_END, re.I)
|
| 122 |
+
_KRS_RE = re.compile(r"\b\d{6,10}(?!\w|[ \t-]*\d)")
|
| 123 |
+
# Abbreviation or written-out register name, optionally "pod nr/numerem".
|
| 124 |
+
_KRS_LABEL_RE = re.compile(
|
| 125 |
+
r"(?:\bKRS\b|\bKrajow\w*[ \t]+Rejestr\w*[ \t]+S[aą]dow\w*)"
|
| 126 |
+
r"(?:[ \t]{0,8},?[ \t]{0,8}pod[ \t]{1,8}(?:nr\.?|numerem))?" + _ID_LABEL_END, re.I)
|
| 127 |
+
# Separators are short and local: one newline *or* a few punct/spaces.
|
| 128 |
+
# Letters and blank lines break the match so body text is not swallowed.
|
| 129 |
+
_SEP = r"(?:[ \t./\-()\u2013\u2014]{0,3}|\n)"
|
| 130 |
+
_VANITY_PHONE_RE = re.compile(
|
| 131 |
+
r"(?<!\w)\+1[ \t-]+\d{3}[ \t-]+(?:\d{3}-[A-Z]{4}|[A-Z]{7})"
|
| 132 |
+
r"(?:[ \t]+\(\d{4,7}\))?(?!\w)")
|
| 133 |
+
# Flattened tables glue labels to both neighbours: "Mödlingtel.: +43 (0) 512 34-56712faks:".
|
| 134 |
+
# Glued "tel" needs its dot and glued "fax" a non-letter before it ("Hotel:", "Halifax:").
|
| 135 |
+
_GLUED_PHONE_LABEL = r"(?:tel\.|telefon|t[ée]l[ée]phone|t[ée]l[ée]copieur|(?<![^\W\d_])fa(?:x|ks))[ \t]*:"
|
| 136 |
+
# A number ends at a word boundary or where a glued contact label starts.
|
| 137 |
+
_PHONE_END = r"(?:(?!\w)|(?=" + _GLUED_PHONE_LABEL + r"|(?:e-?mail|t[ée]lex)[ \t]*:))"
|
| 138 |
+
_PHONE_RE = re.compile(
|
| 139 |
+
_VANITY_PHONE_RE.pattern + r"|(?:(?<!\w)|(?<=telefon)|(?<=tel)|(?<=faks)|(?<=fax))(?:\(?\+[ \t]*)?"
|
| 140 |
+
r"(?:\((?:0|0?\d{2,4}(?:-\d{1,2})?)\)" + _SEP + r")?"
|
| 141 |
+
r"\d(?:" + _SEP + r"\d){4,}" + _PHONE_END, re.I,
|
| 142 |
+
)
|
| 143 |
+
_SHORT_PHONE_LABEL_RE = re.compile(
|
| 144 |
+
r"(?:\b(?:tel(?:efon\w*)?(?:(?:\.[ \t]*|[ \t]+)(?:kom[oó]rk\w*|lokaln\w*|stacjonarn\w*))?|fax|faks(?:em)?|phone|"
|
| 145 |
+
r"toll[ -]free|numer[ \t]+bezpłatny)|[☎☏📞📱])[.: \t/]{0,8}\Z|"
|
| 146 |
+
r"\binfolini[^\W\d_]*[^\d\n]{0,80}:[ \t]*\Z|" + _GLUED_PHONE_LABEL + r"[ \t]*\Z", re.I)
|
| 147 |
+
_PHONE_LABEL_RE = re.compile(
|
| 148 |
+
r"\b(?:tel(?:efon[^\W\d_]*)?|fax|faks(?:em)?|kom(?:[oó]rk[^\W\d_]*)?|gsm|"
|
| 149 |
+
r"infolini[^\W\d_]*|(?:za)?dzwo[ńn][^\W\d_]*|Blikiem|"
|
| 150 |
+
r"lini[ęaąie][ \t]+wsparcia)(?![^\W\d_])|"
|
| 151 |
+
r"\bkontakt(?=tel[.:])|[☎☏📞📱]", re.I,
|
| 152 |
+
)
|
| 153 |
+
_PHONE_BREAK_RE = re.compile(
|
| 154 |
+
r"\n|\.[ \t]+(?=\d)|[ \t]+(?=\(?(?:[01]?\d|2[0-3])[.:][0-5]\d"
|
| 155 |
+
r"|\d{1,2}[.)][ \t]+\d{1,2}[./]\d{1,2}[./](?:19|20)\d{2})")
|
| 156 |
+
_SECTION_MARKER_RE = re.compile(r"\d{1,2}[.)](?!\d)")
|
| 157 |
+
_PHONE_EXTENSION_RE =re.compile(r"(?:(?:[ \t]+,?[ \t]*|,[ \t]*)wew(?:n(?:ętrzny)?)?\.?[ \t]*\d{1,5}|[ \t]+do[ \t]+\d{1,3}(?=[ \t]*(?:[.;,](?!\d)|\r?\n|\Z)))(?!\w|[ \t]*\d)", re.I)
|
| 158 |
+
_SERVICE_PHONE_RE = re.compile(r"(?<![\w+])[1-9]\d{2}(?![\w\d]|[ \t./()-]*\d)")
|
| 159 |
+
_SERVICE_PHONE_LABEL_RE = re.compile(
|
| 160 |
+
r"(?:\b(?:tel(?:efon\w*)?|phone)[.: \t]{0,8}(?:alarmow\w*[.: \t]{0,8})?|"
|
| 161 |
+
r"\b(?:numer[ \t]+)?bezpłatny[.: \t]{0,8}|"
|
| 162 |
+
r"\b(?:bezpłatny[ \t]+)?numer[ \t]+alarmowy[.: \t]{0,8}|"
|
| 163 |
+
r"(?:^|\n)[ \t]*(?:Straż pożarna|Policja|Pogotowie|Ratunek|Lekarz pogotowia ratunkowego)"
|
| 164 |
+
r"(?:[ \t]+\([^()\n]{1,60}\))?[ \t]*:[ \t]*)"
|
| 165 |
+
r"(?:[1-9]\d{2}[ \t]*(?:[,;]|i|lub|oraz)[ \t]*)*\Z", re.I)
|
| 166 |
+
_OTHER_NUMBER_LABEL_RE = re.compile(r"\b(?:NIP|REGON|PESEL|KRS|ISBN|kod)\b[^\d\n]{0,20}\Z", re.I)
|
| 167 |
+
_PHONE_LIST_HEADING_RE = re.compile(
|
| 168 |
+
r"[ \t]*(?:numery[ \t]+kontaktowe|telefony(?:[ \t]+(?:kontaktowe|zaufania))?|fax|faks|zapisz[ \t]+się|"
|
| 169 |
+
r"(?:[^.!?\n]{0,40}\?[ \t]*)?(?:za)?dzwoń!?)"
|
| 170 |
+
r"(?:[ \t]*\([^()\n]{1,80}\))?[ \t]*:?[ \t]*", re.I)
|
| 171 |
+
_CONTACT_INTRO_RE = re.compile(
|
| 172 |
+
r"\b(?:kontakt\b|zapisy\b|zapisz[ \t]+się\b|rejestracja\b)"
|
| 173 |
+
r"(?:nr\.|[^\n.!?;]){0,75}\Z", re.I)
|
| 174 |
+
_PHONE_HEADING_RE = re.compile(
|
| 175 |
+
r"(?<!\[)\b(?:tel(?:efon[^\W\d_]*)?\b|kontakt\b|zapisy\b|zapisz[ \t]+się\b|"
|
| 176 |
+
r"(?:za)?dzwo[ńn][^\W\d_]*\b|(?:pod|na)[ \t]+numer(?:em)?)"
|
| 177 |
+
r"(?:tj\.|nr\.|[^\d\n.!?;\[\]]){0,75}:?[ \t]*(?:\n[ \t]*){1,4}\Z", re.I)
|
| 178 |
+
_CONTACT_EXCLUDE_RE = re.compile(r"\b(?:spraw\w*|kod\w*|kwot\w*|statystyk\w*|taryf\w*)\b", re.I)
|
| 179 |
+
_PHONE_LINE_RE = re.compile(r"(?<!\w)\d{2,3}(?:[ \t-]\d{2,4}){2,3}(?!\w)")
|
| 180 |
+
_STAFF_ROLE_RE = re.compile(
|
| 181 |
+
r"\b(?:inspektor|referent|koordynator|psycholog|księgowość|księgowy|księgowa|"
|
| 182 |
+
r"pracownicy[ \t]+socjalni|łowczy|podłowczy)\b", re.I)
|
| 183 |
+
|
| 184 |
+
|
| 185 |
+
def _phone_list_context(text: str, start: int) -> bool:
|
| 186 |
+
# ponytail: 2,000-character lookback and 400-character entry suffix; longer
|
| 187 |
+
# lists need repeated headings or a section parser. Never cross prose.
|
| 188 |
+
end = text.find('\n', start, start + 401)
|
| 189 |
+
if end < 0 and len(text) > start + 400:
|
| 190 |
+
return False
|
| 191 |
+
lines = text[max(0, start - 2000):end if end >= 0 else len(text)].splitlines()
|
| 192 |
+
if start > 2000:
|
| 193 |
+
lines = lines[1:] # A truncated line cannot establish a heading.
|
| 194 |
+
for i, line in enumerate(reversed(lines)):
|
| 195 |
+
if _PHONE_LIST_HEADING_RE.fullmatch(line):
|
| 196 |
+
return i > 0
|
| 197 |
+
if not line.strip():
|
| 198 |
+
continue
|
| 199 |
+
phone = _PHONE_RE.search(line)
|
| 200 |
+
if not phone or not _phone_ok(phone[0], short=True) or _is_amount(line, *phone.span()):
|
| 201 |
+
return False
|
| 202 |
+
if i == 0 and phone.start() != start - (text.rfind('\n', 0, start) + 1):
|
| 203 |
+
return False # Numbers in an entry's description are not list items.
|
| 204 |
+
before, after = line[:phone.start()].strip(), line[phone.end():].strip()
|
| 205 |
+
if _OTHER_NUMBER_LABEL_RE.search(before):
|
| 206 |
+
return False
|
| 207 |
+
# Either "City: number" or "number – helpline description". Bare
|
| 208 |
+
# numeric rows are ambiguous even below an earlier contact heading.
|
| 209 |
+
named = re.fullmatch(r"[^\W\d_][^\d\n:;]{0,79}[:–—-]", before)
|
| 210 |
+
described = before in ('', '•', '-', '*') and re.match(r"[–—-][ \t]+[^\W\d_]", after)
|
| 211 |
+
dialling = re.fullmatch(r"[-•*]?[ \t]*(?:dzwoniąc[ \t]+)?z[ \t]+[^\W\d_][^\d\n:;.!?]{0,100}", before, re.I)
|
| 212 |
+
if not (((named or dialling) and after in ('', ',', ';', '.')) or described):
|
| 213 |
+
return False
|
| 214 |
+
return False
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
def _phone_context(text: str, start: int, raw: str) -> bool:
|
| 218 |
+
# A previous redaction is a boundary, not a fresh "Telefon" cue. Include
|
| 219 |
+
# one marker's extra width so the 75-character window cannot bisect it.
|
| 220 |
+
before = re.split(r"\n[ \t]*\n|\[Telefon\]|\[PII\]",
|
| 221 |
+
text[max(0, start - 75 - len(PHONE_TAG)):start])[-1][-75:]
|
| 222 |
+
if _OTHER_NUMBER_LABEL_RE.search(before):
|
| 223 |
+
return False
|
| 224 |
+
intro = _CONTACT_INTRO_RE.search(before)
|
| 225 |
+
if intro and not _CONTACT_EXCLUDE_RE.search(intro[0]):
|
| 226 |
+
return True
|
| 227 |
+
# A contact email must not erase an explicit contact label for the phone
|
| 228 |
+
# following it. Other redactions still form boundaries.
|
| 229 |
+
if re.search(r'\bkontakt[ \t]*:[ \t]*\[PII\][ \t,;]*\Z',
|
| 230 |
+
text[max(0, start-75):start], re.I):
|
| 231 |
+
return True
|
| 232 |
+
lead = re.split(r'\[Telefon\]|\[PII\]', text[max(0, start-240):start])[-1]
|
| 233 |
+
# "Pod numerem" also introduces contract/case references. Require a
|
| 234 |
+
# communication cue when the number has no explicit telephone label.
|
| 235 |
+
number_contact = bool((_PHONE_LABEL_RE.search(lead) or re.search(
|
| 236 |
+
r'\b(?:informacj[^\W\d_]*|zgłosz[^\W\d_]*|zapis[^\W\d_]*|SMS(?:-a)?|połączeni[^\W\d_]*)\b', lead, re.I))
|
| 237 |
+
and not _CONTACT_EXCLUDE_RE.search(lead)
|
| 238 |
+
and not re.search(r'\b(?:umow|faktur|dokument)[^\W\d_]*\b', lead, re.I))
|
| 239 |
+
heading = _PHONE_HEADING_RE.search(text[max(0, start-160):start])
|
| 240 |
+
if (heading and not _CONTACT_EXCLUDE_RE.search(heading[0])
|
| 241 |
+
and (not re.match(r'(?:pod|na)\b', heading[0], re.I)
|
| 242 |
+
or _PHONE_LABEL_RE.search(heading[0]) or number_contact)):
|
| 243 |
+
return True
|
| 244 |
+
if (number_contact and re.search(r'\b(?:pod[ \t]+numerem|na[ \t]+numer)[ \t]*:?[ \t]*\Z', before, re.I)):
|
| 245 |
+
return True
|
| 246 |
+
if re.search(r'\bтелефону[ \t]*:[ \t]*\Z', before, re.I):
|
| 247 |
+
return True
|
| 248 |
+
# An international phone may precede a labelled fax in the same contact list.
|
| 249 |
+
# Keep this cue adjacent: arbitrary later contact prose is not evidence.
|
| 250 |
+
after = text[start + len(raw):start + len(raw) + 40]
|
| 251 |
+
if re.match(r'[ \t]*[–—-][ \t]+telefon\b', after, re.I):
|
| 252 |
+
return True
|
| 253 |
+
if (re.match(r"\(?(?:\+|00)", raw)
|
| 254 |
+
and re.match(r"[ \t]*[,;][ \t]*(?:fax|faks(?:em)?)[.: \t]*(?=[+(0-9])", after, re.I)):
|
| 255 |
+
return True
|
| 256 |
+
# A foreign-looking +number can also be an increment in a financial table.
|
| 257 |
+
return bool(re.match(r"\(?(?:\+|00)[ \t]*48|\+\d{1,3}[ \t]+\(0\)", raw)
|
| 258 |
+
or _PHONE_LABEL_RE.search(before))
|
| 259 |
+
|
| 260 |
+
|
| 261 |
+
def _directory_context(text: str, start: int, end: int) -> bool:
|
| 262 |
+
"""Recognize phones in bounded staff and named-address contact records."""
|
| 263 |
+
raw = text[start:end]
|
| 264 |
+
foreign = raw.startswith('+') and _phone_ok(raw)
|
| 265 |
+
if not foreign and (not _pl_national_ok(_digits(raw)) or not re.search(r'[ \t-]', raw)):
|
| 266 |
+
return False
|
| 267 |
+
# ponytail: local 1,600-character directory evidence; longer isolated rows
|
| 268 |
+
# need preserved upstream table structure, not an unbounded document cue.
|
| 269 |
+
left, right = max(0, start-800), min(len(text), end+800)
|
| 270 |
+
nearby = text[left:right]
|
| 271 |
+
line_start, line_end = text.rfind('\n', 0, start)+1, text.find('\n', end)
|
| 272 |
+
line_end = len(text) if line_end < 0 else line_end
|
| 273 |
+
before, after = text[line_start:start], text[end:line_end]
|
| 274 |
+
if _CONTACT_EXCLUDE_RE.search(before) or _OTHER_NUMBER_LABEL_RE.search(before):
|
| 275 |
+
return False
|
| 276 |
+
if foreign:
|
| 277 |
+
# A comma-delimited name/address entry needs a labelled phone nearby;
|
| 278 |
+
# a +number alone could be an increment in a financial table.
|
| 279 |
+
address = re.search(r'(?:^|,)[ \t]*[^\W\d_][^,\n]{1,120},'
|
| 280 |
+
r'[^\n]{0,160}\d[^\n]{0,80},[ \t]*\Z', before)
|
| 281 |
+
return bool(address and any(
|
| 282 |
+
_SHORT_PHONE_LABEL_RE.search(nearby[max(0,m.start()-80):m.start()])
|
| 283 |
+
for m in _PHONE_RE.finditer(nearby) if m[0].startswith('+') and _phone_ok(m[0])))
|
| 284 |
+
standalone = not before.strip() and not after.strip()
|
| 285 |
+
lead = text[max(0, start-400):start]
|
| 286 |
+
if (standalone and re.search(r'\b(?:infolini[^\W\d_]*|helpline)\b', lead, re.I)
|
| 287 |
+
and not _CONTACT_EXCLUDE_RE.search(lead)):
|
| 288 |
+
return True
|
| 289 |
+
if (standalone and re.search(r'\bnr\.?[ \t]+telefonu\b', nearby, re.I)
|
| 290 |
+
and re.search(r'\b(?:pok[oó]j|pokoju)\b', nearby, re.I)
|
| 291 |
+
and re.search(r'\bnazwisko\b', nearby, re.I)):
|
| 292 |
+
return True
|
| 293 |
+
# Email/name/role records and phone-led service directories repeat formatted
|
| 294 |
+
# values. A lone staff title beside a number is insufficient evidence.
|
| 295 |
+
phones = [m for m in _PHONE_LINE_RE.finditer(nearby)
|
| 296 |
+
if _pl_national_ok(_digits(m[0])) and not _is_amount(nearby, *m.span())]
|
| 297 |
+
roles = list(_STAFF_ROLE_RE.finditer(nearby))
|
| 298 |
+
if len(phones) < 2 or len(roles) < 2:
|
| 299 |
+
return False
|
| 300 |
+
return bool((re.search(r'\[PII\]', before[-100:] + after[:100]) and
|
| 301 |
+
re.search(r'[^\W\d_]', before[-100:]))
|
| 302 |
+
or (not before.strip() and re.match(r'[ \t]+[^\W\d_]', after))
|
| 303 |
+
or _STAFF_ROLE_RE.search(before[-100:]))
|
| 304 |
+
|
| 305 |
+
|
| 306 |
+
# Amounts can look like phones or checksum-valid national IDs. Keep the
|
| 307 |
+
# currency adjacent to the number, allowing a decimal, multiplier and line wrap.
|
| 308 |
+
_MONEY_GAP = r"[ \t]*(?:\n[ \t]*)?"
|
| 309 |
+
_CURRENCY = r"(?:PLN|z[łl](?:ot(?:ych|ego|emu|ymi|ym|y|e))?|EUR(?:O)?|USD|GBP|CHF)\b|[%€$£]"
|
| 310 |
+
_MONEY_AFTER_RE = re.compile(
|
| 311 |
+
r"(?:[,.][0-9]+)?" + _MONEY_GAP
|
| 312 |
+
+ r"(?:(?:tys\.?|mln|mld|bln)\b\.?" + _MONEY_GAP + r")?"
|
| 313 |
+
+ r"(?:" + _CURRENCY + r")", re.I,
|
| 314 |
+
)
|
| 315 |
+
_MONEY_BEFORE_RE = re.compile(r"(?:\b(?:PLN|EUR|USD|GBP|CHF)|[€$£])" + _MONEY_GAP + r"$", re.I)
|
| 316 |
+
|
| 317 |
+
|
| 318 |
+
def _is_amount(text: str, start: int, end: int) -> bool:
|
| 319 |
+
return bool(_MONEY_AFTER_RE.match(text, end) or _MONEY_BEFORE_RE.search(text[max(0, start - 32):start]))
|
| 320 |
+
|
| 321 |
+
|
| 322 |
+
def _digits(s: str) -> str:
|
| 323 |
+
return re.sub(r"\D", "", s)
|
| 324 |
+
|
| 325 |
+
|
| 326 |
+
def _pesel_ok(d: str) -> bool:
|
| 327 |
+
if len(d) != 11 or not d.isdigit():
|
| 328 |
+
return False
|
| 329 |
+
weights = (1, 3, 7, 9, 1, 3, 7, 9, 1, 3)
|
| 330 |
+
check = sum(w * int(x) for w, x in zip(weights, d[:-1]))
|
| 331 |
+
if str((10 - check % 10) % 10) != d[-1]:
|
| 332 |
+
return False
|
| 333 |
+
yy, mm, dd = int(d[0:2]), int(d[2:4]), int(d[4:6])
|
| 334 |
+
century = {0: 1900, 1: 2000, 2: 2100, 3: 2200, 4: 1800}.get(mm // 20)
|
| 335 |
+
if century is None:
|
| 336 |
+
return False
|
| 337 |
+
try:
|
| 338 |
+
dt.date(century + yy, mm % 20, dd)
|
| 339 |
+
except ValueError:
|
| 340 |
+
return False
|
| 341 |
+
return True
|
| 342 |
+
|
| 343 |
+
|
| 344 |
+
def _nip_ok(d: str) -> bool:
|
| 345 |
+
if len(d) != 10 or not d.isdigit():
|
| 346 |
+
return False
|
| 347 |
+
weights = (6, 5, 7, 2, 3, 4, 5, 6, 7)
|
| 348 |
+
rem = sum(w * int(x) for w, x in zip(weights, d[:-1])) % 11
|
| 349 |
+
return rem != 10 and rem == int(d[-1])
|
| 350 |
+
|
| 351 |
+
|
| 352 |
+
def _regon_ok(d: str) -> bool:
|
| 353 |
+
if not d.isdigit() or len(d) not in (9, 14):
|
| 354 |
+
return False
|
| 355 |
+
weights = ((8, 9, 2, 3, 4, 5, 6, 7) if len(d) == 9
|
| 356 |
+
else (2, 4, 8, 5, 0, 9, 7, 3, 6, 1, 2, 4, 8))
|
| 357 |
+
rem = sum(w * int(x) for w, x in zip(weights, d[:-1])) % 11
|
| 358 |
+
if rem == 10:
|
| 359 |
+
rem = 0
|
| 360 |
+
return rem == int(d[-1])
|
| 361 |
+
|
| 362 |
+
|
| 363 |
+
def _land_register_ok(court: str, number: str, check: str) -> bool:
|
| 364 |
+
code = court + number.zfill(8)
|
| 365 |
+
if not all(c in _LAND_REGISTER_VALUES for c in code):
|
| 366 |
+
return False
|
| 367 |
+
weights = (1, 3, 7) * 4
|
| 368 |
+
return sum(w * _LAND_REGISTER_VALUES[c] for w, c in zip(weights, code)) % 10 == int(check)
|
| 369 |
+
|
| 370 |
+
|
| 371 |
+
def _iban_ok(raw: str) -> bool:
|
| 372 |
+
# Do not assemble an alphanumeric account from neighbouring prose words.
|
| 373 |
+
# Compact and ordinary four-character printed groups remain supported.
|
| 374 |
+
groups = re.split(r'[ \t-]+', raw)
|
| 375 |
+
if len(groups) > 1 and any(re.search(r'[A-Za-z]{5}', group) for group in groups):
|
| 376 |
+
return False
|
| 377 |
+
compact = re.sub(r"[\s-]+", "", raw).upper()
|
| 378 |
+
if compact.isdigit() and len(compact) == 26:
|
| 379 |
+
compact = "PL" + compact
|
| 380 |
+
if not re.fullmatch(r"[A-Z]{2}[0-9]{2}[A-Z0-9]+", compact):
|
| 381 |
+
return False
|
| 382 |
+
if len(compact) != _IBAN_LENGTHS.get(compact[:2]) or not 2 <= int(compact[2:4]) <= 98:
|
| 383 |
+
return False
|
| 384 |
+
if compact.startswith("PL") and not compact[2:].isdigit():
|
| 385 |
+
return False
|
| 386 |
+
rearranged = compact[4:] + compact[:4]
|
| 387 |
+
nums = "".join(str(ord(c) - 55) if c.isalpha() else c for c in rearranged)
|
| 388 |
+
return int(nums) % 97 == 1
|
| 389 |
+
|
| 390 |
+
|
| 391 |
+
def _tag(start, end, tag):
|
| 392 |
+
return tag
|
| 393 |
+
|
| 394 |
+
|
| 395 |
+
def _replace_epuap(text: str, *, mask=_tag) -> tuple[str, int]:
|
| 396 |
+
last_end, count = None, 0
|
| 397 |
+
|
| 398 |
+
def replace(m):
|
| 399 |
+
nonlocal last_end, count
|
| 400 |
+
labelled = _EPUAP_LABEL_RE.search(text, max(0, m.start()-64), m.start())
|
| 401 |
+
continued = last_end is not None and re.fullmatch(r'[ \t]*[;,][ \t]*|[ \t]+(?:lub|albo|i|oraz)[ \t]+', text[last_end:m.start()], re.I)
|
| 402 |
+
if not labelled and not continued:
|
| 403 |
+
return m[0]
|
| 404 |
+
last_end = m.end()
|
| 405 |
+
count += 1
|
| 406 |
+
return mask(m.start(), m.end(), PII_TAG)
|
| 407 |
+
|
| 408 |
+
out = _EPUAP_RE.sub(replace, text)
|
| 409 |
+
return out, count
|
| 410 |
+
|
| 411 |
+
|
| 412 |
+
def _replace_documents(text: str, *, mask=_tag) -> tuple[str, int]:
|
| 413 |
+
n = 0
|
| 414 |
+
|
| 415 |
+
def replace(m):
|
| 416 |
+
nonlocal n
|
| 417 |
+
number = re.sub(r"[ \t-]", "", m["number"]).upper()
|
| 418 |
+
passport = m["label"].lower().startswith("paszport")
|
| 419 |
+
if passport:
|
| 420 |
+
short_diplomatic = ('dyplomatyczn' in m['label'].lower()
|
| 421 |
+
and re.search(r'\b(?:nr|numer)\b', m['label'], re.I)
|
| 422 |
+
and re.fullmatch(r'\d{4,5}', number))
|
| 423 |
+
if not short_diplomatic and not re.fullmatch(r"[A-Z]{2}[0-9OIL]{7}|[A-Z][0-9]{6,9}|[0-9]{6,12}", number):
|
| 424 |
+
return m[0]
|
| 425 |
+
elif not re.fullmatch(r"[A-Z]{3}[0-9OIL]{6}", number):
|
| 426 |
+
return m[0]
|
| 427 |
+
n += 1
|
| 428 |
+
return m["label"] + mask(m.start('number'), m.end('number'), PII_TAG)
|
| 429 |
+
|
| 430 |
+
# Passport detection is label/format based; no MRZ check digit is present here.
|
| 431 |
+
return _DOCUMENT_RE.sub(replace, text), n
|
| 432 |
+
|
| 433 |
+
|
| 434 |
+
def _replace_land_registers(text: str, *, mask=_tag) -> tuple[str, int]:
|
| 435 |
+
n = 0
|
| 436 |
+
|
| 437 |
+
def replace(m):
|
| 438 |
+
nonlocal n
|
| 439 |
+
# A label covers shortened, spaced or mistyped forms, as for labelled
|
| 440 |
+
# PESEL/NIP. Unlabelled numbers need the full form and a valid check digit.
|
| 441 |
+
labelled = _LAND_REGISTER_LABEL_RE.search(text, max(0, m.start()-64), m.start())
|
| 442 |
+
full = re.fullmatch(r"[A-Z]{2}\d[A-Z]/\d{8}/\d", m[0])
|
| 443 |
+
if not labelled and not (full and _land_register_ok(m['court'], m['number'], m['check'])):
|
| 444 |
+
return m[0]
|
| 445 |
+
n += 1
|
| 446 |
+
return mask(m.start(), m.end(), PII_TAG)
|
| 447 |
+
|
| 448 |
+
return _LAND_REGISTER_RE.sub(replace, text), n
|
| 449 |
+
|
| 450 |
+
|
| 451 |
+
def _pl_national_ok(d: str) -> bool:
|
| 452 |
+
return len(d) == 9 and d.isdigit() and d[:2] in _PL_PREFIX
|
| 453 |
+
|
| 454 |
+
|
| 455 |
+
def _phone_ok(raw: str, short: bool = False) -> bool:
|
| 456 |
+
s = raw.strip()
|
| 457 |
+
if _VANITY_PHONE_RE.fullmatch(s):
|
| 458 |
+
return short
|
| 459 |
+
if s.startswith('/') or s.endswith('/'):
|
| 460 |
+
return False
|
| 461 |
+
s = re.sub(r'^\(\+', '+', s)
|
| 462 |
+
s = re.sub(r'^\(00\)[ \t]*', '00', s)
|
| 463 |
+
if re.fullmatch(r"\d{4}[-./]\d{2}[-./]\d{2}", s):
|
| 464 |
+
return False
|
| 465 |
+
if re.fullmatch(r'\(?(?:[01]?\d|2[0-3])[.:][0-5]\d[ \t]*[–—-][ \t]*(?:[01]?\d|2[0-3])[.:][0-5]\d\)?', s):
|
| 466 |
+
return False
|
| 467 |
+
if re.fullmatch(r"\d{2}-\d{3}", s): # postal code
|
| 468 |
+
return False
|
| 469 |
+
# Ministry / court file numbers: BPRM.4820.2.3.2020, LUB-OMK.601.1.2024.3.
|
| 470 |
+
# They never start with "+"; dotted international phones do: +218.21.555.0123.
|
| 471 |
+
if not s.startswith('+') and (
|
| 472 |
+
s.count(".") >= 3 or (s.count(".") >= 1 and re.search(r"(?<!\d)20\d{2}(?!\d)", s))):
|
| 473 |
+
return False
|
| 474 |
+
if re.search(r"(?<!\d)\d{1,2}[-./]\d{1,2}[-./](?:19|20)\d{2}(?!\d)", s):
|
| 475 |
+
return False
|
| 476 |
+
d = _digits(raw)
|
| 477 |
+
if short and re.fullmatch(r'[2-9]\d{2}-[2-9]\d{2}-\d{4}', s):
|
| 478 |
+
return True
|
| 479 |
+
if short and s.startswith('(0)'):
|
| 480 |
+
return 7 <= len(d) <= 12
|
| 481 |
+
# ponytail: label + shape + length, not a country numbering-plan validator.
|
| 482 |
+
if short and re.match(r'^\(0?\d{2,4}(?:-\d{1,2})?\)', s):
|
| 483 |
+
return 8 <= len(d) <= 15
|
| 484 |
+
# EUR-Lex puts country and area code in parentheses: "(32-2) 299 11 11" is +32 2 299 11 11.
|
| 485 |
+
s = re.sub(r'^\(([1-9]\d{1,2})-(\d{1,2})\)', r'+\1 \2', s)
|
| 486 |
+
if s.startswith(('+', '00')):
|
| 487 |
+
international = _digits(s.replace('(0)', ''))
|
| 488 |
+
if s.startswith('00'):
|
| 489 |
+
international = international[2:]
|
| 490 |
+
if not international.startswith('48'):
|
| 491 |
+
# ponytail: explicit prefix + length, not a global numbering-plan
|
| 492 |
+
# validator. Add country metadata if review finds false positives.
|
| 493 |
+
return (7 if '(0)' in s else 8) <= len(international) <= 15 and international[0] != '0'
|
| 494 |
+
if d.startswith("00"):
|
| 495 |
+
d = d[2:]
|
| 496 |
+
if d.startswith("48") and len(d) >= 11:
|
| 497 |
+
rest = d[2:]
|
| 498 |
+
if rest.startswith("0"):
|
| 499 |
+
rest = rest[1:]
|
| 500 |
+
return _pl_national_ok(rest)
|
| 501 |
+
if d.startswith("0") and len(d) == 11 and d[:2] in {"01", "02", "07"}:
|
| 502 |
+
return True
|
| 503 |
+
if d.startswith("0") and len(d) >= 10:
|
| 504 |
+
return _pl_national_ok(d.lstrip("0")) or (short and len(d) <= 12)
|
| 505 |
+
return (_pl_national_ok(d)
|
| 506 |
+
or (short and 5 <= len(d) <= 8 and d[0] != '0')
|
| 507 |
+
or (short and 9 <= len(d) <= 12 and bool(re.fullmatch(r'\d{2,4}(?:[ \t-]\d{2,4}){2,3}', s)))
|
| 508 |
+
or (short and 9 <= len(d) <= 12 and bool(re.fullmatch(r'\d{2,4}\.\d{6,8}', s)))
|
| 509 |
+
or (short and 9 <= len(d) <= 12 and d.startswith('0')))
|
| 510 |
+
|
| 511 |
+
|
| 512 |
+
def _replace_checked(text: str, pattern: re.Pattern, tag: str, ok,
|
| 513 |
+
label: re.Pattern | None = None, *, mask=_tag) -> tuple[str, int]:
|
| 514 |
+
n = 0
|
| 515 |
+
|
| 516 |
+
def _sub(m):
|
| 517 |
+
nonlocal n
|
| 518 |
+
if label is not None and not label.search(text, max(0, m.start() - 64), m.start()):
|
| 519 |
+
return m.group(0)
|
| 520 |
+
if pattern is _PESEL_RE and re.search(r'[/?&][^\s]*\Z', text[max(0, m.start()-500):m.start()]):
|
| 521 |
+
return m.group(0) # Bare URL path/query digits are not personal identifiers.
|
| 522 |
+
if not ok(m.group(0)) or (pattern not in (_EMAIL_RE, _EMAIL_WRAP_RE, _EMAIL_FRAGMENT_RE)
|
| 523 |
+
and _is_amount(text, m.start(), m.end())):
|
| 524 |
+
return m.group(0)
|
| 525 |
+
n += 1
|
| 526 |
+
return mask(m.start(), m.end(), tag)
|
| 527 |
+
|
| 528 |
+
return pattern.sub(_sub, text), n
|
| 529 |
+
|
| 530 |
+
|
| 531 |
+
def _replace_phones(text: str, *, mask=_tag) -> tuple[str, int]:
|
| 532 |
+
n = 0
|
| 533 |
+
out = []
|
| 534 |
+
pos = 0
|
| 535 |
+
last_phone_end = None
|
| 536 |
+
last_phone_complete = False
|
| 537 |
+
while True:
|
| 538 |
+
m = _PHONE_RE.search(text, pos)
|
| 539 |
+
if not m:
|
| 540 |
+
out.append(text[pos:])
|
| 541 |
+
break
|
| 542 |
+
raw = m[0]
|
| 543 |
+
end = m.end()
|
| 544 |
+
replacement = raw
|
| 545 |
+
line_end = text.find('\n', m.start(), end)
|
| 546 |
+
if line_end >= 0 and re.fullmatch(r'\d{1,3}', text[m.start():line_end]):
|
| 547 |
+
next_end = text.find('\n', line_end+1, end)
|
| 548 |
+
if _directory_context(text, line_end+1, next_end if next_end >= 0 else end):
|
| 549 |
+
# A flattened table's room cell is not a phone country prefix.
|
| 550 |
+
out.append(text[pos:line_end+1])
|
| 551 |
+
pos = line_end+1
|
| 552 |
+
continue
|
| 553 |
+
short = bool(_SHORT_PHONE_LABEL_RE.search(text[max(0, m.start()-100):m.start()]))
|
| 554 |
+
short = short or (re.fullmatch(r'116[ \t]?\d{3}', raw)
|
| 555 |
+
and _phone_context(text, m.start(), raw))
|
| 556 |
+
short = short or _phone_list_context(text, m.start())
|
| 557 |
+
# A bare five-digit continuation followed by a place/name can be a
|
| 558 |
+
# postal address. It needs its own phone label to override that ambiguity.
|
| 559 |
+
postal = re.fullmatch(r'\d{5}', raw) and re.match(r'[ \t]+[^\W\d_]', text[end:])
|
| 560 |
+
# So is a 5-6 digit number after a complete phone and a bare line break: a table
|
| 561 |
+
# cell, "Fax (32-2) 287 25 24\n28 648,00". After a comma a list goes on,
|
| 562 |
+
# "95-1-511098,\n514262"; after a fragment the next line is its wrapped rest,
|
| 563 |
+
# "tel. +48 609 \n953 709", or the next local number, "(30) 25 10 22 33 25,\n22 33 28".
|
| 564 |
+
cell = (last_phone_complete and 5 <= len(_digits(raw)) <= 6
|
| 565 |
+
and re.fullmatch(r'[ \t]*\n[ \t\n]*', text[last_phone_end:m.start()]))
|
| 566 |
+
if (last_phone_end is not None and not postal and not cell
|
| 567 |
+
and re.fullmatch(r"[ \t\n,;]{1,12}|[ \t]+(?:lub|albo)[ \t]+", text[last_phone_end:m.start()], re.I)):
|
| 568 |
+
short = True
|
| 569 |
+
if (last_phone_end is not None
|
| 570 |
+
and re.fullmatch(r'(?:[ \t]+(?:w sprawie\b|\((?:w godzinach|dostępny)\b)'
|
| 571 |
+
r'|[ \t]*[–—-][ \t]+)[^\d\n.!?]{1,200}\n[ \t\n]{0,8}',
|
| 572 |
+
text[last_phone_end:m.start()], re.I)
|
| 573 |
+
and _pl_national_ok(_digits(raw))):
|
| 574 |
+
short = True
|
| 575 |
+
if (short or _phone_context(text, m.start(), raw)
|
| 576 |
+
or _directory_context(text, m.start(), line_end if line_end >= 0 else end)):
|
| 577 |
+
if (_is_amount(text, m.start(), end)
|
| 578 |
+
and re.fullmatch(r'\d{1,3}(?:\.[ \t]*\d{3})+', raw)):
|
| 579 |
+
out.extend((text[pos:m.start()], raw))
|
| 580 |
+
pos = end
|
| 581 |
+
continue
|
| 582 |
+
# Stop before a new line or opening hours, but only after a complete
|
| 583 |
+
# phone. Look past the greedy match's last digit to recognize HH:MM.
|
| 584 |
+
for gap in _PHONE_BREAK_RE.finditer(text, m.start(), end + 6):
|
| 585 |
+
if gap.start() >= end:
|
| 586 |
+
break
|
| 587 |
+
# Foreign lengths vary: a valid-looking prefix may still be an
|
| 588 |
+
# incomplete wrapped number. Only a section marker ends it.
|
| 589 |
+
if (raw.startswith(('+', '00')) and not re.match(r'(?:\+|00)[ \t]*48', raw)
|
| 590 |
+
and gap[0] == '\n'
|
| 591 |
+
and not _SECTION_MARKER_RE.match(text, gap.end())):
|
| 592 |
+
continue
|
| 593 |
+
prefix = text[m.start():gap.start()].rstrip(". \t")
|
| 594 |
+
# A short number also ends at a section marker: "albo 1234567\n2. Szkoła".
|
| 595 |
+
section = gap[0] == '\n' and bool(_SECTION_MARKER_RE.match(text, gap.end()))
|
| 596 |
+
if _phone_ok(prefix, short and section) and not _is_amount(text, m.start(), gap.start()):
|
| 597 |
+
end = m.start() + len(prefix)
|
| 598 |
+
raw = prefix
|
| 599 |
+
replacement = raw
|
| 600 |
+
break
|
| 601 |
+
if _is_amount(text, m.start(), end):
|
| 602 |
+
out.extend((text[pos:m.start()], raw))
|
| 603 |
+
pos = end
|
| 604 |
+
continue
|
| 605 |
+
# A suffix range repeats the final extension digits, not a second
|
| 606 |
+
# complete phone. Only accept it after a valid full base number.
|
| 607 |
+
extension = re.fullmatch(r"(.+?)[/-](\d{1,3})", raw)
|
| 608 |
+
parenthesized_extension = re.fullmatch(r'(.+?)[ \t]+\(\d{1,5}', raw)
|
| 609 |
+
if not (text[end:end+1] == ')' and parenthesized_extension
|
| 610 |
+
and _phone_ok(parenthesized_extension[1])):
|
| 611 |
+
parenthesized_extension = None
|
| 612 |
+
pair = re.fullmatch(r"(.+?)[ \t]+[–—-][ \t]+(.+)", raw)
|
| 613 |
+
shared = re.match(r"\(\d{2,4}\)[ \t]*", pair[1]) if pair else None
|
| 614 |
+
full_range = (pair and _phone_ok(pair[1], short)
|
| 615 |
+
and _phone_ok((shared[0] if shared else '') + pair[2], short))
|
| 616 |
+
if (_phone_ok(raw, short) or (extension and _phone_ok(extension[1]))
|
| 617 |
+
or parenthesized_extension or full_range):
|
| 618 |
+
if raw.count('(') > raw.count(')') and text[end:end+1] == ')':
|
| 619 |
+
end += 1
|
| 620 |
+
suffix = _PHONE_EXTENSION_RE.match(text, end)
|
| 621 |
+
if suffix and not _is_amount(text, suffix.start(), suffix.end()):
|
| 622 |
+
end = suffix.end()
|
| 623 |
+
start = m.start()
|
| 624 |
+
# Parentheses enclosing prose are punctuation, not phone syntax.
|
| 625 |
+
if raw.startswith('(+') and raw.count('(') > raw.count(')') and text[end-1:end] != ')':
|
| 626 |
+
start += 1
|
| 627 |
+
replacement = text[m.start():start] + mask(start, end, PHONE_TAG)
|
| 628 |
+
n += 1
|
| 629 |
+
# ponytail: split only pairs (<=22 digits); longer lists need label
|
| 630 |
+
# context to distinguish them from accounts. Never redact ID tails.
|
| 631 |
+
elif 18 <= len(_digits(raw)) <= 22 and "." not in raw:
|
| 632 |
+
# A slash separates alternatives, the second without the shared
|
| 633 |
+
# prefix: "(31-30) 274 44 13/274 44 01". Try slashes first.
|
| 634 |
+
gaps = sorted(re.finditer(r"[ \t]*/[ \t]*|[ \t\n]+", raw), key=lambda g: '/' not in g[0])
|
| 635 |
+
for gap in gaps:
|
| 636 |
+
if _phone_ok(raw[:gap.start()]) and _phone_ok(raw[gap.end():], '/' in gap[0]):
|
| 637 |
+
replacement = (mask(m.start(), m.start()+gap.start(), PHONE_TAG)
|
| 638 |
+
+ gap[0] + mask(m.start()+gap.end(), end, PHONE_TAG))
|
| 639 |
+
n += 2
|
| 640 |
+
break
|
| 641 |
+
out.extend((text[pos:m.start()], replacement))
|
| 642 |
+
if replacement != raw:
|
| 643 |
+
last_phone_end = end
|
| 644 |
+
last_phone_complete = _phone_ok(raw)
|
| 645 |
+
pos = end
|
| 646 |
+
return "".join(out), n
|
| 647 |
+
|
| 648 |
+
|
| 649 |
+
def _replace_extensions(text: str, *, mask=_tag) -> tuple[str, int]:
|
| 650 |
+
count = 0
|
| 651 |
+
|
| 652 |
+
def replace(m):
|
| 653 |
+
nonlocal count
|
| 654 |
+
before = text[max(0, m.start()-500):m.start()]
|
| 655 |
+
# Only a telephone followed by extension/name entries establishes scope.
|
| 656 |
+
heading = re.search(r'\btel(?:efon)?[.: \t]*\[Telefon\]'
|
| 657 |
+
r'(?P<entries>(?:\s*wewn?\.?[ \t]+\d{1,5}[ \t]+[^\d\n]+)*\s*)\Z', before, re.I)
|
| 658 |
+
if not heading or _is_amount(text, m.start('number'), m.end('number')):
|
| 659 |
+
return m[0]
|
| 660 |
+
count += 1
|
| 661 |
+
return m['label'] + mask(m.start('number'), m.end('number'), PHONE_TAG)
|
| 662 |
+
|
| 663 |
+
output = re.sub(r'(?P<label>(?:^|\n)[ \t]*wewn?\.?[ \t]+)(?P<number>\d{1,5})(?!\w)',
|
| 664 |
+
replace, text, flags=re.I)
|
| 665 |
+
return output, count
|
| 666 |
+
|
| 667 |
+
|
| 668 |
+
def scrub_pii(text: str, *, spans: list | None = None) -> tuple[str, dict[str, int]]:
|
| 669 |
+
"""Return (scrubbed_text, counts); optionally append exact original spans.
|
| 670 |
+
|
| 671 |
+
Offset tracking is opt-in; corpus ingestion keeps its existing return type
|
| 672 |
+
and does not allocate a character map. Unicode offsets count code points.
|
| 673 |
+
"""
|
| 674 |
+
counts = {k: 0 for k in COUNTS}
|
| 675 |
+
if not text:
|
| 676 |
+
return text, counts
|
| 677 |
+
original = text
|
| 678 |
+
text = text.translate(_SPACES) # One character per character: offsets survive.
|
| 679 |
+
offsets = list(range(len(text))) if spans is not None else None
|
| 680 |
+
edits = []
|
| 681 |
+
|
| 682 |
+
def mask(start, end, tag):
|
| 683 |
+
if spans is not None:
|
| 684 |
+
a, b = offsets[start], offsets[end-1] + 1
|
| 685 |
+
spans.append(dict(start=a, end=b, text=original[a:b],
|
| 686 |
+
label='phone' if tag == PHONE_TAG else 'pii'))
|
| 687 |
+
edits.append((start, end, tag))
|
| 688 |
+
return tag
|
| 689 |
+
|
| 690 |
+
def apply(replacer, *args):
|
| 691 |
+
nonlocal text
|
| 692 |
+
text, result = replacer(text, *args, mask=mask)
|
| 693 |
+
# Each replacement pass uses one coordinate system. Apply its edits
|
| 694 |
+
# backwards only after all regex callbacks have finished.
|
| 695 |
+
for start, end, tag in reversed(edits):
|
| 696 |
+
offsets[start:end] = [offsets[start]] * len(tag)
|
| 697 |
+
edits.clear()
|
| 698 |
+
return result
|
| 699 |
+
|
| 700 |
+
# Longest / most specific first so a 26-digit account is not sliced
|
| 701 |
+
# into REGON / PESEL / NIP / phone. Checksums live in the replace callback.
|
| 702 |
+
def pins(value, *, mask):
|
| 703 |
+
def replace(m):
|
| 704 |
+
if _is_amount(value, m.start('number'), m.end('number')):
|
| 705 |
+
return m[0]
|
| 706 |
+
counts['pin'] += 1
|
| 707 |
+
return m['label'] + mask(m.start('number'), m.end('number'), PII_TAG)
|
| 708 |
+
return _PIN_RE.sub(replace, value), None
|
| 709 |
+
apply(pins)
|
| 710 |
+
foreign = apply(_replace_checked, _IBAN_RE, PII_TAG, _iban_ok)
|
| 711 |
+
bare = apply(_replace_checked, _ACCOUNT_RE, PII_TAG, _iban_ok)
|
| 712 |
+
counts["account"] = foreign + bare
|
| 713 |
+
counts['electronic_address'] = apply(_replace_checked, _EDELIVERY_RE, PII_TAG, lambda _: True)
|
| 714 |
+
counts['electronic_address'] += apply(_replace_epuap)
|
| 715 |
+
counts['krs'] = apply(_replace_checked, _KRS_RE, PII_TAG, lambda _: True, _KRS_LABEL_RE)
|
| 716 |
+
counts["document"] = apply(_replace_documents)
|
| 717 |
+
counts["land_register"] = apply(_replace_land_registers)
|
| 718 |
+
def labelled_ids(value, *, mask):
|
| 719 |
+
def replace(m):
|
| 720 |
+
if len(_digits(m['number'])) < 7 or _is_amount(value, m.start('number'), m.end()):
|
| 721 |
+
return m[0]
|
| 722 |
+
counts[m['kind'].lower()] += 1
|
| 723 |
+
return m['label'] + mask(m.start('number'), m.end('number'), PII_TAG)
|
| 724 |
+
return _LABELLED_ID_RE.sub(replace, value), None
|
| 725 |
+
apply(labelled_ids)
|
| 726 |
+
n14 = apply(_replace_checked, _REGON14_RE, PII_TAG, _regon_ok, _REGON_LABEL_RE)
|
| 727 |
+
n9 = apply(_replace_checked, _REGON9_RE, PII_TAG, _regon_ok, _REGON_LABEL_RE)
|
| 728 |
+
counts["regon"] += n14 + n9
|
| 729 |
+
n_dash = apply(_replace_checked, _NIP_DASH_RE, PII_TAG, lambda s: _nip_ok(_digits(s)), _NIP_LABEL_RE)
|
| 730 |
+
n_plain = apply(_replace_checked, _NIP_RE, PII_TAG, lambda s: _nip_ok(_digits(s)), _NIP_LABEL_RE)
|
| 731 |
+
counts["nip"] += n_dash + n_plain
|
| 732 |
+
# The domain break is the last newline; a hyphenated local part may add an earlier one.
|
| 733 |
+
counts["email"] = apply(_replace_checked, _EMAIL_WRAP_RE, PII_TAG,
|
| 734 |
+
lambda s: not _EMAIL_RE.fullmatch(s[:s.rindex('\n')].rstrip('. \t\r')))
|
| 735 |
+
counts["email"] += apply(_replace_checked, _EMAIL_RE, PII_TAG, lambda _: True)
|
| 736 |
+
counts["email"] += apply(_replace_checked, _EMAIL_FRAGMENT_RE, PII_TAG,
|
| 737 |
+
lambda _: True, _EMAIL_LABEL_RE)
|
| 738 |
+
counts["phone"] = apply(_replace_phones)
|
| 739 |
+
counts["phone"] += apply(_replace_checked, _SERVICE_PHONE_RE, PHONE_TAG,
|
| 740 |
+
lambda _: True, _SERVICE_PHONE_LABEL_RE)
|
| 741 |
+
counts['phone'] += apply(_replace_extensions)
|
| 742 |
+
# Contact context takes precedence over coincidental PESEL checksums in
|
| 743 |
+
# foreign phone numbers; explicitly labelled identifiers were handled first.
|
| 744 |
+
bare_pesel = apply(_replace_checked, _PESEL_RE, PII_TAG, _pesel_ok)
|
| 745 |
+
counts['pesel'] += bare_pesel
|
| 746 |
+
return text, counts
|