File size: 5,292 Bytes
3b3c03e
 
b72d02d
3b3c03e
b72d02d
 
 
 
 
3b3c03e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
import spacy
import re
import subprocess

try:
    nlp = spacy.load("en_core_web_sm")
except OSError:
    subprocess.run(["python", "-m", "spacy", "download", "en_core_web_sm"], check=True)
    nlp = spacy.load("en_core_web_sm")


FULL_NAME_PATTERN = r'My name is ([A-Za-z\s\.\-]+)[\.,]|Name\s*:\s*([A-Za-z\s\.\-]+)[\.,]'
EMAIL_PATTERN = r'You can reach me at ([A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,})|Email\s*:\s*([A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,})'
PHONE_PATTERN = r'My [Cc]ontact number is\s*([+\d\s\-\(\)\.]+)|Phone\s*:\s*([+\d\s\-\(\)\.]+)|(\+?\d{1,4}?[-.\s]?\(?\d{1,3}?\)?[-.\s]?\d{1,4}[-.\s]?\d{1,4}[-.\s]?\d{1,9})'
DOB_PATTERN = r'[Dd]ate of [Bb]irth\s*:?\s*(\d{1,2}[-/\.]\d{1,2}[-/\.]\d{2,4}|\d{2,4}[-/\.]\d{1,2}[-/\.]\d{1,2})'
AADHAR_PATTERN = r'[Aa]adhar(?:\s*[Cc]ard)?\s*(?:[Nn]umber)?\s*:?\s*(\d{4}\s*\d{4}\s*\d{4}|\d{12})'
CREDIT_DEBIT_PATTERN = r'[Cc](?:redit|ard)\s*(?:[Nn]umber)?\s*:?\s*(\d{4}[-\s]?\d{4}[-\s]?\d{4}[-\s]?\d{4}|\d{16})'
CVV_PATTERN = r'[Cc][Vv][Vv]\s*(?:[Nn]umber)?\s*:?\s*(\d{3,4})'
EXPIRY_PATTERN = r'[Ee]xpir(?:y|ation)\s*[Dd]ate\s*:?\s*(\d{1,2}[-/\.]\d{2,4}|\d{2}[-/\.]\d{2})'

def preprocess_text(text):
    """Clean and normalize text before processing."""
    text = re.sub(r'\n{2,}', '\n', text)
    text = text.replace('\n', ' ')
    text = re.sub(r'\s+', ' ', text)
    text = text.replace('\xa0', ' ')
    text = text.replace('"', '')

    return text.strip()

def mask_pii(email_text):
    """Mask PII in emails using regex patterns with the required entity fields."""
    if not email_text or len(email_text.strip()) == 0:
        return ""

    masked_text = preprocess_text(email_text)
    pii_entities = {}
    name_matches = re.finditer(FULL_NAME_PATTERN, masked_text)
    for match in name_matches:
        full_match = match.group(0)
        name = next((g for g in match.groups() if g), "")
        if name:
            pii_entities["full_name"] = name
            if "My name is" in full_match:
                masked_text = masked_text.replace(full_match, "My name is <full_name>.")
            elif "Name:" in full_match:
                masked_text = masked_text.replace(full_match, "Name: <full_name>.")
            else:
                masked_text = masked_text.replace(name, "<full_name>")

    email_matches = re.finditer(EMAIL_PATTERN, masked_text)
    for match in email_matches:
        full_match = match.group(0)
        email = next((g for g in match.groups() if g), "")
        if email:
            pii_entities["email"] = email
            if "You can reach me at" in full_match:
                masked_text = masked_text.replace(full_match, "You can reach me at <email>")
            elif "Email" in full_match:
                masked_text = masked_text.replace(full_match, "Email: <email>")
            else:
                masked_text = masked_text.replace(email, "<email>")

    phone_matches = re.finditer(PHONE_PATTERN, masked_text)
    for match in phone_matches:
        full_match = match.group(0)
        phone = next((g for g in match.groups() if g), "")
        if phone:
            pii_entities["phone_number"] = phone
            if "My contact number is" in full_match or "My Contact number is" in full_match:
                masked_text = masked_text.replace(full_match, "My contact number is <phone_number>")
            elif "Phone" in full_match:
                masked_text = masked_text.replace(full_match, "Phone: <phone_number>")
            else:
                masked_text = masked_text.replace(phone, "<phone_number>")

    dob_matches = re.finditer(DOB_PATTERN, masked_text)
    for match in dob_matches:
        full_match = match.group(0)
        dob = match.group(1)
        if dob:
            pii_entities["dob"] = dob
            masked_text = masked_text.replace(full_match, "Date of Birth: <dob>")

    aadhar_matches = re.finditer(AADHAR_PATTERN, masked_text)
    for match in aadhar_matches:
        full_match = match.group(0)
        aadhar = match.group(1)
        if aadhar:
            pii_entities["aadhar_num"] = aadhar
            masked_text = masked_text.replace(full_match, "Aadhar Number: <aadhar_num>")

    card_matches = re.finditer(CREDIT_DEBIT_PATTERN, masked_text)
    for match in card_matches:
        full_match = match.group(0)
        card = match.group(1)
        if card:
            pii_entities["credit_debit_no"] = card
            masked_text = masked_text.replace(full_match, "Card Number: <credit_debit_no>")

    cvv_matches = re.finditer(CVV_PATTERN, masked_text)
    for match in cvv_matches:
        full_match = match.group(0)
        cvv = match.group(1)
        if cvv:
            pii_entities["cvv_no"] = cvv
            masked_text = masked_text.replace(full_match, "CVV: <cvv_no>")

    expiry_matches = re.finditer(EXPIRY_PATTERN, masked_text)
    for match in expiry_matches:
        full_match = match.group(0)
        expiry = match.group(1)
        if expiry:
            pii_entities["expiry_no"] = expiry
            masked_text = masked_text.replace(full_match, "Expiry Date: <expiry_no>")

    doc = nlp(masked_text)
    for ent in doc.ents:
        if ent.label_ == "PERSON" and "full_name" not in pii_entities:
            masked_text = masked_text.replace(ent.text, "<full_name>")

    return masked_text