Spaces:
Paused
Paused
File size: 4,302 Bytes
b06be80 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 | import re
from typing import Dict, List
# Common abbreviations in Indian addresses
ABBREVIATIONS = {
'po': 'post office',
'bo': 'branch office',
'so': 'sub office',
'ho': 'head office',
'dist': 'district',
'nr': 'near',
'rd': 'road',
'st': 'street',
'ave': 'avenue',
'blvd': 'boulevard',
'apt': 'apartment',
'bldg': 'building',
'flr': 'floor',
'dept': 'department',
'no': 'number',
'nagar': 'nagar',
'gali': 'gali',
'marg': 'marg',
'colony': 'colony',
'sector': 'sector',
'phase': 'phase',
'block': 'block',
'h/o': 'house of',
's/o': 'son of',
'd/o': 'daughter of',
'w/o': 'wife of',
'c/o': 'care of',
'tel': 'telangana',
'tg': 'telangana',
'ap': 'andhra pradesh',
'up': 'uttar pradesh',
'hp': 'himachal pradesh',
'mp': 'madhya pradesh',
'tn': 'tamil nadu',
'wb': 'west bengal',
'ka': 'karnataka',
'mh': 'maharashtra',
'dl': 'delhi',
'rj': 'rajasthan',
'pb': 'punjab',
'hr': 'haryana',
'jk': 'jammu kashmir',
'gj': 'gujarat',
'or': 'odisha',
'br': 'bihar',
'jh': 'jharkhand',
'as': 'assam',
'uk': 'uttarakhand',
}
def normalize_text(text: str) -> str:
if not text:
return ""
# Convert to lowercase
text = text.lower()
# Remove special characters but keep spaces and alphanumeric
text = re.sub(r'[^a-z0-9\s]', ' ', text)
# Remove extra whitespace
text = re.sub(r'\s+', ' ', text)
# Strip leading/trailing whitespace
text = text.strip()
return text
def expand_abbreviations(text: str) -> str:
words = text.split()
expanded_words = []
for word in words:
# Check if word is an abbreviation
if word.lower() in ABBREVIATIONS:
expanded_words.append(ABBREVIATIONS[word.lower()])
else:
expanded_words.append(word)
return ' '.join(expanded_words)
def extract_pincode(text: str) -> str:
# Look for 6-digit numbers
match = re.search(r'\b\d{6}\b', text)
if match:
return match.group(0)
return ""
def remove_personal_info(text: str) -> str:
# Remove mobile numbers (10 digits)
text = re.sub(r'\b\d{10}\b', '', text)
# Remove email addresses
text = re.sub(r'\S+@\S+', '', text)
# Remove patterns like "S/O", "D/O", "W/O" followed by names
text = re.sub(r'\b[sdw]/o\s+\w+\s+\w+', '', text, flags=re.IGNORECASE)
# Remove "Mr.", "Mrs.", "Ms.", "Dr." titles
text = re.sub(r'\b(mr|mrs|ms|dr|shri|smt)\.?\s+', '', text, flags=re.IGNORECASE)
return text
def clean_address(text: str) -> str:
if not text:
return ""
# Step 1: Remove personal info
text = remove_personal_info(text)
# Step 2: Normalize
text = normalize_text(text)
# Step 3: Expand abbreviations
text = expand_abbreviations(text)
# Step 4: Final normalization
text = normalize_text(text)
return text
def extract_address_components(text: str) -> Dict[str, str]:
components = {
'pincode': '',
'city': '',
'state': '',
'district': '',
'raw_text': text
}
# Extract PIN code
components['pincode'] = extract_pincode(text)
# Extract state (if matches known abbreviations)
text_lower = text.lower()
for abbr, full_name in ABBREVIATIONS.items():
if len(abbr) == 2 and abbr in text_lower.split():
components['state'] = full_name
break
return components
def calculate_text_similarity(text1: str, text2: str) -> float:
# Normalize both texts
text1 = normalize_text(text1)
text2 = normalize_text(text2)
# Get word sets
set1 = set(text1.split())
set2 = set(text2.split())
# Calculate Jaccard similarity
if not set1 or not set2:
return 0.0
intersection = len(set1 & set2)
union = len(set1 | set2)
return intersection / union if union > 0 else 0.0
def highlight_matching_tokens(query: str, target: str) -> List[str]:
query_tokens = set(normalize_text(query).split())
target_tokens = set(normalize_text(target).split())
return list(query_tokens & target_tokens)
|