Download ner_extractor.py from NabilHzs/payparse: direct link, hf CLI and curl.
- Browser
- Download file 15.7 kB
-
https://huggingface.co/spaces/NabilHzs/payparse/resolve/main/ner_extractor.py
- Command line
-
hf download hf://spaces/NabilHzs/payparse/ner_extractor.py
-
curl -L -o ner_extractor.py https://huggingface.co/spaces/NabilHzs/payparse/resolve/main/ner_extractor.py
15.7 kB
| """ | |
| ner_extractor.py | |
| ---------------- | |
| Rule-based Named Entity Recognition (NER) layer for Indonesian financial | |
| utterances. Runs BEFORE the LLM call as a "pre-scan" to: | |
| 1. Extract entities the LLM might miss (robust regex + slang dictionary). | |
| 2. Validate LLM output against NER findings (cross-check). | |
| 3. Serve as fallback when the LLM is rate-limited. | |
| This is NOT a replacement for the LLM — it's a safety net that catches | |
| high-confidence entities (amounts, phone numbers, PLN IDs) that have | |
| deterministic patterns, leaving ambiguous entities (contact names, | |
| intent) to the LLM. | |
| Extraction capabilities: | |
| - Amounts: slang (goceng, ceban, gocap, cepek, seceng, sejuta) + numeric | |
| (50rb, 100.000, 2jt, 75 ribu). | |
| - Phone numbers: 08xxxxxxxxxx, +62xxxxxxxxxx, 628xxxxxxxxxx. | |
| - PLN customer IDs: 8-12 digit sequences. | |
| - Contact names: pattern "ke/buat/untuk [name]" with honorific stripping. | |
| - Intent keywords: transfer/pulsa/listrik with typo tolerance. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from dataclasses import dataclass, field | |
| from typing import Optional | |
| from schema import IntentType, TransactionEntities | |
| # --------------------------------------------------------------------------- | |
| # Indonesian financial slang → integer amount | |
| # --------------------------------------------------------------------------- | |
| SLANG_AMOUNTS: dict[str, int] = { | |
| "goceng": 5_000, | |
| "ceban": 10_000, | |
| "gocap": 50_000, | |
| "cepek": 100_000, | |
| "seceng": 1_000, | |
| "sejuta": 1_000_000, | |
| "sejutaan": 1_000_000, | |
| "gocengan": 5_000, | |
| "cebuan": 10_000, | |
| "gocapan": 50_000, | |
| "cepekan": 100_000, | |
| "secengan": 1_000, | |
| } | |
| # Numeric abbreviations: "50rb", "100ribu", "2jt", "75 k" | |
| NUMERIC_ABBREV = { | |
| "rb": 1_000, | |
| "ribu": 1_000, | |
| "k": 1_000, | |
| "jt": 1_000_000, | |
| "juta": 1_000_000, | |
| "jutaan": 1_000_000, | |
| } | |
| # Intent keywords with common typos | |
| INTENT_KEYWORDS: dict[IntentType, list[str]] = { | |
| IntentType.TRANSFER_UANG: [ | |
| "transfer", "trasnfer", "tf", "kirim", "kirimin", "kirimin", | |
| "transferin", "ngirim", "ngirimin", "send", "kirim uang", | |
| ], | |
| IntentType.BELI_PULSA: [ | |
| "pulsa", "pusla", "pls", "isi pulsa", "isiin pulsa", "beli pulsa", | |
| "beliin pulsa", "isiin", "top up pulsa", "isipulsa", | |
| ], | |
| IntentType.BAYAR_PLN: [ | |
| "listrik", "pln", "bayar listrik", "tagihan listrik", "listrik id", | |
| "bayar pln", "token listrik", "tagihan pln", "tagihan listrik saya", | |
| "bayar tagihan pln", "bayar tagihan listrik", "listrik saya", | |
| "bayar tagihan pln saya", | |
| ], | |
| IntentType.PESAN_GOJEK: [ | |
| "gojek", "go jek", "pesan gojek", "order gojek", "booking gojek", | |
| "goride", "go ride", "naik gojek", "call gojek", | |
| ], | |
| IntentType.PESAN_GOFOOD: [ | |
| "gofood", "go food", "pesan gofood", "order gofood", "booking gofood", | |
| "beli makanan", "pesan makan", "beli makan", | |
| ], | |
| } | |
| # Honorifics to strip from contact names | |
| HONORIFICS = {"si", "bang", "mbak", "pak", "bu", "mas", "mbah", "kak", "ade", "adik"} | |
| # --------------------------------------------------------------------------- | |
| # NER result | |
| # --------------------------------------------------------------------------- | |
| class NERResult: | |
| """Entities extracted by the rule-based NER layer.""" | |
| intent: Optional[IntentType] = None | |
| amount: Optional[int] = None | |
| phone_number: Optional[str] = None | |
| recipient_phone: Optional[str] = None | |
| recipient: Optional[str] = None | |
| target_kontak: Optional[str] = None | |
| customer_id: Optional[str] = None | |
| provider: Optional[str] = None | |
| asal: Optional[str] = None | |
| tujuan: Optional[str] = None | |
| makanan: Optional[str] = None | |
| confidence: float = 0.0 | |
| """Which fields were extracted (for merge logic).""" | |
| extracted_fields: set[str] = field(default_factory=set) | |
| def to_entities(self) -> TransactionEntities: | |
| return TransactionEntities( | |
| recipient=self.recipient, | |
| recipient_phone=self.recipient_phone, | |
| amount=self.amount, | |
| phone_number=self.phone_number, | |
| target_kontak=self.target_kontak, | |
| customer_id=self.customer_id, | |
| provider=self.provider, | |
| asal=self.asal, | |
| tujuan=self.tujuan, | |
| makanan=self.makanan, | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # Extractor | |
| # --------------------------------------------------------------------------- | |
| class NERExtractor: | |
| """Rule-based NER for Indonesian financial utterances.""" | |
| def extract(self, text: str) -> NERResult: | |
| lowered = text.lower().strip() | |
| result = NERResult() | |
| # --- Intent classification (keyword + typo tolerant) --- | |
| result.intent = self._classify_intent(lowered) | |
| if result.intent is not None: | |
| result.extracted_fields.add("intent") | |
| result.confidence = 0.7 | |
| # --- Amount extraction --- | |
| amount = self._extract_amount(lowered) | |
| if amount is not None: | |
| result.amount = amount | |
| result.extracted_fields.add("amount") | |
| # --- Phone number extraction --- | |
| phone = self._extract_phone_number(lowered) | |
| if phone is not None: | |
| # Assign to the right field based on intent | |
| if result.intent == IntentType.TRANSFER_UANG: | |
| result.recipient_phone = phone | |
| result.extracted_fields.add("recipient_phone") | |
| else: | |
| result.phone_number = phone | |
| result.extracted_fields.add("phone_number") | |
| # --- PLN customer ID --- | |
| if result.intent == IntentType.BAYAR_PLN: | |
| cust_id = self._extract_customer_id(lowered) | |
| if cust_id is not None: | |
| result.customer_id = cust_id | |
| result.extracted_fields.add("customer_id") | |
| # --- Contact name / recipient --- | |
| if result.intent in (IntentType.TRANSFER_UANG, IntentType.BELI_PULSA): | |
| contact = self._extract_contact_name(lowered) | |
| if contact is not None: | |
| if result.intent == IntentType.TRANSFER_UANG: | |
| result.recipient = contact | |
| if result.recipient_phone is None: | |
| result.target_kontak = contact | |
| result.extracted_fields.add("target_kontak") | |
| result.extracted_fields.add("recipient") | |
| else: | |
| # beli_pulsa: only set target_kontak if no phone digits | |
| if result.phone_number is None: | |
| result.target_kontak = contact | |
| result.extracted_fields.add("target_kontak") | |
| # --- Provider (telco) --- | |
| provider = self._extract_provider(lowered) | |
| if provider is not None: | |
| result.provider = provider | |
| result.extracted_fields.add("provider") | |
| # --- Gojek: extract tujuan (destination) --- | |
| if result.intent == IntentType.PESAN_GOJEK: | |
| tujuan = self._extract_tujuan(lowered) | |
| if tujuan is not None: | |
| result.tujuan = tujuan | |
| result.extracted_fields.add("tujuan") | |
| # Asal defaults to Bogor; extract if "dari X" is mentioned | |
| asal = self._extract_asal(lowered) | |
| if asal is not None: | |
| result.asal = asal | |
| result.extracted_fields.add("asal") | |
| # --- GoFood: extract makanan (food item) --- | |
| if result.intent == IntentType.PESAN_GOFOOD: | |
| makanan = self._extract_makanan(lowered) | |
| if makanan is not None: | |
| result.makanan = makanan | |
| result.extracted_fields.add("makanan") | |
| return result | |
| # ------------------------------------------------------------------ | |
| # Intent classification | |
| # ------------------------------------------------------------------ | |
| def _classify_intent(lowered: str) -> Optional[IntentType]: | |
| # Check each intent's keywords (including typos) | |
| for intent, keywords in INTENT_KEYWORDS.items(): | |
| for kw in keywords: | |
| if kw in lowered: | |
| return intent | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Amount extraction | |
| # ------------------------------------------------------------------ | |
| def _extract_amount(lowered: str) -> Optional[int]: | |
| # 1. Slang amounts (highest priority) | |
| for slang, value in SLANG_AMOUNTS.items(): | |
| if slang in lowered: | |
| return value | |
| # 2. Numeric + abbreviation: "50rb", "100 ribu", "2jt", "75k" | |
| m = re.search(r"(\d+(?:[.,]\d+)?)\s*(rb|ribu|k|jt|juta|jutaan)\b", lowered) | |
| if m: | |
| base = float(m.group(1).replace(",", ".")) | |
| mult = NUMERIC_ABBREV.get(m.group(2), 1) | |
| return int(base * mult) | |
| # 3. Plain large number: "50000", "100000" (but not phone numbers) | |
| m = re.search(r"\b(\d{4,9})\b(?!\s*(?:rb|ribu|k|jt|juta))", lowered) | |
| if m and not m.group(1).startswith("08"): | |
| value = int(m.group(1)) | |
| if 500 <= value <= 100_000_000: | |
| return value | |
| # 4. "seratus ribu", "lima puluh ribu" (word-based, basic) | |
| word_amounts = { | |
| "seratus ribu": 100_000, | |
| "lima puluh ribu": 50_000, | |
| "sepuluh ribu": 10_000, | |
| "dua puluh ribu": 20_000, | |
| "tiga puluh ribu": 30_000, | |
| "empat puluh ribu": 40_000, | |
| "tujuh puluh ribu": 70_000, | |
| "delapan puluh ribu": 80_000, | |
| "sembilan puluh ribu": 90_000, | |
| "seribu": 1_000, | |
| "dua ribu": 2_000, | |
| "lima ribu": 5_000, | |
| } | |
| for phrase, value in word_amounts.items(): | |
| if phrase in lowered: | |
| return value | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Phone number extraction | |
| # ------------------------------------------------------------------ | |
| def _extract_phone_number(lowered: str) -> Optional[str]: | |
| # Match 08xxxxxxxxxx (9-13 digits), +62xxxxxxxxxx, 62xxxxxxxxxx | |
| patterns = [ | |
| r"\b08\d{8,12}\b", | |
| r"\+62\d{8,12}\b", | |
| r"\b62\d{8,12}\b", | |
| ] | |
| for pat in patterns: | |
| m = re.search(pat, lowered) | |
| if m: | |
| digits = re.sub(r"\D", "", m.group()) | |
| # Normalize +62 / 62 to 08 | |
| if digits.startswith("62"): | |
| digits = "0" + digits[2:] | |
| return digits | |
| return None | |
| # ------------------------------------------------------------------ | |
| # PLN customer ID | |
| # ------------------------------------------------------------------ | |
| def _extract_customer_id(lowered: str) -> Optional[str]: | |
| # PLN IDs are typically 8-12 digits, often starting with 4 or 5 | |
| m = re.search(r"\b(\d{8,12})\b", lowered) | |
| if m: | |
| return m.group(1) | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Contact name extraction | |
| # ------------------------------------------------------------------ | |
| def _extract_contact_name(lowered: str) -> Optional[str]: | |
| # Pronouns | |
| for pronoun in ("nomor ini", "nomer ini", "nomorku", "nomerku", "nomor saya"): | |
| if pronoun in lowered: | |
| return pronoun | |
| # "ke [name]", "buat [name]", "untuk [name]" with optional honorific | |
| m = re.search( | |
| r"\b(?:ke|buat|untuk)\s+(?:(?:si|bang|mbak|pak|bu|mas|mbah|kak|ade|adik)\s+)?([a-z]+)", | |
| lowered, | |
| ) | |
| if m: | |
| name = m.group(1) | |
| if name not in {"nomor", "nomer", "hp", "rekening", "pulsa", "aku", "saya", "ini"}: | |
| return name.capitalize() | |
| # "beliin [name] pulsa", "isiin [name] pulsa" | |
| m = re.search( | |
| r"\b(?:beliin|isiin|isi|beli)\s+([a-z]+)\s+pulsa", | |
| lowered, | |
| ) | |
| if m and m.group(1) not in {"pulsa", "nomor", "nomer"}: | |
| return m.group(1).capitalize() | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Telco provider | |
| # ------------------------------------------------------------------ | |
| def _extract_provider(lowered: str) -> Optional[str]: | |
| providers = { | |
| "telkomsel": "Telkomsel", | |
| "kartu as": "Telkomsel", | |
| "xl": "XL", | |
| "axis": "XL", | |
| "indosat": "Indosat", | |
| "im3": "Indosat", | |
| "mentari": "Indosat", | |
| "tri": "Tri", | |
| "smartfren": "Smartfren", | |
| } | |
| for key, value in providers.items(): | |
| if key in lowered: | |
| return value | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Gojek destination | |
| # ------------------------------------------------------------------ | |
| def _extract_tujuan(lowered: str) -> Optional[str]: | |
| # "gojek ke stasiun", "gojek ke bandara", "gojek ke mall botani" | |
| m = re.search(r"\bgojek\s+(?:ke|buat|untuk)\s+(.+?)(?:\s*$|\s*dari\s)", lowered) | |
| if m: | |
| dest = m.group(1).strip() | |
| if dest and dest not in {"dari", "ke", "buat"}: | |
| return dest.capitalize() | |
| # "pesan gojek ke X" | |
| m = re.search(r"\bpesan\s+gojek\s+(?:ke|buat|untuk)\s+(.+?)(?:\s*$|\s*dari\s)", lowered) | |
| if m: | |
| dest = m.group(1).strip() | |
| if dest: | |
| return dest.capitalize() | |
| # "gojek X" (without "ke") | |
| m = re.search(r"\bgojek\s+([a-z][a-z\s]+)", lowered) | |
| if m: | |
| dest = m.group(1).strip() | |
| # Exclude if it's just "ke" or intent keywords | |
| if dest and dest not in {"ke", "dari", "pesan", "order"}: | |
| return dest.capitalize() | |
| return None | |
| # ------------------------------------------------------------------ | |
| # Gojek origin | |
| # ------------------------------------------------------------------ | |
| def _extract_asal(lowered: str) -> Optional[str]: | |
| # "dari bogor", "dari stasiun" | |
| m = re.search(r"\bdari\s+([a-z][a-z\s]+?)(?:\s+ke\s|$)", lowered) | |
| if m: | |
| origin = m.group(1).strip() | |
| if origin: | |
| return origin.capitalize() | |
| return None | |
| # ------------------------------------------------------------------ | |
| # GoFood food item | |
| # ------------------------------------------------------------------ | |
| def _extract_makanan(lowered: str) -> Optional[str]: | |
| # "gofood nasi goreng", "pesan gofood ayam geprek" | |
| m = re.search(r"\bgofood\s+(.+?)(?:\s*$)", lowered) | |
| if m: | |
| food = m.group(1).strip() | |
| if food and food not in {"pesan", "order", "beli", "mau"}: | |
| return food.capitalize() | |
| m = re.search(r"\bpesan\s+gofood\s+(.+?)(?:\s*$)", lowered) | |
| if m: | |
| food = m.group(1).strip() | |
| if food: | |
| return food.capitalize() | |
| # "beli makan nasi goreng", "pesan makan ayam" | |
| m = re.search(r"\b(?:beli|pesan)\s+makan(?:an)?\s+(.+?)(?:\s*$)", lowered) | |
| if m: | |
| food = m.group(1).strip() | |
| if food: | |
| return food.capitalize() | |
| return None | |
| # Singleton instance | |
| ner_extractor = NERExtractor() | |