Instructions to use SlayerLab/NERGAL with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SlayerLab/NERGAL with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("token-classification", model="SlayerLab/NERGAL")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForTokenClassification tokenizer = AutoTokenizer.from_pretrained("SlayerLab/NERGAL") model = AutoModelForTokenClassification.from_pretrained("SlayerLab/NERGAL", device_map="auto") - Notebooks
- Google Colab
- Kaggle
2.0.0: [PHONE] replaces [Telefon]; [Telefon] stays a rules boundary
Browse files- hybrid.json +1 -1
- nergal.py +2 -2
- scrub_pii.py +9 -6
- test_nergal.py +24 -6
hybrid.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
| 6 |
"epoch": 5,
|
| 7 |
"seed": 202609160,
|
| 8 |
"threshold": 0.95,
|
| 9 |
-
"rules_sha256": "
|
| 10 |
"eval": {
|
| 11 |
"split": "841-dev",
|
| 12 |
"gold_amendments": [
|
|
|
|
| 6 |
"epoch": 5,
|
| 7 |
"seed": 202609160,
|
| 8 |
"threshold": 0.95,
|
| 9 |
+
"rules_sha256": "d18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357",
|
| 10 |
"eval": {
|
| 11 |
"split": "841-dev",
|
| 12 |
"gold_amendments": [
|
nergal.py
CHANGED
|
@@ -23,7 +23,7 @@ GAP_IDS = [250002, 250003]
|
|
| 23 |
BIO_LABELS = ['O', 'B-phone', 'I-phone', 'B-pii', 'I-pii']
|
| 24 |
LABELS = ['phone', 'pii']
|
| 25 |
THRESHOLD = 0.95
|
| 26 |
-
RULES_SHA = '
|
| 27 |
DTYPES = ('float32', 'float16')
|
| 28 |
|
| 29 |
|
|
@@ -168,7 +168,7 @@ class Encoding:
|
|
| 168 |
|
| 169 |
|
| 170 |
def spans_from(module, text):
|
| 171 |
-
"""Rule spans on the original text. Existing [PII]/[Telefon] placeholders do not switch the rules off."""
|
| 172 |
result = []
|
| 173 |
module.scrub_pii(text, spans=result)
|
| 174 |
return sorted(({k: s[k] for k in ('start', 'end', 'label')} | {'score': 1.0} for s in result),
|
|
|
|
| 23 |
BIO_LABELS = ['O', 'B-phone', 'I-phone', 'B-pii', 'I-pii']
|
| 24 |
LABELS = ['phone', 'pii']
|
| 25 |
THRESHOLD = 0.95
|
| 26 |
+
RULES_SHA = 'd18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357'
|
| 27 |
DTYPES = ('float32', 'float16')
|
| 28 |
|
| 29 |
|
|
|
|
| 168 |
|
| 169 |
|
| 170 |
def spans_from(module, text):
|
| 171 |
+
"""Rule spans on the original text. Existing [PHONE]/[PII]/[Telefon] placeholders do not switch the rules off."""
|
| 172 |
result = []
|
| 173 |
module.scrub_pii(text, spans=result)
|
| 174 |
return sorted(({k: s[k] for k in ('start', 'end', 'label')} | {'score': 1.0} for s in result),
|
scrub_pii.py
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
|
| 3 |
Replaces emails, phones, PESEL/NIP/REGON/KRS, land-register (KW) numbers, electronic contact addresses and
|
| 4 |
account numbers in place so sentence structure survives. Names of public officials are left untouched —
|
| 5 |
-
that is intentional, not a gap. Phones map to [
|
| 6 |
to [PII]. Bare PESEL requires checksum and date validation; explicit identifier
|
| 7 |
labels also redact damaged numbers (including common OCR O/I/l substitutions).
|
| 8 |
NIP/REGON/KRS and identity documents require explicit nearby labels. KW numbers
|
|
@@ -27,7 +27,7 @@ Phones require a nearby contact cue, a Polish +48/0048 prefix, or explicit
|
|
| 27 |
international country/trunk notation such as +CC (0). With a cue, the EUR-Lex
|
| 28 |
"(32-2) 299 11 11" country-area form counts as international. Strong labels also
|
| 29 |
admit one-digit country codes and wider hyphenated area codes. Phone policy v3: each
|
| 30 |
-
number of at least 7 digits (keypad letters count) is its own [
|
| 31 |
between numbers stay unmasked, and a shorter part (an extension, "/90") stays inside
|
| 32 |
its number's span. A phone match with no such number (a short service or emergency
|
| 33 |
number, a lone extension) is left as text; a line break alone never splits a
|
|
@@ -43,8 +43,11 @@ from __future__ import annotations
|
|
| 43 |
import datetime as dt
|
| 44 |
import re
|
| 45 |
|
| 46 |
-
PHONE_TAG = "[
|
| 47 |
PII_TAG = "[PII]"
|
|
|
|
|
|
|
|
|
|
| 48 |
COUNTS = ("email", "phone", "pesel", "nip", "regon", "account", "document", "krs", "electronic_address", "pin",
|
| 49 |
"land_register", "registry")
|
| 50 |
|
|
@@ -248,8 +251,8 @@ def _phone_list_context(text: str, start: int) -> bool:
|
|
| 248 |
def _phone_context(text: str, start: int, raw: str) -> bool:
|
| 249 |
# A previous redaction is a boundary, not a fresh "Telefon" cue. Include
|
| 250 |
# one marker's extra width so the 75-character window cannot bisect it.
|
| 251 |
-
before = re.split(r"\n[ \t]*\n|
|
| 252 |
-
text[max(0, start - 75 -
|
| 253 |
if _OTHER_NUMBER_LABEL_RE.search(before):
|
| 254 |
return False
|
| 255 |
intro = _CONTACT_INTRO_RE.search(before)
|
|
@@ -260,7 +263,7 @@ def _phone_context(text: str, start: int, raw: str) -> bool:
|
|
| 260 |
if re.search(r'\bkontakt[ \t]*:[ \t]*\[PII\][ \t,;]*\Z',
|
| 261 |
text[max(0, start-75):start], re.I):
|
| 262 |
return True
|
| 263 |
-
lead = re.split(
|
| 264 |
# "Pod numerem" also introduces contract/case references. Require a
|
| 265 |
# communication cue when the number has no explicit telephone label.
|
| 266 |
number_contact = bool((_PHONE_LABEL_RE.search(lead) or re.search(
|
|
|
|
| 2 |
|
| 3 |
Replaces emails, phones, PESEL/NIP/REGON/KRS, land-register (KW) numbers, electronic contact addresses and
|
| 4 |
account numbers in place so sentence structure survives. Names of public officials are left untouched —
|
| 5 |
+
that is intentional, not a gap. Phones map to [PHONE]; everything else
|
| 6 |
to [PII]. Bare PESEL requires checksum and date validation; explicit identifier
|
| 7 |
labels also redact damaged numbers (including common OCR O/I/l substitutions).
|
| 8 |
NIP/REGON/KRS and identity documents require explicit nearby labels. KW numbers
|
|
|
|
| 27 |
international country/trunk notation such as +CC (0). With a cue, the EUR-Lex
|
| 28 |
"(32-2) 299 11 11" country-area form counts as international. Strong labels also
|
| 29 |
admit one-digit country codes and wider hyphenated area codes. Phone policy v3: each
|
| 30 |
+
number of at least 7 digits (keypad letters count) is its own [PHONE], connectors
|
| 31 |
between numbers stay unmasked, and a shorter part (an extension, "/90") stays inside
|
| 32 |
its number's span. A phone match with no such number (a short service or emergency
|
| 33 |
number, a lone extension) is left as text; a line break alone never splits a
|
|
|
|
| 43 |
import datetime as dt
|
| 44 |
import re
|
| 45 |
|
| 46 |
+
PHONE_TAG = "[PHONE]"
|
| 47 |
PII_TAG = "[PII]"
|
| 48 |
+
LEGACY_PHONE_TAG = "[Telefon]" # NERGAL 1.x output; still a boundary and a heading in text scrubbed before 2.0
|
| 49 |
+
_BOUNDARY = "|".join(map(re.escape, (PHONE_TAG, PII_TAG, LEGACY_PHONE_TAG)))
|
| 50 |
+
_BOUNDARY_WIDTH = max(map(len, (PHONE_TAG, PII_TAG, LEGACY_PHONE_TAG))) # 9: same window as 1.x
|
| 51 |
COUNTS = ("email", "phone", "pesel", "nip", "regon", "account", "document", "krs", "electronic_address", "pin",
|
| 52 |
"land_register", "registry")
|
| 53 |
|
|
|
|
| 251 |
def _phone_context(text: str, start: int, raw: str) -> bool:
|
| 252 |
# A previous redaction is a boundary, not a fresh "Telefon" cue. Include
|
| 253 |
# one marker's extra width so the 75-character window cannot bisect it.
|
| 254 |
+
before = re.split(r"\n[ \t]*\n|" + _BOUNDARY,
|
| 255 |
+
text[max(0, start - 75 - _BOUNDARY_WIDTH):start])[-1][-75:]
|
| 256 |
if _OTHER_NUMBER_LABEL_RE.search(before):
|
| 257 |
return False
|
| 258 |
intro = _CONTACT_INTRO_RE.search(before)
|
|
|
|
| 263 |
if re.search(r'\bkontakt[ \t]*:[ \t]*\[PII\][ \t,;]*\Z',
|
| 264 |
text[max(0, start-75):start], re.I):
|
| 265 |
return True
|
| 266 |
+
lead = re.split(_BOUNDARY, text[max(0, start-240):start])[-1]
|
| 267 |
# "Pod numerem" also introduces contract/case references. Require a
|
| 268 |
# communication cue when the number has no explicit telephone label.
|
| 269 |
number_contact = bool((_PHONE_LABEL_RE.search(lead) or re.search(
|
test_nergal.py
CHANGED
|
@@ -5,7 +5,7 @@ import unittest
|
|
| 5 |
from pathlib import Path
|
| 6 |
|
| 7 |
HERE = Path(__file__).resolve().parent
|
| 8 |
-
RULES_SHA = '
|
| 9 |
|
| 10 |
|
| 11 |
class NergalTests(unittest.TestCase):
|
|
@@ -58,10 +58,28 @@ class NergalTests(unittest.TestCase):
|
|
| 58 |
|
| 59 |
def test_existing_placeholders_do_not_switch_the_rules_off(self):
|
| 60 |
from nergal import rules
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
|
| 64 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
|
| 66 |
def test_grouped_national_phones_mask_without_a_cue(self):
|
| 67 |
from nergal import rules
|
|
@@ -97,7 +115,7 @@ class NergalTests(unittest.TestCase):
|
|
| 97 |
{'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
|
| 98 |
]
|
| 99 |
masked, counts = scrub_spans(text, rules, model, threshold=0.95)
|
| 100 |
-
self.assertIn('[
|
| 101 |
self.assertIn('[PII]', masked)
|
| 102 |
self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
|
| 103 |
self.assertEqual(counts['model_extra_spans'], 1)
|
|
|
|
| 5 |
from pathlib import Path
|
| 6 |
|
| 7 |
HERE = Path(__file__).resolve().parent
|
| 8 |
+
RULES_SHA = 'd18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357'
|
| 9 |
|
| 10 |
|
| 11 |
class NergalTests(unittest.TestCase):
|
|
|
|
| 58 |
|
| 59 |
def test_existing_placeholders_do_not_switch_the_rules_off(self):
|
| 60 |
from nergal import rules
|
| 61 |
+
for marker in ('[PHONE]', '[Telefon]', '[PII]', '[PERSON]'):
|
| 62 |
+
with self.subTest(marker=marker):
|
| 63 |
+
text = f'Kontakt {marker}, NIP 1234567802.' # invented, checksum-valid
|
| 64 |
+
[span] = rules(text)
|
| 65 |
+
self.assertEqual(text[span['start']:span['end']], '1234567802')
|
| 66 |
+
self.assertEqual(rules('a [PII] b [PHONE] c [PERSON] d [Telefon] e'), [])
|
| 67 |
+
|
| 68 |
+
def test_phones_are_tagged_phone(self):
|
| 69 |
+
from nergal import apply_union, rules
|
| 70 |
+
text = 'Biuro: (22) 123 45 67.'
|
| 71 |
+
masked = apply_union(text, rules(text))[0]
|
| 72 |
+
self.assertEqual(masked, 'Biuro: [PHONE].')
|
| 73 |
+
|
| 74 |
+
def test_new_and_legacy_phone_tags_are_the_same_boundary(self):
|
| 75 |
+
from nergal import rules
|
| 76 |
+
def found(text): # values, not offsets: the two tags differ in length
|
| 77 |
+
return [(text[s['start']:s['end']], s['label']) for s in rules(text)]
|
| 78 |
+
for text in ('Telefon: {} lub 601234567', 'Kontakt: {}, 601234567', # plain 9 digits: cue-gated
|
| 79 |
+
'tel. {}\nwew. 123 Jan Nowak\nwew. 456 sekretariat',
|
| 80 |
+
'Kontakt {}: e-mail biuro@example.pl, 601 234 567'): # invented
|
| 81 |
+
with self.subTest(text=text):
|
| 82 |
+
self.assertEqual(found(text.format('[Telefon]')), found(text.format('[PHONE]')))
|
| 83 |
|
| 84 |
def test_grouped_national_phones_mask_without_a_cue(self):
|
| 85 |
from nergal import rules
|
|
|
|
| 115 |
{'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
|
| 116 |
]
|
| 117 |
masked, counts = scrub_spans(text, rules, model, threshold=0.95)
|
| 118 |
+
self.assertIn('[PHONE]', masked)
|
| 119 |
self.assertIn('[PII]', masked)
|
| 120 |
self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
|
| 121 |
self.assertEqual(counts['model_extra_spans'], 1)
|