ppuzio Claude Opus 5.5 commited on
Commit
75130f0
·
1 Parent(s): bbe063e

2.0.0: [PHONE] replaces [Telefon]; [Telefon] stays a rules boundary

Browse files
Files changed (4) hide show
  1. hybrid.json +1 -1
  2. nergal.py +2 -2
  3. scrub_pii.py +9 -6
  4. test_nergal.py +24 -6
hybrid.json CHANGED
@@ -6,7 +6,7 @@
6
  "epoch": 5,
7
  "seed": 202609160,
8
  "threshold": 0.95,
9
- "rules_sha256": "b238d5b88aa3f3d55a24bb051ec93f9179dfb14b2c650441c8e0acb225d81d59",
10
  "eval": {
11
  "split": "841-dev",
12
  "gold_amendments": [
 
6
  "epoch": 5,
7
  "seed": 202609160,
8
  "threshold": 0.95,
9
+ "rules_sha256": "d18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357",
10
  "eval": {
11
  "split": "841-dev",
12
  "gold_amendments": [
nergal.py CHANGED
@@ -23,7 +23,7 @@ GAP_IDS = [250002, 250003]
23
  BIO_LABELS = ['O', 'B-phone', 'I-phone', 'B-pii', 'I-pii']
24
  LABELS = ['phone', 'pii']
25
  THRESHOLD = 0.95
26
- RULES_SHA = 'b238d5b88aa3f3d55a24bb051ec93f9179dfb14b2c650441c8e0acb225d81d59'
27
  DTYPES = ('float32', 'float16')
28
 
29
 
@@ -168,7 +168,7 @@ class Encoding:
168
 
169
 
170
  def spans_from(module, text):
171
- """Rule spans on the original text. Existing [PII]/[Telefon] placeholders do not switch the rules off."""
172
  result = []
173
  module.scrub_pii(text, spans=result)
174
  return sorted(({k: s[k] for k in ('start', 'end', 'label')} | {'score': 1.0} for s in result),
 
23
  BIO_LABELS = ['O', 'B-phone', 'I-phone', 'B-pii', 'I-pii']
24
  LABELS = ['phone', 'pii']
25
  THRESHOLD = 0.95
26
+ RULES_SHA = 'd18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357'
27
  DTYPES = ('float32', 'float16')
28
 
29
 
 
168
 
169
 
170
  def spans_from(module, text):
171
+ """Rule spans on the original text. Existing [PHONE]/[PII]/[Telefon] placeholders do not switch the rules off."""
172
  result = []
173
  module.scrub_pii(text, spans=result)
174
  return sorted(({k: s[k] for k in ('start', 'end', 'label')} | {'score': 1.0} for s in result),
scrub_pii.py CHANGED
@@ -2,7 +2,7 @@
2
 
3
  Replaces emails, phones, PESEL/NIP/REGON/KRS, land-register (KW) numbers, electronic contact addresses and
4
  account numbers in place so sentence structure survives. Names of public officials are left untouched —
5
- that is intentional, not a gap. Phones map to [Telefon]; everything else
6
  to [PII]. Bare PESEL requires checksum and date validation; explicit identifier
7
  labels also redact damaged numbers (including common OCR O/I/l substitutions).
8
  NIP/REGON/KRS and identity documents require explicit nearby labels. KW numbers
@@ -27,7 +27,7 @@ Phones require a nearby contact cue, a Polish +48/0048 prefix, or explicit
27
  international country/trunk notation such as +CC (0). With a cue, the EUR-Lex
28
  "(32-2) 299 11 11" country-area form counts as international. Strong labels also
29
  admit one-digit country codes and wider hyphenated area codes. Phone policy v3: each
30
- number of at least 7 digits (keypad letters count) is its own [Telefon], connectors
31
  between numbers stay unmasked, and a shorter part (an extension, "/90") stays inside
32
  its number's span. A phone match with no such number (a short service or emergency
33
  number, a lone extension) is left as text; a line break alone never splits a
@@ -43,8 +43,11 @@ from __future__ import annotations
43
  import datetime as dt
44
  import re
45
 
46
- PHONE_TAG = "[Telefon]"
47
  PII_TAG = "[PII]"
 
 
 
48
  COUNTS = ("email", "phone", "pesel", "nip", "regon", "account", "document", "krs", "electronic_address", "pin",
49
  "land_register", "registry")
50
 
@@ -248,8 +251,8 @@ def _phone_list_context(text: str, start: int) -> bool:
248
  def _phone_context(text: str, start: int, raw: str) -> bool:
249
  # A previous redaction is a boundary, not a fresh "Telefon" cue. Include
250
  # one marker's extra width so the 75-character window cannot bisect it.
251
- before = re.split(r"\n[ \t]*\n|\[Telefon\]|\[PII\]",
252
- text[max(0, start - 75 - len(PHONE_TAG)):start])[-1][-75:]
253
  if _OTHER_NUMBER_LABEL_RE.search(before):
254
  return False
255
  intro = _CONTACT_INTRO_RE.search(before)
@@ -260,7 +263,7 @@ def _phone_context(text: str, start: int, raw: str) -> bool:
260
  if re.search(r'\bkontakt[ \t]*:[ \t]*\[PII\][ \t,;]*\Z',
261
  text[max(0, start-75):start], re.I):
262
  return True
263
- lead = re.split(r'\[Telefon\]|\[PII\]', text[max(0, start-240):start])[-1]
264
  # "Pod numerem" also introduces contract/case references. Require a
265
  # communication cue when the number has no explicit telephone label.
266
  number_contact = bool((_PHONE_LABEL_RE.search(lead) or re.search(
 
2
 
3
  Replaces emails, phones, PESEL/NIP/REGON/KRS, land-register (KW) numbers, electronic contact addresses and
4
  account numbers in place so sentence structure survives. Names of public officials are left untouched —
5
+ that is intentional, not a gap. Phones map to [PHONE]; everything else
6
  to [PII]. Bare PESEL requires checksum and date validation; explicit identifier
7
  labels also redact damaged numbers (including common OCR O/I/l substitutions).
8
  NIP/REGON/KRS and identity documents require explicit nearby labels. KW numbers
 
27
  international country/trunk notation such as +CC (0). With a cue, the EUR-Lex
28
  "(32-2) 299 11 11" country-area form counts as international. Strong labels also
29
  admit one-digit country codes and wider hyphenated area codes. Phone policy v3: each
30
+ number of at least 7 digits (keypad letters count) is its own [PHONE], connectors
31
  between numbers stay unmasked, and a shorter part (an extension, "/90") stays inside
32
  its number's span. A phone match with no such number (a short service or emergency
33
  number, a lone extension) is left as text; a line break alone never splits a
 
43
  import datetime as dt
44
  import re
45
 
46
+ PHONE_TAG = "[PHONE]"
47
  PII_TAG = "[PII]"
48
+ LEGACY_PHONE_TAG = "[Telefon]" # NERGAL 1.x output; still a boundary and a heading in text scrubbed before 2.0
49
+ _BOUNDARY = "|".join(map(re.escape, (PHONE_TAG, PII_TAG, LEGACY_PHONE_TAG)))
50
+ _BOUNDARY_WIDTH = max(map(len, (PHONE_TAG, PII_TAG, LEGACY_PHONE_TAG))) # 9: same window as 1.x
51
  COUNTS = ("email", "phone", "pesel", "nip", "regon", "account", "document", "krs", "electronic_address", "pin",
52
  "land_register", "registry")
53
 
 
251
  def _phone_context(text: str, start: int, raw: str) -> bool:
252
  # A previous redaction is a boundary, not a fresh "Telefon" cue. Include
253
  # one marker's extra width so the 75-character window cannot bisect it.
254
+ before = re.split(r"\n[ \t]*\n|" + _BOUNDARY,
255
+ text[max(0, start - 75 - _BOUNDARY_WIDTH):start])[-1][-75:]
256
  if _OTHER_NUMBER_LABEL_RE.search(before):
257
  return False
258
  intro = _CONTACT_INTRO_RE.search(before)
 
263
  if re.search(r'\bkontakt[ \t]*:[ \t]*\[PII\][ \t,;]*\Z',
264
  text[max(0, start-75):start], re.I):
265
  return True
266
+ lead = re.split(_BOUNDARY, text[max(0, start-240):start])[-1]
267
  # "Pod numerem" also introduces contract/case references. Require a
268
  # communication cue when the number has no explicit telephone label.
269
  number_contact = bool((_PHONE_LABEL_RE.search(lead) or re.search(
test_nergal.py CHANGED
@@ -5,7 +5,7 @@ import unittest
5
  from pathlib import Path
6
 
7
  HERE = Path(__file__).resolve().parent
8
- RULES_SHA = 'b238d5b88aa3f3d55a24bb051ec93f9179dfb14b2c650441c8e0acb225d81d59'
9
 
10
 
11
  class NergalTests(unittest.TestCase):
@@ -58,10 +58,28 @@ class NergalTests(unittest.TestCase):
58
 
59
  def test_existing_placeholders_do_not_switch_the_rules_off(self):
60
  from nergal import rules
61
- text = 'Kontakt [Telefon], NIP 1234567802.' # invented, checksum-valid
62
- [span] = rules(text)
63
- self.assertEqual(text[span['start']:span['end']], '1234567802')
64
- self.assertEqual(rules('a [PII] b [Telefon] c'), [])
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
65
 
66
  def test_grouped_national_phones_mask_without_a_cue(self):
67
  from nergal import rules
@@ -97,7 +115,7 @@ class NergalTests(unittest.TestCase):
97
  {'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
98
  ]
99
  masked, counts = scrub_spans(text, rules, model, threshold=0.95)
100
- self.assertIn('[Telefon]', masked)
101
  self.assertIn('[PII]', masked)
102
  self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
103
  self.assertEqual(counts['model_extra_spans'], 1)
 
5
  from pathlib import Path
6
 
7
  HERE = Path(__file__).resolve().parent
8
+ RULES_SHA = 'd18662434b3bab69d74e7f63f76a646122cd9d7420de1d70914aaeb903ff2357'
9
 
10
 
11
  class NergalTests(unittest.TestCase):
 
58
 
59
  def test_existing_placeholders_do_not_switch_the_rules_off(self):
60
  from nergal import rules
61
+ for marker in ('[PHONE]', '[Telefon]', '[PII]', '[PERSON]'):
62
+ with self.subTest(marker=marker):
63
+ text = f'Kontakt {marker}, NIP 1234567802.' # invented, checksum-valid
64
+ [span] = rules(text)
65
+ self.assertEqual(text[span['start']:span['end']], '1234567802')
66
+ self.assertEqual(rules('a [PII] b [PHONE] c [PERSON] d [Telefon] e'), [])
67
+
68
+ def test_phones_are_tagged_phone(self):
69
+ from nergal import apply_union, rules
70
+ text = 'Biuro: (22) 123 45 67.'
71
+ masked = apply_union(text, rules(text))[0]
72
+ self.assertEqual(masked, 'Biuro: [PHONE].')
73
+
74
+ def test_new_and_legacy_phone_tags_are_the_same_boundary(self):
75
+ from nergal import rules
76
+ def found(text): # values, not offsets: the two tags differ in length
77
+ return [(text[s['start']:s['end']], s['label']) for s in rules(text)]
78
+ for text in ('Telefon: {} lub 601234567', 'Kontakt: {}, 601234567', # plain 9 digits: cue-gated
79
+ 'tel. {}\nwew. 123 Jan Nowak\nwew. 456 sekretariat',
80
+ 'Kontakt {}: e-mail biuro@example.pl, 601 234 567'): # invented
81
+ with self.subTest(text=text):
82
+ self.assertEqual(found(text.format('[Telefon]')), found(text.format('[PHONE]')))
83
 
84
  def test_grouped_national_phones_mask_without_a_cue(self):
85
  from nergal import rules
 
115
  {'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
116
  ]
117
  masked, counts = scrub_spans(text, rules, model, threshold=0.95)
118
+ self.assertIn('[PHONE]', masked)
119
  self.assertIn('[PII]', masked)
120
  self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
121
  self.assertEqual(counts['model_extra_spans'], 1)