File size: 1,097 Bytes
ade9388
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
import re
import ftfy
import nltk
from nltk.tokenize import PunktSentenceTokenizer

# Ensure required NLTK tokenizer resources are downloaded
for pkg in ['punkt', 'punkt_tab']:
    try:
        nltk.data.find(f'tokenizers/{pkg}')
    except LookupError:
        nltk.download(pkg, quiet=True)


def clean_text_pipeline(raw_text: str) -> str:
    """Deterministic cleaning pipeline for legal documents:
    1. Repairs encoding artifacts and Mojibake (via ftfy).
    2. Normalizes hard line wraps into single spaces while preserving paragraph breaks.
    3. Normalizes consecutive whitespace and tabs.
    """
    if not raw_text:
        return ""
    text = ftfy.fix_text(raw_text)
    text = re.sub(r'(?<!\n)\n(?!\n)', ' ', text)
    text = re.sub(r'[ \t]+', ' ', text)
    return text.strip()


def segment_sentences(cleaned_text: str):
    """Tokenize cleaned legal text into character spans: [(start_char, end_char), ...]."""
    if not cleaned_text or not cleaned_text.strip():
        return []
    tokenizer = PunktSentenceTokenizer()
    return list(tokenizer.span_tokenize(cleaned_text))