File size: 1,152 Bytes
ade9388
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
import pytest
from gotcha.preprocessor import clean_text_pipeline, segment_sentences


def test_clean_text_mojibake_fix():
    # Broken UTF-8 quotation marks and apostrophes
    raw = "The company’s services are provided “as is”."
    cleaned = clean_text_pipeline(raw)
    assert "company’s" in cleaned or "company's" in cleaned
    assert "“as is”" in cleaned or '"as is"' in cleaned


def test_clean_text_hard_wraps():
    raw = "This is a sentence\nsplit across lines.\n\nThis is a new paragraph."
    cleaned = clean_text_pipeline(raw)
    assert "This is a sentence split across lines." in cleaned
    assert "\n\n" in cleaned or "This is a new paragraph." in cleaned


def test_clean_text_whitespace_normalization():
    raw = "Multiple    spaces \t and tabs   normalized."
    cleaned = clean_text_pipeline(raw)
    assert cleaned == "Multiple spaces and tabs normalized."


def test_segment_sentences():
    text = "First sentence here. Second sentence here. Third sentence."
    spans = segment_sentences(text)
    assert len(spans) == 3
    s0 = text[spans[0][0]:spans[0][1]]
    assert s0 == "First sentence here."