File size: 6,037 Bytes
efff1ff
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
"""Unit tests for the pure, fast layers: parsing, validation, status mapping.

No model inference here, so this file runs in seconds.
"""

from __future__ import annotations

import io

import pytest

from docxextract import parsing
from docxextract.schema import (ExtractRequest, FieldStatus, UnsupportedMediaError,
                                ValidationError)


# ---------------------------------------------------------------- media sniff
def test_sniff_pdf():
    assert parsing.sniff(b"%PDF-1.7\n...") == "pdf"


@pytest.mark.parametrize("data,expected", [
    (b"\x89PNG\r\n\x1a\n", "png"),
    (b"\xff\xd8\xff\xe0", "jpeg"),
    (b"GIF89a...", "gif"),
    (b"BM....", "bmp"),
    (b"not a document", "unknown"),
])
def test_sniff_images(data, expected):
    assert parsing.sniff(data) == expected


def test_rejects_unknown_type(settings):
    with pytest.raises(UnsupportedMediaError):
        parsing.parse(b"just some text", settings)


def test_rejects_empty(settings):
    with pytest.raises(Exception):
        parsing.parse(b"", settings)


# ---------------------------------------------------------------- normalisation
def test_boxes_are_clamped_to_1000(settings):
    """LayoutLM indexes learned position embeddings; out-of-range raises."""
    words, boxes = parsing._normalize(
        ["a", "b"], [[-50, -50, 5000, 5000], [0, 0, 100, 100]], 100, 100)
    assert all(0 <= v <= 1000 for b in boxes for v in b)


def test_boxes_ordered_lowercase_first():
    _words, boxes = parsing._normalize(["a"], [[300, 400, 100, 200]], 1000, 1000)
    x0, y0, x1, y1 = boxes[0]
    assert x0 <= x1 and y0 <= y1


def test_zero_size_page_does_not_divide_by_zero():
    words, boxes = parsing._normalize(["a"], [[10, 10, 20, 20]], 0, 0)
    assert len(words) == 1 and len(boxes) == 1


# ---------------------------------------------------------------- parsing
def test_parses_pdf_text_layer(corpus_dir, settings, ground_truth):
    from pathlib import Path
    data = Path(ground_truth[0]["path"]).read_bytes()
    doc = parsing.parse(data, settings)
    assert doc.word_count > 50
    assert doc.source_type.value == "pdf_text_layer"
    assert doc.page_count >= 1


def test_parse_cache_hits_on_repeat(corpus_dir, settings, ground_truth):
    from pathlib import Path
    data = Path(ground_truth[0]["path"]).read_bytes()
    first = parsing.parse(data, settings)
    before = parsing.cache_stats()["hits"]
    second = parsing.parse(data, settings)
    assert second.document_id == first.document_id
    assert parsing.cache_stats()["hits"] > before


def test_blank_pdf_reports_no_words(tmp_path, settings):
    from reportlab.lib.pagesizes import A4
    from reportlab.pdfgen import canvas

    pytest.importorskip("pytesseract")
    if not parsing.tesseract_available():
        pytest.skip("Tesseract binary not on PATH")

    path = tmp_path / "blank.pdf"
    c = canvas.Canvas(str(path), pagesize=A4)
    c.showPage()
    c.save()
    doc = parsing.parse(path.read_bytes(), settings)
    # Either genuinely empty, or OCR of a blank page yields nothing usable.
    assert doc.word_count < 5


def test_tesseract_probe_does_not_raise():
    """The availability probe must never raise, whatever the environment."""
    assert isinstance(parsing.tesseract_available(), bool)


# ---------------------------------------------------------------- validation
def test_extract_request_rejects_empty_keys():
    with pytest.raises(Exception):
        ExtractRequest(keys=["", "   "])


def test_extract_request_deduplicates_preserving_order():
    req = ExtractRequest(keys=["b", "a", "b", "c"])
    assert req.keys == ["b", "a", "c"]


def test_extract_request_rejects_overlong_key():
    with pytest.raises(Exception):
        ExtractRequest(keys=["x" * 500])


# ---------------------------------------------------------------- status mapping
def test_status_below_threshold_is_low_confidence(settings):
    from docxextract.engine import Span
    from docxextract.service import ExtractionService

    service = ExtractionService(settings)
    span = Span(key="k", answer="$10.00", confidence=0.2, page=1, start=0, end=0)
    field = service._status_for(span, threshold=0.5, page_has_words=True)
    assert field.status is FieldStatus.LOW_CONFIDENCE
    # The value is still returned so a human reviewer can judge it.
    assert field.value == "$10.00"


def test_status_empty_answer_is_not_found(settings):
    from docxextract.engine import Span
    from docxextract.service import ExtractionService

    service = ExtractionService(settings)
    span = Span(key="k", answer="", confidence=0.99, page=1, start=-1, end=-1)
    field = service._status_for(span, threshold=0.5, page_has_words=True)
    assert field.status is FieldStatus.NOT_FOUND
    assert field.value is None


def test_status_high_confidence_is_extracted(settings):
    from docxextract.engine import Span
    from docxextract.service import ExtractionService

    service = ExtractionService(settings)
    span = Span(key="k", answer="INV-1", confidence=0.95, page=1, start=0, end=0)
    field = service._status_for(span, threshold=0.5, page_has_words=True)
    assert field.status is FieldStatus.EXTRACTED
    assert field.is_usable


def test_service_rejects_no_keys(settings, sample_pdf):
    from docxextract.service import ExtractionService
    service = ExtractionService(settings)
    with pytest.raises(ValidationError):
        service.extract(sample_pdf, [])


def test_service_rejects_blank_document(settings, tmp_path):
    from reportlab.lib.pagesizes import A4
    from reportlab.pdfgen import canvas

    from docxextract.service import ExtractionService

    pytest.importorskip("pytesseract")
    if not parsing.tesseract_available():
        pytest.skip("Tesseract binary not on PATH")

    path = tmp_path / "empty.pdf"
    c = canvas.Canvas(str(path), pagesize=A4)
    c.showPage()
    c.save()
    service = ExtractionService(settings)
    with pytest.raises(ValidationError):
        service.extract(path.read_bytes(), ["Total"])