Spaces:
Running
Running
File size: 6,037 Bytes
efff1ff | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 | """Unit tests for the pure, fast layers: parsing, validation, status mapping.
No model inference here, so this file runs in seconds.
"""
from __future__ import annotations
import io
import pytest
from docxextract import parsing
from docxextract.schema import (ExtractRequest, FieldStatus, UnsupportedMediaError,
ValidationError)
# ---------------------------------------------------------------- media sniff
def test_sniff_pdf():
assert parsing.sniff(b"%PDF-1.7\n...") == "pdf"
@pytest.mark.parametrize("data,expected", [
(b"\x89PNG\r\n\x1a\n", "png"),
(b"\xff\xd8\xff\xe0", "jpeg"),
(b"GIF89a...", "gif"),
(b"BM....", "bmp"),
(b"not a document", "unknown"),
])
def test_sniff_images(data, expected):
assert parsing.sniff(data) == expected
def test_rejects_unknown_type(settings):
with pytest.raises(UnsupportedMediaError):
parsing.parse(b"just some text", settings)
def test_rejects_empty(settings):
with pytest.raises(Exception):
parsing.parse(b"", settings)
# ---------------------------------------------------------------- normalisation
def test_boxes_are_clamped_to_1000(settings):
"""LayoutLM indexes learned position embeddings; out-of-range raises."""
words, boxes = parsing._normalize(
["a", "b"], [[-50, -50, 5000, 5000], [0, 0, 100, 100]], 100, 100)
assert all(0 <= v <= 1000 for b in boxes for v in b)
def test_boxes_ordered_lowercase_first():
_words, boxes = parsing._normalize(["a"], [[300, 400, 100, 200]], 1000, 1000)
x0, y0, x1, y1 = boxes[0]
assert x0 <= x1 and y0 <= y1
def test_zero_size_page_does_not_divide_by_zero():
words, boxes = parsing._normalize(["a"], [[10, 10, 20, 20]], 0, 0)
assert len(words) == 1 and len(boxes) == 1
# ---------------------------------------------------------------- parsing
def test_parses_pdf_text_layer(corpus_dir, settings, ground_truth):
from pathlib import Path
data = Path(ground_truth[0]["path"]).read_bytes()
doc = parsing.parse(data, settings)
assert doc.word_count > 50
assert doc.source_type.value == "pdf_text_layer"
assert doc.page_count >= 1
def test_parse_cache_hits_on_repeat(corpus_dir, settings, ground_truth):
from pathlib import Path
data = Path(ground_truth[0]["path"]).read_bytes()
first = parsing.parse(data, settings)
before = parsing.cache_stats()["hits"]
second = parsing.parse(data, settings)
assert second.document_id == first.document_id
assert parsing.cache_stats()["hits"] > before
def test_blank_pdf_reports_no_words(tmp_path, settings):
from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas
pytest.importorskip("pytesseract")
if not parsing.tesseract_available():
pytest.skip("Tesseract binary not on PATH")
path = tmp_path / "blank.pdf"
c = canvas.Canvas(str(path), pagesize=A4)
c.showPage()
c.save()
doc = parsing.parse(path.read_bytes(), settings)
# Either genuinely empty, or OCR of a blank page yields nothing usable.
assert doc.word_count < 5
def test_tesseract_probe_does_not_raise():
"""The availability probe must never raise, whatever the environment."""
assert isinstance(parsing.tesseract_available(), bool)
# ---------------------------------------------------------------- validation
def test_extract_request_rejects_empty_keys():
with pytest.raises(Exception):
ExtractRequest(keys=["", " "])
def test_extract_request_deduplicates_preserving_order():
req = ExtractRequest(keys=["b", "a", "b", "c"])
assert req.keys == ["b", "a", "c"]
def test_extract_request_rejects_overlong_key():
with pytest.raises(Exception):
ExtractRequest(keys=["x" * 500])
# ---------------------------------------------------------------- status mapping
def test_status_below_threshold_is_low_confidence(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="$10.00", confidence=0.2, page=1, start=0, end=0)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.LOW_CONFIDENCE
# The value is still returned so a human reviewer can judge it.
assert field.value == "$10.00"
def test_status_empty_answer_is_not_found(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="", confidence=0.99, page=1, start=-1, end=-1)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.NOT_FOUND
assert field.value is None
def test_status_high_confidence_is_extracted(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="INV-1", confidence=0.95, page=1, start=0, end=0)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.EXTRACTED
assert field.is_usable
def test_service_rejects_no_keys(settings, sample_pdf):
from docxextract.service import ExtractionService
service = ExtractionService(settings)
with pytest.raises(ValidationError):
service.extract(sample_pdf, [])
def test_service_rejects_blank_document(settings, tmp_path):
from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas
from docxextract.service import ExtractionService
pytest.importorskip("pytesseract")
if not parsing.tesseract_available():
pytest.skip("Tesseract binary not on PATH")
path = tmp_path / "empty.pdf"
c = canvas.Canvas(str(path), pagesize=A4)
c.showPage()
c.save()
service = ExtractionService(settings)
with pytest.raises(ValidationError):
service.extract(path.read_bytes(), ["Total"])
|