instance-2 / docqa /tests /test_unit.py
validops-east-3's picture
Deploy 72701a1d9ec33b1842d0a516ffdee753aaa754a9
efff1ff verified
Raw History Blame Contribute Delete
6.04 kB
"""Unit tests for the pure, fast layers: parsing, validation, status mapping.
No model inference here, so this file runs in seconds.
"""
from __future__ import annotations
import io
import pytest
from docxextract import parsing
from docxextract.schema import (ExtractRequest, FieldStatus, UnsupportedMediaError,
ValidationError)
# ---------------------------------------------------------------- media sniff
def test_sniff_pdf():
assert parsing.sniff(b"%PDF-1.7\n...") == "pdf"
@pytest.mark.parametrize("data,expected", [
(b"\x89PNG\r\n\x1a\n", "png"),
(b"\xff\xd8\xff\xe0", "jpeg"),
(b"GIF89a...", "gif"),
(b"BM....", "bmp"),
(b"not a document", "unknown"),
])
def test_sniff_images(data, expected):
assert parsing.sniff(data) == expected
def test_rejects_unknown_type(settings):
with pytest.raises(UnsupportedMediaError):
parsing.parse(b"just some text", settings)
def test_rejects_empty(settings):
with pytest.raises(Exception):
parsing.parse(b"", settings)
# ---------------------------------------------------------------- normalisation
def test_boxes_are_clamped_to_1000(settings):
"""LayoutLM indexes learned position embeddings; out-of-range raises."""
words, boxes = parsing._normalize(
["a", "b"], [[-50, -50, 5000, 5000], [0, 0, 100, 100]], 100, 100)
assert all(0 <= v <= 1000 for b in boxes for v in b)
def test_boxes_ordered_lowercase_first():
_words, boxes = parsing._normalize(["a"], [[300, 400, 100, 200]], 1000, 1000)
x0, y0, x1, y1 = boxes[0]
assert x0 <= x1 and y0 <= y1
def test_zero_size_page_does_not_divide_by_zero():
words, boxes = parsing._normalize(["a"], [[10, 10, 20, 20]], 0, 0)
assert len(words) == 1 and len(boxes) == 1
# ---------------------------------------------------------------- parsing
def test_parses_pdf_text_layer(corpus_dir, settings, ground_truth):
from pathlib import Path
data = Path(ground_truth[0]["path"]).read_bytes()
doc = parsing.parse(data, settings)
assert doc.word_count > 50
assert doc.source_type.value == "pdf_text_layer"
assert doc.page_count >= 1
def test_parse_cache_hits_on_repeat(corpus_dir, settings, ground_truth):
from pathlib import Path
data = Path(ground_truth[0]["path"]).read_bytes()
first = parsing.parse(data, settings)
before = parsing.cache_stats()["hits"]
second = parsing.parse(data, settings)
assert second.document_id == first.document_id
assert parsing.cache_stats()["hits"] > before
def test_blank_pdf_reports_no_words(tmp_path, settings):
from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas
pytest.importorskip("pytesseract")
if not parsing.tesseract_available():
pytest.skip("Tesseract binary not on PATH")
path = tmp_path / "blank.pdf"
c = canvas.Canvas(str(path), pagesize=A4)
c.showPage()
c.save()
doc = parsing.parse(path.read_bytes(), settings)
# Either genuinely empty, or OCR of a blank page yields nothing usable.
assert doc.word_count < 5
def test_tesseract_probe_does_not_raise():
"""The availability probe must never raise, whatever the environment."""
assert isinstance(parsing.tesseract_available(), bool)
# ---------------------------------------------------------------- validation
def test_extract_request_rejects_empty_keys():
with pytest.raises(Exception):
ExtractRequest(keys=["", " "])
def test_extract_request_deduplicates_preserving_order():
req = ExtractRequest(keys=["b", "a", "b", "c"])
assert req.keys == ["b", "a", "c"]
def test_extract_request_rejects_overlong_key():
with pytest.raises(Exception):
ExtractRequest(keys=["x" * 500])
# ---------------------------------------------------------------- status mapping
def test_status_below_threshold_is_low_confidence(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="$10.00", confidence=0.2, page=1, start=0, end=0)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.LOW_CONFIDENCE
# The value is still returned so a human reviewer can judge it.
assert field.value == "$10.00"
def test_status_empty_answer_is_not_found(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="", confidence=0.99, page=1, start=-1, end=-1)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.NOT_FOUND
assert field.value is None
def test_status_high_confidence_is_extracted(settings):
from docxextract.engine import Span
from docxextract.service import ExtractionService
service = ExtractionService(settings)
span = Span(key="k", answer="INV-1", confidence=0.95, page=1, start=0, end=0)
field = service._status_for(span, threshold=0.5, page_has_words=True)
assert field.status is FieldStatus.EXTRACTED
assert field.is_usable
def test_service_rejects_no_keys(settings, sample_pdf):
from docxextract.service import ExtractionService
service = ExtractionService(settings)
with pytest.raises(ValidationError):
service.extract(sample_pdf, [])
def test_service_rejects_blank_document(settings, tmp_path):
from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas
from docxextract.service import ExtractionService
pytest.importorskip("pytesseract")
if not parsing.tesseract_available():
pytest.skip("Tesseract binary not on PATH")
path = tmp_path / "empty.pdf"
c = canvas.Canvas(str(path), pagesize=A4)
c.showPage()
c.save()
service = ExtractionService(settings)
with pytest.raises(ValidationError):
service.extract(path.read_bytes(), ["Total"])