SERPent / tests /test_ops_parsing.py
Claude
Add pytest harness with characterization tests for scrap/ops/serp/utils
5796881 unverified
Raw History Blame Contribute Delete
7.32 kB
"""Unit tests for ops.py's pure JSON-walking helpers.
OPS's XML-to-JSON conversion is the trickiest part of this module: text
nodes become {"$": "..."}, attributes become "@name" keys, and a repeated
XML element becomes a list *unless there's only one*, in which case it stays
a bare dict. Every helper here has to handle both shapes; these tests pin
both for each one.
"""
import pytest
from ops import (
_as_list,
_text,
_pick_lang,
_epodoc_number,
_extract_title,
_extract_abstract,
_extract_classifications,
_fulltext_paragraphs,
_to_cql,
_normalize_epodoc,
)
class TestAsList:
def test_none_becomes_empty_list(self):
assert _as_list(None) == []
def test_single_dict_is_wrapped(self):
assert _as_list({"a": 1}) == [{"a": 1}]
def test_list_passes_through_unchanged(self):
assert _as_list([{"a": 1}, {"a": 2}]) == [{"a": 1}, {"a": 2}]
class TestText:
def test_extracts_dollar_key(self):
assert _text({"$": "hello"}) == "hello"
def test_plain_string_passes_through(self):
assert _text("hello") == "hello"
def test_none_for_anything_else(self):
assert _text(None) is None
assert _text({"@lang": "en"}) is None
class TestPickLang:
def test_picks_requested_language_from_list(self):
nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "en", "$": "Hello"}]
assert _pick_lang(nodes) == {"@lang": "en", "$": "Hello"}
def test_falls_back_to_first_when_language_missing(self):
nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "fr", "$": "Bonjour"}]
assert _pick_lang(nodes) == {"@lang": "de", "$": "Hallo"}
def test_single_dict_not_wrapped_in_a_list_still_works(self):
node = {"@lang": "en", "$": "Hello"}
assert _pick_lang(node) == node
def test_empty_input_returns_none(self):
assert _pick_lang(None) is None
assert _pick_lang([]) is None
class TestEpodocNumber:
def test_prefers_the_epodoc_document_id(self):
exchange_doc = {
"@country": "US",
"@doc-number": "11930446",
"bibliographic-data": {
"publication-reference": {
"document-id": [
{"@document-id-type": "docdb", "doc-number": {"$": "11930446"}},
{"@document-id-type": "epodoc", "doc-number": {"$": "US11930446"}},
]
}
},
}
assert _epodoc_number(exchange_doc) == "US11930446"
def test_falls_back_to_country_plus_doc_number(self):
exchange_doc = {"@country": "US", "@doc-number": "11930446", "bibliographic-data": {}}
assert _epodoc_number(exchange_doc) == "US11930446"
def test_none_when_nothing_available(self):
assert _epodoc_number({"bibliographic-data": {}}) is None
class TestExtractTitle:
def test_from_a_list_of_languages(self):
bd = {"invention-title": [{"@lang": "en", "$": "Widget"}, {"@lang": "de", "$": "Widget-DE"}]}
assert _extract_title(bd) == "Widget"
def test_from_a_single_bare_dict(self):
bd = {"invention-title": {"@lang": "en", "$": "Widget"}}
assert _extract_title(bd) == "Widget"
def test_none_when_missing(self):
assert _extract_title({}) is None
class TestExtractAbstract:
def test_joins_paragraphs_from_a_list(self):
doc = {"abstract": [{"@lang": "en", "p": [{"$": "Part one."}, {"$": "Part two."}]}]}
assert _extract_abstract(doc) == "Part one. Part two."
def test_single_paragraph_not_wrapped_in_a_list(self):
doc = {"abstract": {"@lang": "en", "p": {"$": "Only paragraph."}}}
assert _extract_abstract(doc) == "Only paragraph."
def test_none_when_missing(self):
assert _extract_abstract({}) is None
class TestExtractClassifications:
def test_builds_codes_from_the_five_part_symbol(self):
bd = {
"patent-classifications": {
"patent-classification": [
{
"section": {"$": "G"},
"class": {"$": "06"},
"subclass": {"$": "F"},
"main-group": {"$": "17"},
"subgroup": {"$": "30"},
}
]
}
}
codes = _extract_classifications(bd)
assert len(codes) == 1
assert codes[0].code == "G06F17/30"
assert codes[0].description == ""
def test_dedupes_and_skips_incomplete_entries(self):
bd = {
"patent-classifications": {
"patent-classification": [
{"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}},
{"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}},
{"section": {"$": "H"}}, # incomplete - no class/subclass/main-group
]
}
}
codes = _extract_classifications(bd)
assert [c.code for c in codes] == ["G06F17/30"]
def test_none_when_no_classifications(self):
assert _extract_classifications({}) is None
class TestFulltextParagraphs:
def test_claims_join_claim_text_lines(self):
data = {
"ops:world-patent-data": {
"ftxt:fulltext-documents": {
"ftxt:fulltext-document": {
"claims": {
"@lang": "en",
"claim": [
{"claim-text": {"$": "1. A widget."}},
{"claim-text": [{"$": "2. A gadget,"}, {"$": "further comprising a mechanism."}]},
],
}
}
}
}
}
assert _fulltext_paragraphs(data, "claims") == (
"1. A widget.\n2. A gadget,\nfurther comprising a mechanism."
)
def test_description_joins_paragraphs(self):
data = {
"ops:world-patent-data": {
"ftxt:fulltext-documents": {
"ftxt:fulltext-document": {
"description": {"@lang": "en", "p": [{"$": "Para one."}, {"$": "Para two."}]}
}
}
}
}
assert _fulltext_paragraphs(data, "description") == "Para one.\nPara two."
def test_none_when_document_missing(self):
assert _fulltext_paragraphs({"ops:world-patent-data": {}}, "claims") is None
@pytest.mark.parametrize(
"raw, expected",
[
("widget", 'txt all "widget"'),
("agentic ai", 'txt all "agentic ai"'),
('pn = "US1234"', 'pn = "US1234"'),
("", ""),
('has "quotes"', 'txt all "has quotes"'),
],
)
def test_to_cql(raw, expected):
assert _to_cql(raw) == expected
@pytest.mark.parametrize(
"raw, expected",
[
("US11930446B2", "US11930446"),
("EP4760514A1", "EP4760514"),
("US 11930446 B2", "US11930446"),
("US11930446", "US11930446"),
],
)
def test_normalize_epodoc(raw, expected):
assert _normalize_epodoc(raw) == expected