"""Unit tests for ops.py's pure JSON-walking helpers. OPS's XML-to-JSON conversion is the trickiest part of this module: text nodes become {"$": "..."}, attributes become "@name" keys, and a repeated XML element becomes a list *unless there's only one*, in which case it stays a bare dict. Every helper here has to handle both shapes; these tests pin both for each one. """ import pytest from ops import ( _as_list, _text, _pick_lang, _epodoc_number, _extract_title, _extract_abstract, _extract_classifications, _fulltext_paragraphs, _to_cql, _normalize_epodoc, ) class TestAsList: def test_none_becomes_empty_list(self): assert _as_list(None) == [] def test_single_dict_is_wrapped(self): assert _as_list({"a": 1}) == [{"a": 1}] def test_list_passes_through_unchanged(self): assert _as_list([{"a": 1}, {"a": 2}]) == [{"a": 1}, {"a": 2}] class TestText: def test_extracts_dollar_key(self): assert _text({"$": "hello"}) == "hello" def test_plain_string_passes_through(self): assert _text("hello") == "hello" def test_none_for_anything_else(self): assert _text(None) is None assert _text({"@lang": "en"}) is None class TestPickLang: def test_picks_requested_language_from_list(self): nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "en", "$": "Hello"}] assert _pick_lang(nodes) == {"@lang": "en", "$": "Hello"} def test_falls_back_to_first_when_language_missing(self): nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "fr", "$": "Bonjour"}] assert _pick_lang(nodes) == {"@lang": "de", "$": "Hallo"} def test_single_dict_not_wrapped_in_a_list_still_works(self): node = {"@lang": "en", "$": "Hello"} assert _pick_lang(node) == node def test_empty_input_returns_none(self): assert _pick_lang(None) is None assert _pick_lang([]) is None class TestEpodocNumber: def test_prefers_the_epodoc_document_id(self): exchange_doc = { "@country": "US", "@doc-number": "11930446", "bibliographic-data": { "publication-reference": { "document-id": [ {"@document-id-type": "docdb", "doc-number": {"$": "11930446"}}, {"@document-id-type": "epodoc", "doc-number": {"$": "US11930446"}}, ] } }, } assert _epodoc_number(exchange_doc) == "US11930446" def test_falls_back_to_country_plus_doc_number(self): exchange_doc = {"@country": "US", "@doc-number": "11930446", "bibliographic-data": {}} assert _epodoc_number(exchange_doc) == "US11930446" def test_none_when_nothing_available(self): assert _epodoc_number({"bibliographic-data": {}}) is None class TestExtractTitle: def test_from_a_list_of_languages(self): bd = {"invention-title": [{"@lang": "en", "$": "Widget"}, {"@lang": "de", "$": "Widget-DE"}]} assert _extract_title(bd) == "Widget" def test_from_a_single_bare_dict(self): bd = {"invention-title": {"@lang": "en", "$": "Widget"}} assert _extract_title(bd) == "Widget" def test_none_when_missing(self): assert _extract_title({}) is None class TestExtractAbstract: def test_joins_paragraphs_from_a_list(self): doc = {"abstract": [{"@lang": "en", "p": [{"$": "Part one."}, {"$": "Part two."}]}]} assert _extract_abstract(doc) == "Part one. Part two." def test_single_paragraph_not_wrapped_in_a_list(self): doc = {"abstract": {"@lang": "en", "p": {"$": "Only paragraph."}}} assert _extract_abstract(doc) == "Only paragraph." def test_none_when_missing(self): assert _extract_abstract({}) is None class TestExtractClassifications: def test_builds_codes_from_the_five_part_symbol(self): bd = { "patent-classifications": { "patent-classification": [ { "section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}, } ] } } codes = _extract_classifications(bd) assert len(codes) == 1 assert codes[0].code == "G06F17/30" assert codes[0].description == "" def test_dedupes_and_skips_incomplete_entries(self): bd = { "patent-classifications": { "patent-classification": [ {"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}}, {"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}}, {"section": {"$": "H"}}, # incomplete - no class/subclass/main-group ] } } codes = _extract_classifications(bd) assert [c.code for c in codes] == ["G06F17/30"] def test_none_when_no_classifications(self): assert _extract_classifications({}) is None class TestFulltextParagraphs: def test_claims_join_claim_text_lines(self): data = { "ops:world-patent-data": { "ftxt:fulltext-documents": { "ftxt:fulltext-document": { "claims": { "@lang": "en", "claim": [ {"claim-text": {"$": "1. A widget."}}, {"claim-text": [{"$": "2. A gadget,"}, {"$": "further comprising a mechanism."}]}, ], } } } } } assert _fulltext_paragraphs(data, "claims") == ( "1. A widget.\n2. A gadget,\nfurther comprising a mechanism." ) def test_description_joins_paragraphs(self): data = { "ops:world-patent-data": { "ftxt:fulltext-documents": { "ftxt:fulltext-document": { "description": {"@lang": "en", "p": [{"$": "Para one."}, {"$": "Para two."}]} } } } } assert _fulltext_paragraphs(data, "description") == "Para one.\nPara two." def test_none_when_document_missing(self): assert _fulltext_paragraphs({"ops:world-patent-data": {}}, "claims") is None @pytest.mark.parametrize( "raw, expected", [ ("widget", 'txt all "widget"'), ("agentic ai", 'txt all "agentic ai"'), ('pn = "US1234"', 'pn = "US1234"'), ("", ""), ('has "quotes"', 'txt all "has quotes"'), ], ) def test_to_cql(raw, expected): assert _to_cql(raw) == expected @pytest.mark.parametrize( "raw, expected", [ ("US11930446B2", "US11930446"), ("EP4760514A1", "EP4760514"), ("US 11930446 B2", "US11930446"), ("US11930446", "US11930446"), ], ) def test_normalize_epodoc(raw, expected): assert _normalize_epodoc(raw) == expected