Spaces:
Running
Running
Download tests/test_ops_parsing.py from OrganizedProgrammers/SERPent: direct link, hf CLI and curl.
- Browser
- Download file 7.32 kB
-
https://huggingface.co/spaces/OrganizedProgrammers/SERPent/resolve/main/tests/test_ops_parsing.py
- Command line
-
hf download hf://spaces/OrganizedProgrammers/SERPent/tests/test_ops_parsing.py
-
curl -L -o test_ops_parsing.py https://huggingface.co/spaces/OrganizedProgrammers/SERPent/resolve/main/tests/test_ops_parsing.py
7.32 kB
| """Unit tests for ops.py's pure JSON-walking helpers. | |
| OPS's XML-to-JSON conversion is the trickiest part of this module: text | |
| nodes become {"$": "..."}, attributes become "@name" keys, and a repeated | |
| XML element becomes a list *unless there's only one*, in which case it stays | |
| a bare dict. Every helper here has to handle both shapes; these tests pin | |
| both for each one. | |
| """ | |
| import pytest | |
| from ops import ( | |
| _as_list, | |
| _text, | |
| _pick_lang, | |
| _epodoc_number, | |
| _extract_title, | |
| _extract_abstract, | |
| _extract_classifications, | |
| _fulltext_paragraphs, | |
| _to_cql, | |
| _normalize_epodoc, | |
| ) | |
| class TestAsList: | |
| def test_none_becomes_empty_list(self): | |
| assert _as_list(None) == [] | |
| def test_single_dict_is_wrapped(self): | |
| assert _as_list({"a": 1}) == [{"a": 1}] | |
| def test_list_passes_through_unchanged(self): | |
| assert _as_list([{"a": 1}, {"a": 2}]) == [{"a": 1}, {"a": 2}] | |
| class TestText: | |
| def test_extracts_dollar_key(self): | |
| assert _text({"$": "hello"}) == "hello" | |
| def test_plain_string_passes_through(self): | |
| assert _text("hello") == "hello" | |
| def test_none_for_anything_else(self): | |
| assert _text(None) is None | |
| assert _text({"@lang": "en"}) is None | |
| class TestPickLang: | |
| def test_picks_requested_language_from_list(self): | |
| nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "en", "$": "Hello"}] | |
| assert _pick_lang(nodes) == {"@lang": "en", "$": "Hello"} | |
| def test_falls_back_to_first_when_language_missing(self): | |
| nodes = [{"@lang": "de", "$": "Hallo"}, {"@lang": "fr", "$": "Bonjour"}] | |
| assert _pick_lang(nodes) == {"@lang": "de", "$": "Hallo"} | |
| def test_single_dict_not_wrapped_in_a_list_still_works(self): | |
| node = {"@lang": "en", "$": "Hello"} | |
| assert _pick_lang(node) == node | |
| def test_empty_input_returns_none(self): | |
| assert _pick_lang(None) is None | |
| assert _pick_lang([]) is None | |
| class TestEpodocNumber: | |
| def test_prefers_the_epodoc_document_id(self): | |
| exchange_doc = { | |
| "@country": "US", | |
| "@doc-number": "11930446", | |
| "bibliographic-data": { | |
| "publication-reference": { | |
| "document-id": [ | |
| {"@document-id-type": "docdb", "doc-number": {"$": "11930446"}}, | |
| {"@document-id-type": "epodoc", "doc-number": {"$": "US11930446"}}, | |
| ] | |
| } | |
| }, | |
| } | |
| assert _epodoc_number(exchange_doc) == "US11930446" | |
| def test_falls_back_to_country_plus_doc_number(self): | |
| exchange_doc = {"@country": "US", "@doc-number": "11930446", "bibliographic-data": {}} | |
| assert _epodoc_number(exchange_doc) == "US11930446" | |
| def test_none_when_nothing_available(self): | |
| assert _epodoc_number({"bibliographic-data": {}}) is None | |
| class TestExtractTitle: | |
| def test_from_a_list_of_languages(self): | |
| bd = {"invention-title": [{"@lang": "en", "$": "Widget"}, {"@lang": "de", "$": "Widget-DE"}]} | |
| assert _extract_title(bd) == "Widget" | |
| def test_from_a_single_bare_dict(self): | |
| bd = {"invention-title": {"@lang": "en", "$": "Widget"}} | |
| assert _extract_title(bd) == "Widget" | |
| def test_none_when_missing(self): | |
| assert _extract_title({}) is None | |
| class TestExtractAbstract: | |
| def test_joins_paragraphs_from_a_list(self): | |
| doc = {"abstract": [{"@lang": "en", "p": [{"$": "Part one."}, {"$": "Part two."}]}]} | |
| assert _extract_abstract(doc) == "Part one. Part two." | |
| def test_single_paragraph_not_wrapped_in_a_list(self): | |
| doc = {"abstract": {"@lang": "en", "p": {"$": "Only paragraph."}}} | |
| assert _extract_abstract(doc) == "Only paragraph." | |
| def test_none_when_missing(self): | |
| assert _extract_abstract({}) is None | |
| class TestExtractClassifications: | |
| def test_builds_codes_from_the_five_part_symbol(self): | |
| bd = { | |
| "patent-classifications": { | |
| "patent-classification": [ | |
| { | |
| "section": {"$": "G"}, | |
| "class": {"$": "06"}, | |
| "subclass": {"$": "F"}, | |
| "main-group": {"$": "17"}, | |
| "subgroup": {"$": "30"}, | |
| } | |
| ] | |
| } | |
| } | |
| codes = _extract_classifications(bd) | |
| assert len(codes) == 1 | |
| assert codes[0].code == "G06F17/30" | |
| assert codes[0].description == "" | |
| def test_dedupes_and_skips_incomplete_entries(self): | |
| bd = { | |
| "patent-classifications": { | |
| "patent-classification": [ | |
| {"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}}, | |
| {"section": {"$": "G"}, "class": {"$": "06"}, "subclass": {"$": "F"}, "main-group": {"$": "17"}, "subgroup": {"$": "30"}}, | |
| {"section": {"$": "H"}}, # incomplete - no class/subclass/main-group | |
| ] | |
| } | |
| } | |
| codes = _extract_classifications(bd) | |
| assert [c.code for c in codes] == ["G06F17/30"] | |
| def test_none_when_no_classifications(self): | |
| assert _extract_classifications({}) is None | |
| class TestFulltextParagraphs: | |
| def test_claims_join_claim_text_lines(self): | |
| data = { | |
| "ops:world-patent-data": { | |
| "ftxt:fulltext-documents": { | |
| "ftxt:fulltext-document": { | |
| "claims": { | |
| "@lang": "en", | |
| "claim": [ | |
| {"claim-text": {"$": "1. A widget."}}, | |
| {"claim-text": [{"$": "2. A gadget,"}, {"$": "further comprising a mechanism."}]}, | |
| ], | |
| } | |
| } | |
| } | |
| } | |
| } | |
| assert _fulltext_paragraphs(data, "claims") == ( | |
| "1. A widget.\n2. A gadget,\nfurther comprising a mechanism." | |
| ) | |
| def test_description_joins_paragraphs(self): | |
| data = { | |
| "ops:world-patent-data": { | |
| "ftxt:fulltext-documents": { | |
| "ftxt:fulltext-document": { | |
| "description": {"@lang": "en", "p": [{"$": "Para one."}, {"$": "Para two."}]} | |
| } | |
| } | |
| } | |
| } | |
| assert _fulltext_paragraphs(data, "description") == "Para one.\nPara two." | |
| def test_none_when_document_missing(self): | |
| assert _fulltext_paragraphs({"ops:world-patent-data": {}}, "claims") is None | |
| def test_to_cql(raw, expected): | |
| assert _to_cql(raw) == expected | |
| def test_normalize_epodoc(raw, expected): | |
| assert _normalize_epodoc(raw) == expected | |