Spaces:
Running
Running
Download tests/test_term_fields.py from TransLegal/grading-answers: direct link, hf CLI and curl.
- Browser
- Download file 7.3 kB
-
https://huggingface.co/spaces/TransLegal/grading-answers/resolve/main/tests/test_term_fields.py
- Command line
-
hf download hf://spaces/TransLegal/grading-answers/tests/test_term_fields.py
-
curl -L -o test_term_fields.py https://huggingface.co/spaces/TransLegal/grading-answers/resolve/main/tests/test_term_fields.py
7.3 kB
| """CSV field-of-law names must resolve, drop out-of-vocab terms, and refill.""" | |
| from pathlib import Path | |
| import term_fields as tf | |
| VOCAB = ( | |
| "Business / Company Law", | |
| "Criminal Law", | |
| "Debtor-Creditor Law", | |
| "Immigration Law", | |
| "Personal Injury / Tort Law", | |
| "Property Law", | |
| ) | |
| def test_aliases_resolve_onto_the_canonical_26() -> None: | |
| assert tf.resolve_csv_field("Criminal law", VOCAB) == "Criminal Law" | |
| assert ( | |
| tf.resolve_csv_field("Business/Company Law", VOCAB) == "Business / Company Law" | |
| ) | |
| assert tf.resolve_csv_field("Company law", VOCAB) == "Business / Company Law" | |
| assert ( | |
| tf.resolve_csv_field("Personal injury/Tort", VOCAB) | |
| == "Personal Injury / Tort Law" | |
| ) | |
| assert ( | |
| tf.resolve_csv_field("Personal injutry/tort law", VOCAB) | |
| == "Personal Injury / Tort Law" | |
| ) | |
| assert tf.resolve_csv_field("Debtor-Creditor", VOCAB) == "Debtor-Creditor Law" | |
| assert tf.resolve_csv_field("Immigration law", VOCAB) == "Immigration Law" | |
| def test_out_of_vocabulary_and_empty_fields_do_not_resolve() -> None: | |
| assert tf.resolve_csv_field("Procedural Law and Evidence", VOCAB) is None | |
| assert tf.resolve_csv_field("Constitutional law", VOCAB) is None | |
| assert tf.resolve_csv_field("Trusts", VOCAB) is None | |
| assert tf.resolve_csv_field("Transportation Law", VOCAB) is None | |
| assert tf.resolve_csv_field("Government and Politics", VOCAB) is None | |
| assert tf.resolve_csv_field("", VOCAB) is None | |
| def test_choose_drops_exclude_out_of_vocab_and_missing_source() -> None: | |
| rows = ( | |
| tf.CsvFieldRow("keep-me", "Criminal law", False), | |
| tf.CsvFieldRow("procedural", "Procedural Law and Evidence", False), | |
| tf.CsvFieldRow("no-source", "Immigration law", False), | |
| tf.CsvFieldRow("active duty", "", True), | |
| tf.CsvFieldRow("fill-a", "Property law", False), | |
| tf.CsvFieldRow("fill-b", "Company law", False), | |
| tf.CsvFieldRow("fill-skip", "Trusts", False), | |
| tf.CsvFieldRow("fill-c", "Debtor-Creditor", False), | |
| ) | |
| source = { | |
| ("keep-me", "Criminal Law"), | |
| ("fill-a", "Property Law"), | |
| ("fill-b", "Business / Company Law"), | |
| ("fill-c", "Debtor-Creditor Law"), | |
| } | |
| chosen = tf.choose_terms_from_csv( | |
| ["keep-me", "procedural", "no-source", "active duty"], | |
| rows, | |
| vocabulary=VOCAB, | |
| source_pairs=source, | |
| target_count=3, | |
| ) | |
| assert chosen.kept == ("keep-me",) | |
| assert chosen.dropped == ("procedural", "no-source", "active duty") | |
| assert chosen.added == ("fill-a", "fill-b") | |
| assert chosen.by_term == { | |
| "keep-me": "Criminal Law", | |
| "fill-a": "Property Law", | |
| "fill-b": "Business / Company Law", | |
| } | |
| def test_prefers_same_concept_over_earlier_csv_term() -> None: | |
| rows = ( | |
| tf.CsvFieldRow("keep", "Criminal law", False), | |
| tf.CsvFieldRow("drop-me", "Trusts", False), | |
| tf.CsvFieldRow("other-concept", "Property law", False), | |
| tf.CsvFieldRow("same-concept", "Property law", False), | |
| ) | |
| source = { | |
| ("keep", "Criminal Law"), | |
| ("other-concept", "Property Law"), | |
| ("same-concept", "Property Law"), | |
| } | |
| chosen = tf.choose_terms_from_csv( | |
| ["keep", "drop-me"], | |
| rows, | |
| vocabulary=VOCAB, | |
| source_pairs=source, | |
| concept_by_term={ | |
| "keep": "A", | |
| "drop-me": "B", | |
| "other-concept": "A", | |
| "same-concept": "B", | |
| }, | |
| target_count=2, | |
| ) | |
| assert chosen.added == ("same-concept",) | |
| discarded = [row.term for row in chosen.actions if row.action == "discarded"] | |
| assert discarded == [] | |
| def test_picks_scarcest_fol_in_the_selected_set() -> None: | |
| rows = ( | |
| tf.CsvFieldRow("keep-criminal", "Criminal law", False), | |
| tf.CsvFieldRow("drop-me", "Trusts", False), | |
| tf.CsvFieldRow("more-criminal", "Criminal law", False), | |
| tf.CsvFieldRow("property", "Property law", False), | |
| ) | |
| concepts = { | |
| "keep-criminal": "X", | |
| "drop-me": "X", | |
| "more-criminal": "X", | |
| "property": "X", | |
| } | |
| source = { | |
| ("keep-criminal", "Criminal Law"), | |
| ("more-criminal", "Criminal Law"), | |
| ("property", "Property Law"), | |
| } | |
| chosen = tf.choose_terms_from_csv( | |
| ["keep-criminal", "drop-me"], | |
| rows, | |
| vocabulary=VOCAB, | |
| source_pairs=source, | |
| concept_by_term=concepts, | |
| target_count=2, | |
| ) | |
| assert chosen.added == ("property",) | |
| discarded = [row.term for row in chosen.actions if row.action == "discarded"] | |
| assert discarded == ["more-criminal"] | |
| def test_falls_back_to_any_concept_when_same_class_is_empty() -> None: | |
| rows = ( | |
| tf.CsvFieldRow("keep", "Criminal law", False), | |
| tf.CsvFieldRow("drop-me", "Trusts", False), | |
| tf.CsvFieldRow("other-concept", "Property law", False), | |
| ) | |
| chosen = tf.choose_terms_from_csv( | |
| ["keep", "drop-me"], | |
| rows, | |
| vocabulary=VOCAB, | |
| source_pairs={("keep", "Criminal Law"), ("other-concept", "Property Law")}, | |
| concept_by_term={"keep": "A", "drop-me": "B", "other-concept": "A"}, | |
| target_count=2, | |
| ) | |
| assert chosen.added == ("other-concept",) | |
| added = next(row for row in chosen.actions if row.action == "added") | |
| assert "no same-concept replacement" in added.reason | |
| def test_seed_terms_keep_first_seen_order() -> None: | |
| assert tf.seed_terms_from_template(["b", "a", "b", "c"]) == ["b", "a", "c"] | |
| def test_read_term_fields_csv_parses_exclude_column(tmp_path: Path) -> None: | |
| path = tmp_path / "fields.csv" | |
| path.write_text( | |
| "\ufeffterm,field of law,\nalpha,Criminal law,\nactive duty,,Exclude this term\n", | |
| encoding="utf-8", | |
| ) | |
| rows = tf.read_term_fields_csv(path) | |
| assert rows[0] == tf.CsvFieldRow("alpha", "Criminal law", False) | |
| assert rows[1].term == "active duty" | |
| assert rows[1].exclude is True | |
| assert rows[1].raw_field == "" | |
| def test_write_reassignment_report_freezes_header_and_adds_a_sortable_table( | |
| tmp_path: Path, | |
| ) -> None: | |
| from openpyxl import load_workbook | |
| actions = ( | |
| tf.TermAction( | |
| "alpha", | |
| "kept", | |
| "Criminal Law", | |
| "Contract Law", | |
| "true", | |
| "Label or designation", | |
| "kept: CSV FoL Contract Law is in vocabulary", | |
| ), | |
| tf.TermAction( | |
| "beta", | |
| "added", | |
| "", | |
| "Property Law", | |
| "", | |
| "Property regime", | |
| "added: same concept as drop", | |
| ), | |
| ) | |
| path = tf.write_reassignment_report(actions, tmp_path / "report.csv") | |
| assert path.suffix == ".xlsx" | |
| book = load_workbook(path) | |
| sheet = book.active | |
| assert sheet is not None | |
| assert [cell.value for cell in sheet[1]] == [ | |
| "Term", | |
| "Action", | |
| "Previous field of law", | |
| "New field of law", | |
| "Field of law changed", | |
| "Concept class", | |
| "Reason", | |
| ] | |
| assert sheet["B2"].value == "Kept" | |
| assert sheet["E2"].value == "Yes" | |
| assert sheet["E3"].value in (None, "") | |
| assert sheet.freeze_panes == "A2" | |
| table = next(iter(sheet.tables.values())) | |
| assert table.ref == "A1:G3" | |