Spaces:
Sleeping
Sleeping
diogo.rodrigues.silva
Claude Sonnet 5
Harden screening pipeline and add multi-user access for group use
e5bea2d Download tests/test_reference_parser.py from Heit39/LLM_Screener: direct link, hf CLI and curl.
- Browser
- Download file 5.29 kB
-
https://huggingface.co/spaces/Heit39/LLM_Screener/resolve/main/tests/test_reference_parser.py
- Command line
-
hf download hf://spaces/Heit39/LLM_Screener/tests/test_reference_parser.py
-
curl -L -o test_reference_parser.py https://huggingface.co/spaces/Heit39/LLM_Screener/resolve/main/tests/test_reference_parser.py
5.29 kB
| from pathlib import Path | |
| import pandas as pd | |
| import pytest | |
| import reference_parser as rp | |
| MEDLINE_SAMPLE = """PMID- 12345 | |
| TI - A study about something | |
| AB - This is the abstract | |
| continued on next line | |
| AU - Smith J | |
| AU - Doe J | |
| FAU - Smith, John | |
| FAU - Doe, Jane | |
| JT - Journal of Testing | |
| TA - J Test | |
| DP - 2020 Jan | |
| AID - 10.1000/xyz123 [doi] | |
| LA - eng | |
| PT - Journal Article | |
| MH - Testing | |
| OT - keyword1 | |
| PMID- 67890 | |
| TI - Second study | |
| AB - Another abstract | |
| JT - Journal Two | |
| DP - 2019 | |
| """ | |
| RIS_SAMPLE = """TY - JOUR | |
| TI - RIS Study Title | |
| AU - Author, A | |
| AU - Author, B | |
| AB - RIS abstract text | |
| JO - RIS Journal | |
| PY - 2021 | |
| DO - 10.2000/ABC | |
| KW - kw1 | |
| KW - kw2 | |
| ER - | |
| TY - JOUR | |
| TI - RIS Study Two | |
| AB - Another RIS abstract | |
| JO - RIS Journal Two | |
| PY - 2018 | |
| DO - https://doi.org/10.3000/def | |
| ER - | |
| """ | |
| # ---------- MEDLINE parsing ---------- | |
| def test_parse_medline_text_splits_records_and_joins_continuations(): | |
| records = rp.parse_medline_text(MEDLINE_SAMPLE) | |
| assert len(records) == 2 | |
| assert records[0]["AB"] == "This is the abstract continued on next line" | |
| assert records[0]["AU"] == ["Smith J", "Doe J"] | |
| def test_normalize_medline_records_extracts_expected_fields(): | |
| records = rp.parse_medline_text(MEDLINE_SAMPLE) | |
| rows = rp.normalize_medline_records(records, Path("dummy.txt")) | |
| assert len(rows) == 2 | |
| first = rows[0] | |
| assert first["PMID"] == "12345" | |
| assert first["Title"] == "A study about something" | |
| assert first["Authors"] == "Smith J; Doe J" | |
| assert first["FullAuthors"] == "Smith, John; Doe, Jane" | |
| assert first["Journal"] == "Journal of Testing" | |
| assert first["Year"] == "2020" | |
| assert first["DOI"] == "10.1000/xyz123" | |
| assert first["URL"] == "https://pubmed.ncbi.nlm.nih.gov/12345/" | |
| assert first["SourceFormat"] == "MEDLINE" | |
| second = rows[1] | |
| assert second["PMID"] == "67890" | |
| assert second["DOI"] == "" | |
| # ---------- RIS parsing ---------- | |
| def test_parse_ris_text_splits_on_er_tag(): | |
| records = rp.parse_ris_text(RIS_SAMPLE) | |
| assert len(records) == 2 | |
| assert records[0]["TI"] == ["RIS Study Title"] | |
| assert records[0]["AU"] == ["Author, A", "Author, B"] | |
| def test_normalize_ris_records_builds_doi_url(): | |
| records = rp.parse_ris_text(RIS_SAMPLE) | |
| rows = rp.normalize_ris_records(records, Path("dummy.ris")) | |
| assert len(rows) == 2 | |
| first = rows[0] | |
| assert first["Title"] == "RIS Study Title" | |
| assert first["Authors"] == "Author, A; Author, B" | |
| assert first["DOI"] == "10.2000/ABC" | |
| assert first["URL"] == "https://doi.org/10.2000/ABC" | |
| assert first["Keywords"] == "kw1; kw2" | |
| assert first["SourceFormat"] == "RIS" | |
| second = rows[1] | |
| # Already a full URL DOI value: used as-is, not double-wrapped. | |
| assert second["URL"] == "https://doi.org/10.3000/def" | |
| # ---------- Dedup key ---------- | |
| def test_build_dedup_key_prefers_doi_over_pmid_and_title(): | |
| row = pd.Series({"DOI": "10.1/ABC", "PMID": "111", "Title": "X", "Year": "2020"}) | |
| assert rp.build_dedup_key(row) == "doi:10.1/abc" | |
| def test_build_dedup_key_strips_doi_url_prefix(): | |
| row = pd.Series({"DOI": "https://doi.org/10.1/ABC", "PMID": "", "Title": "", "Year": ""}) | |
| assert rp.build_dedup_key(row) == "doi:10.1/abc" | |
| def test_build_dedup_key_falls_back_to_pmid(): | |
| row = pd.Series({"DOI": "", "PMID": "999", "Title": "X", "Year": "2020"}) | |
| assert rp.build_dedup_key(row) == "pmid:999" | |
| def test_build_dedup_key_falls_back_to_title_year(): | |
| row = pd.Series({"DOI": "", "PMID": "", "Title": " Some Title ", "Year": "2020"}) | |
| assert rp.build_dedup_key(row) == "title_year:some title_2020" | |
| # ---------- End-to-end process_paths ---------- | |
| def test_process_paths_dedups_same_doi_across_formats(tmp_path: Path): | |
| medline_path = tmp_path / "pubmed.txt" | |
| medline_path.write_text( | |
| "PMID- 1\nTI - Shared Study\nAB - abs\nAID - 10.9/shared [doi]\nDP - 2020\n", | |
| encoding="utf-8", | |
| ) | |
| ris_path = tmp_path / "scopus.ris" | |
| ris_path.write_text( | |
| "TY - JOUR\nTI - Shared Study (Scopus copy)\nDO - https://doi.org/10.9/SHARED\nPY - 2020\nER -\n", | |
| encoding="utf-8", | |
| ) | |
| df = rp.process_paths([medline_path, ris_path]) | |
| assert len(df) == 1 | |
| assert "DedupKey" not in df.columns | |
| assert "SourceFormat" not in df.columns | |
| def test_process_paths_keeps_distinct_records(tmp_path: Path): | |
| medline_path = tmp_path / "pubmed.txt" | |
| medline_path.write_text( | |
| "PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n\n" | |
| "PMID- 2\nTI - Study Two\nAB - abs2\nDP - 2021\n", | |
| encoding="utf-8", | |
| ) | |
| df = rp.process_paths([medline_path]) | |
| assert len(df) == 2 | |
| def test_process_paths_returns_empty_df_for_missing_file(tmp_path: Path): | |
| df = rp.process_paths([tmp_path / "does_not_exist.txt"]) | |
| assert df.empty | |
| def test_process_paths_writes_excel_when_output_path_given(tmp_path: Path): | |
| medline_path = tmp_path / "pubmed.txt" | |
| medline_path.write_text("PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n", encoding="utf-8") | |
| output_path = tmp_path / "out.xlsx" | |
| df = rp.parse_references([medline_path], output_path) | |
| assert output_path.exists() | |
| reloaded = pd.read_excel(output_path, engine="openpyxl") | |
| assert len(reloaded) == len(df) == 1 | |