LLM_Screener / tests /test_reference_parser.py
diogo.rodrigues.silva
Claude Sonnet 5
Harden screening pipeline and add multi-user access for group use
e5bea2d
Raw History Blame
5.29 kB
from pathlib import Path
import pandas as pd
import pytest
import reference_parser as rp
MEDLINE_SAMPLE = """PMID- 12345
TI - A study about something
AB - This is the abstract
continued on next line
AU - Smith J
AU - Doe J
FAU - Smith, John
FAU - Doe, Jane
JT - Journal of Testing
TA - J Test
DP - 2020 Jan
AID - 10.1000/xyz123 [doi]
LA - eng
PT - Journal Article
MH - Testing
OT - keyword1
PMID- 67890
TI - Second study
AB - Another abstract
JT - Journal Two
DP - 2019
"""
RIS_SAMPLE = """TY - JOUR
TI - RIS Study Title
AU - Author, A
AU - Author, B
AB - RIS abstract text
JO - RIS Journal
PY - 2021
DO - 10.2000/ABC
KW - kw1
KW - kw2
ER -
TY - JOUR
TI - RIS Study Two
AB - Another RIS abstract
JO - RIS Journal Two
PY - 2018
DO - https://doi.org/10.3000/def
ER -
"""
# ---------- MEDLINE parsing ----------
def test_parse_medline_text_splits_records_and_joins_continuations():
records = rp.parse_medline_text(MEDLINE_SAMPLE)
assert len(records) == 2
assert records[0]["AB"] == "This is the abstract continued on next line"
assert records[0]["AU"] == ["Smith J", "Doe J"]
def test_normalize_medline_records_extracts_expected_fields():
records = rp.parse_medline_text(MEDLINE_SAMPLE)
rows = rp.normalize_medline_records(records, Path("dummy.txt"))
assert len(rows) == 2
first = rows[0]
assert first["PMID"] == "12345"
assert first["Title"] == "A study about something"
assert first["Authors"] == "Smith J; Doe J"
assert first["FullAuthors"] == "Smith, John; Doe, Jane"
assert first["Journal"] == "Journal of Testing"
assert first["Year"] == "2020"
assert first["DOI"] == "10.1000/xyz123"
assert first["URL"] == "https://pubmed.ncbi.nlm.nih.gov/12345/"
assert first["SourceFormat"] == "MEDLINE"
second = rows[1]
assert second["PMID"] == "67890"
assert second["DOI"] == ""
# ---------- RIS parsing ----------
def test_parse_ris_text_splits_on_er_tag():
records = rp.parse_ris_text(RIS_SAMPLE)
assert len(records) == 2
assert records[0]["TI"] == ["RIS Study Title"]
assert records[0]["AU"] == ["Author, A", "Author, B"]
def test_normalize_ris_records_builds_doi_url():
records = rp.parse_ris_text(RIS_SAMPLE)
rows = rp.normalize_ris_records(records, Path("dummy.ris"))
assert len(rows) == 2
first = rows[0]
assert first["Title"] == "RIS Study Title"
assert first["Authors"] == "Author, A; Author, B"
assert first["DOI"] == "10.2000/ABC"
assert first["URL"] == "https://doi.org/10.2000/ABC"
assert first["Keywords"] == "kw1; kw2"
assert first["SourceFormat"] == "RIS"
second = rows[1]
# Already a full URL DOI value: used as-is, not double-wrapped.
assert second["URL"] == "https://doi.org/10.3000/def"
# ---------- Dedup key ----------
def test_build_dedup_key_prefers_doi_over_pmid_and_title():
row = pd.Series({"DOI": "10.1/ABC", "PMID": "111", "Title": "X", "Year": "2020"})
assert rp.build_dedup_key(row) == "doi:10.1/abc"
def test_build_dedup_key_strips_doi_url_prefix():
row = pd.Series({"DOI": "https://doi.org/10.1/ABC", "PMID": "", "Title": "", "Year": ""})
assert rp.build_dedup_key(row) == "doi:10.1/abc"
def test_build_dedup_key_falls_back_to_pmid():
row = pd.Series({"DOI": "", "PMID": "999", "Title": "X", "Year": "2020"})
assert rp.build_dedup_key(row) == "pmid:999"
def test_build_dedup_key_falls_back_to_title_year():
row = pd.Series({"DOI": "", "PMID": "", "Title": " Some Title ", "Year": "2020"})
assert rp.build_dedup_key(row) == "title_year:some title_2020"
# ---------- End-to-end process_paths ----------
def test_process_paths_dedups_same_doi_across_formats(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text(
"PMID- 1\nTI - Shared Study\nAB - abs\nAID - 10.9/shared [doi]\nDP - 2020\n",
encoding="utf-8",
)
ris_path = tmp_path / "scopus.ris"
ris_path.write_text(
"TY - JOUR\nTI - Shared Study (Scopus copy)\nDO - https://doi.org/10.9/SHARED\nPY - 2020\nER -\n",
encoding="utf-8",
)
df = rp.process_paths([medline_path, ris_path])
assert len(df) == 1
assert "DedupKey" not in df.columns
assert "SourceFormat" not in df.columns
def test_process_paths_keeps_distinct_records(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text(
"PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n\n"
"PMID- 2\nTI - Study Two\nAB - abs2\nDP - 2021\n",
encoding="utf-8",
)
df = rp.process_paths([medline_path])
assert len(df) == 2
def test_process_paths_returns_empty_df_for_missing_file(tmp_path: Path):
df = rp.process_paths([tmp_path / "does_not_exist.txt"])
assert df.empty
def test_process_paths_writes_excel_when_output_path_given(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text("PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n", encoding="utf-8")
output_path = tmp_path / "out.xlsx"
df = rp.parse_references([medline_path], output_path)
assert output_path.exists()
reloaded = pd.read_excel(output_path, engine="openpyxl")
assert len(reloaded) == len(df) == 1