Spaces:
Sleeping
Sleeping
File size: 5,291 Bytes
e5bea2d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 | from pathlib import Path
import pandas as pd
import pytest
import reference_parser as rp
MEDLINE_SAMPLE = """PMID- 12345
TI - A study about something
AB - This is the abstract
continued on next line
AU - Smith J
AU - Doe J
FAU - Smith, John
FAU - Doe, Jane
JT - Journal of Testing
TA - J Test
DP - 2020 Jan
AID - 10.1000/xyz123 [doi]
LA - eng
PT - Journal Article
MH - Testing
OT - keyword1
PMID- 67890
TI - Second study
AB - Another abstract
JT - Journal Two
DP - 2019
"""
RIS_SAMPLE = """TY - JOUR
TI - RIS Study Title
AU - Author, A
AU - Author, B
AB - RIS abstract text
JO - RIS Journal
PY - 2021
DO - 10.2000/ABC
KW - kw1
KW - kw2
ER -
TY - JOUR
TI - RIS Study Two
AB - Another RIS abstract
JO - RIS Journal Two
PY - 2018
DO - https://doi.org/10.3000/def
ER -
"""
# ---------- MEDLINE parsing ----------
def test_parse_medline_text_splits_records_and_joins_continuations():
records = rp.parse_medline_text(MEDLINE_SAMPLE)
assert len(records) == 2
assert records[0]["AB"] == "This is the abstract continued on next line"
assert records[0]["AU"] == ["Smith J", "Doe J"]
def test_normalize_medline_records_extracts_expected_fields():
records = rp.parse_medline_text(MEDLINE_SAMPLE)
rows = rp.normalize_medline_records(records, Path("dummy.txt"))
assert len(rows) == 2
first = rows[0]
assert first["PMID"] == "12345"
assert first["Title"] == "A study about something"
assert first["Authors"] == "Smith J; Doe J"
assert first["FullAuthors"] == "Smith, John; Doe, Jane"
assert first["Journal"] == "Journal of Testing"
assert first["Year"] == "2020"
assert first["DOI"] == "10.1000/xyz123"
assert first["URL"] == "https://pubmed.ncbi.nlm.nih.gov/12345/"
assert first["SourceFormat"] == "MEDLINE"
second = rows[1]
assert second["PMID"] == "67890"
assert second["DOI"] == ""
# ---------- RIS parsing ----------
def test_parse_ris_text_splits_on_er_tag():
records = rp.parse_ris_text(RIS_SAMPLE)
assert len(records) == 2
assert records[0]["TI"] == ["RIS Study Title"]
assert records[0]["AU"] == ["Author, A", "Author, B"]
def test_normalize_ris_records_builds_doi_url():
records = rp.parse_ris_text(RIS_SAMPLE)
rows = rp.normalize_ris_records(records, Path("dummy.ris"))
assert len(rows) == 2
first = rows[0]
assert first["Title"] == "RIS Study Title"
assert first["Authors"] == "Author, A; Author, B"
assert first["DOI"] == "10.2000/ABC"
assert first["URL"] == "https://doi.org/10.2000/ABC"
assert first["Keywords"] == "kw1; kw2"
assert first["SourceFormat"] == "RIS"
second = rows[1]
# Already a full URL DOI value: used as-is, not double-wrapped.
assert second["URL"] == "https://doi.org/10.3000/def"
# ---------- Dedup key ----------
def test_build_dedup_key_prefers_doi_over_pmid_and_title():
row = pd.Series({"DOI": "10.1/ABC", "PMID": "111", "Title": "X", "Year": "2020"})
assert rp.build_dedup_key(row) == "doi:10.1/abc"
def test_build_dedup_key_strips_doi_url_prefix():
row = pd.Series({"DOI": "https://doi.org/10.1/ABC", "PMID": "", "Title": "", "Year": ""})
assert rp.build_dedup_key(row) == "doi:10.1/abc"
def test_build_dedup_key_falls_back_to_pmid():
row = pd.Series({"DOI": "", "PMID": "999", "Title": "X", "Year": "2020"})
assert rp.build_dedup_key(row) == "pmid:999"
def test_build_dedup_key_falls_back_to_title_year():
row = pd.Series({"DOI": "", "PMID": "", "Title": " Some Title ", "Year": "2020"})
assert rp.build_dedup_key(row) == "title_year:some title_2020"
# ---------- End-to-end process_paths ----------
def test_process_paths_dedups_same_doi_across_formats(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text(
"PMID- 1\nTI - Shared Study\nAB - abs\nAID - 10.9/shared [doi]\nDP - 2020\n",
encoding="utf-8",
)
ris_path = tmp_path / "scopus.ris"
ris_path.write_text(
"TY - JOUR\nTI - Shared Study (Scopus copy)\nDO - https://doi.org/10.9/SHARED\nPY - 2020\nER -\n",
encoding="utf-8",
)
df = rp.process_paths([medline_path, ris_path])
assert len(df) == 1
assert "DedupKey" not in df.columns
assert "SourceFormat" not in df.columns
def test_process_paths_keeps_distinct_records(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text(
"PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n\n"
"PMID- 2\nTI - Study Two\nAB - abs2\nDP - 2021\n",
encoding="utf-8",
)
df = rp.process_paths([medline_path])
assert len(df) == 2
def test_process_paths_returns_empty_df_for_missing_file(tmp_path: Path):
df = rp.process_paths([tmp_path / "does_not_exist.txt"])
assert df.empty
def test_process_paths_writes_excel_when_output_path_given(tmp_path: Path):
medline_path = tmp_path / "pubmed.txt"
medline_path.write_text("PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n", encoding="utf-8")
output_path = tmp_path / "out.xlsx"
df = rp.parse_references([medline_path], output_path)
assert output_path.exists()
reloaded = pd.read_excel(output_path, engine="openpyxl")
assert len(reloaded) == len(df) == 1
|