File size: 5,291 Bytes
e5bea2d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
from pathlib import Path

import pandas as pd
import pytest

import reference_parser as rp


MEDLINE_SAMPLE = """PMID- 12345
TI  - A study about something
AB  - This is the abstract
      continued on next line
AU  - Smith J
AU  - Doe J
FAU - Smith, John
FAU - Doe, Jane
JT  - Journal of Testing
TA  - J Test
DP  - 2020 Jan
AID - 10.1000/xyz123 [doi]
LA  - eng
PT  - Journal Article
MH  - Testing
OT  - keyword1

PMID- 67890
TI  - Second study
AB  - Another abstract
JT  - Journal Two
DP  - 2019
"""

RIS_SAMPLE = """TY  - JOUR
TI  - RIS Study Title
AU  - Author, A
AU  - Author, B
AB  - RIS abstract text
JO  - RIS Journal
PY  - 2021
DO  - 10.2000/ABC
KW  - kw1
KW  - kw2
ER  -

TY  - JOUR
TI  - RIS Study Two
AB  - Another RIS abstract
JO  - RIS Journal Two
PY  - 2018
DO  - https://doi.org/10.3000/def
ER  -
"""


# ---------- MEDLINE parsing ----------


def test_parse_medline_text_splits_records_and_joins_continuations():
    records = rp.parse_medline_text(MEDLINE_SAMPLE)
    assert len(records) == 2
    assert records[0]["AB"] == "This is the abstract continued on next line"
    assert records[0]["AU"] == ["Smith J", "Doe J"]


def test_normalize_medline_records_extracts_expected_fields():
    records = rp.parse_medline_text(MEDLINE_SAMPLE)
    rows = rp.normalize_medline_records(records, Path("dummy.txt"))
    assert len(rows) == 2

    first = rows[0]
    assert first["PMID"] == "12345"
    assert first["Title"] == "A study about something"
    assert first["Authors"] == "Smith J; Doe J"
    assert first["FullAuthors"] == "Smith, John; Doe, Jane"
    assert first["Journal"] == "Journal of Testing"
    assert first["Year"] == "2020"
    assert first["DOI"] == "10.1000/xyz123"
    assert first["URL"] == "https://pubmed.ncbi.nlm.nih.gov/12345/"
    assert first["SourceFormat"] == "MEDLINE"

    second = rows[1]
    assert second["PMID"] == "67890"
    assert second["DOI"] == ""


# ---------- RIS parsing ----------


def test_parse_ris_text_splits_on_er_tag():
    records = rp.parse_ris_text(RIS_SAMPLE)
    assert len(records) == 2
    assert records[0]["TI"] == ["RIS Study Title"]
    assert records[0]["AU"] == ["Author, A", "Author, B"]


def test_normalize_ris_records_builds_doi_url():
    records = rp.parse_ris_text(RIS_SAMPLE)
    rows = rp.normalize_ris_records(records, Path("dummy.ris"))
    assert len(rows) == 2

    first = rows[0]
    assert first["Title"] == "RIS Study Title"
    assert first["Authors"] == "Author, A; Author, B"
    assert first["DOI"] == "10.2000/ABC"
    assert first["URL"] == "https://doi.org/10.2000/ABC"
    assert first["Keywords"] == "kw1; kw2"
    assert first["SourceFormat"] == "RIS"

    second = rows[1]
    # Already a full URL DOI value: used as-is, not double-wrapped.
    assert second["URL"] == "https://doi.org/10.3000/def"


# ---------- Dedup key ----------


def test_build_dedup_key_prefers_doi_over_pmid_and_title():
    row = pd.Series({"DOI": "10.1/ABC", "PMID": "111", "Title": "X", "Year": "2020"})
    assert rp.build_dedup_key(row) == "doi:10.1/abc"


def test_build_dedup_key_strips_doi_url_prefix():
    row = pd.Series({"DOI": "https://doi.org/10.1/ABC", "PMID": "", "Title": "", "Year": ""})
    assert rp.build_dedup_key(row) == "doi:10.1/abc"


def test_build_dedup_key_falls_back_to_pmid():
    row = pd.Series({"DOI": "", "PMID": "999", "Title": "X", "Year": "2020"})
    assert rp.build_dedup_key(row) == "pmid:999"


def test_build_dedup_key_falls_back_to_title_year():
    row = pd.Series({"DOI": "", "PMID": "", "Title": "  Some   Title  ", "Year": "2020"})
    assert rp.build_dedup_key(row) == "title_year:some title_2020"


# ---------- End-to-end process_paths ----------


def test_process_paths_dedups_same_doi_across_formats(tmp_path: Path):
    medline_path = tmp_path / "pubmed.txt"
    medline_path.write_text(
        "PMID- 1\nTI  - Shared Study\nAB  - abs\nAID - 10.9/shared [doi]\nDP  - 2020\n",
        encoding="utf-8",
    )
    ris_path = tmp_path / "scopus.ris"
    ris_path.write_text(
        "TY  - JOUR\nTI  - Shared Study (Scopus copy)\nDO  - https://doi.org/10.9/SHARED\nPY  - 2020\nER  -\n",
        encoding="utf-8",
    )

    df = rp.process_paths([medline_path, ris_path])
    assert len(df) == 1
    assert "DedupKey" not in df.columns
    assert "SourceFormat" not in df.columns


def test_process_paths_keeps_distinct_records(tmp_path: Path):
    medline_path = tmp_path / "pubmed.txt"
    medline_path.write_text(
        "PMID- 1\nTI  - Study One\nAB  - abs1\nDP  - 2020\n\n"
        "PMID- 2\nTI  - Study Two\nAB  - abs2\nDP  - 2021\n",
        encoding="utf-8",
    )

    df = rp.process_paths([medline_path])
    assert len(df) == 2


def test_process_paths_returns_empty_df_for_missing_file(tmp_path: Path):
    df = rp.process_paths([tmp_path / "does_not_exist.txt"])
    assert df.empty


def test_process_paths_writes_excel_when_output_path_given(tmp_path: Path):
    medline_path = tmp_path / "pubmed.txt"
    medline_path.write_text("PMID- 1\nTI  - Study One\nAB  - abs1\nDP  - 2020\n", encoding="utf-8")
    output_path = tmp_path / "out.xlsx"

    df = rp.parse_references([medline_path], output_path)
    assert output_path.exists()
    reloaded = pd.read_excel(output_path, engine="openpyxl")
    assert len(reloaded) == len(df) == 1