from pathlib import Path import pandas as pd import pytest import reference_parser as rp MEDLINE_SAMPLE = """PMID- 12345 TI - A study about something AB - This is the abstract continued on next line AU - Smith J AU - Doe J FAU - Smith, John FAU - Doe, Jane JT - Journal of Testing TA - J Test DP - 2020 Jan AID - 10.1000/xyz123 [doi] LA - eng PT - Journal Article MH - Testing OT - keyword1 PMID- 67890 TI - Second study AB - Another abstract JT - Journal Two DP - 2019 """ RIS_SAMPLE = """TY - JOUR TI - RIS Study Title AU - Author, A AU - Author, B AB - RIS abstract text JO - RIS Journal PY - 2021 DO - 10.2000/ABC KW - kw1 KW - kw2 ER - TY - JOUR TI - RIS Study Two AB - Another RIS abstract JO - RIS Journal Two PY - 2018 DO - https://doi.org/10.3000/def ER - """ # ---------- MEDLINE parsing ---------- def test_parse_medline_text_splits_records_and_joins_continuations(): records = rp.parse_medline_text(MEDLINE_SAMPLE) assert len(records) == 2 assert records[0]["AB"] == "This is the abstract continued on next line" assert records[0]["AU"] == ["Smith J", "Doe J"] def test_normalize_medline_records_extracts_expected_fields(): records = rp.parse_medline_text(MEDLINE_SAMPLE) rows = rp.normalize_medline_records(records, Path("dummy.txt")) assert len(rows) == 2 first = rows[0] assert first["PMID"] == "12345" assert first["Title"] == "A study about something" assert first["Authors"] == "Smith J; Doe J" assert first["FullAuthors"] == "Smith, John; Doe, Jane" assert first["Journal"] == "Journal of Testing" assert first["Year"] == "2020" assert first["DOI"] == "10.1000/xyz123" assert first["URL"] == "https://pubmed.ncbi.nlm.nih.gov/12345/" assert first["SourceFormat"] == "MEDLINE" second = rows[1] assert second["PMID"] == "67890" assert second["DOI"] == "" # ---------- RIS parsing ---------- def test_parse_ris_text_splits_on_er_tag(): records = rp.parse_ris_text(RIS_SAMPLE) assert len(records) == 2 assert records[0]["TI"] == ["RIS Study Title"] assert records[0]["AU"] == ["Author, A", "Author, B"] def test_normalize_ris_records_builds_doi_url(): records = rp.parse_ris_text(RIS_SAMPLE) rows = rp.normalize_ris_records(records, Path("dummy.ris")) assert len(rows) == 2 first = rows[0] assert first["Title"] == "RIS Study Title" assert first["Authors"] == "Author, A; Author, B" assert first["DOI"] == "10.2000/ABC" assert first["URL"] == "https://doi.org/10.2000/ABC" assert first["Keywords"] == "kw1; kw2" assert first["SourceFormat"] == "RIS" second = rows[1] # Already a full URL DOI value: used as-is, not double-wrapped. assert second["URL"] == "https://doi.org/10.3000/def" # ---------- Dedup key ---------- def test_build_dedup_key_prefers_doi_over_pmid_and_title(): row = pd.Series({"DOI": "10.1/ABC", "PMID": "111", "Title": "X", "Year": "2020"}) assert rp.build_dedup_key(row) == "doi:10.1/abc" def test_build_dedup_key_strips_doi_url_prefix(): row = pd.Series({"DOI": "https://doi.org/10.1/ABC", "PMID": "", "Title": "", "Year": ""}) assert rp.build_dedup_key(row) == "doi:10.1/abc" def test_build_dedup_key_falls_back_to_pmid(): row = pd.Series({"DOI": "", "PMID": "999", "Title": "X", "Year": "2020"}) assert rp.build_dedup_key(row) == "pmid:999" def test_build_dedup_key_falls_back_to_title_year(): row = pd.Series({"DOI": "", "PMID": "", "Title": " Some Title ", "Year": "2020"}) assert rp.build_dedup_key(row) == "title_year:some title_2020" # ---------- End-to-end process_paths ---------- def test_process_paths_dedups_same_doi_across_formats(tmp_path: Path): medline_path = tmp_path / "pubmed.txt" medline_path.write_text( "PMID- 1\nTI - Shared Study\nAB - abs\nAID - 10.9/shared [doi]\nDP - 2020\n", encoding="utf-8", ) ris_path = tmp_path / "scopus.ris" ris_path.write_text( "TY - JOUR\nTI - Shared Study (Scopus copy)\nDO - https://doi.org/10.9/SHARED\nPY - 2020\nER -\n", encoding="utf-8", ) df = rp.process_paths([medline_path, ris_path]) assert len(df) == 1 assert "DedupKey" not in df.columns assert "SourceFormat" not in df.columns def test_process_paths_keeps_distinct_records(tmp_path: Path): medline_path = tmp_path / "pubmed.txt" medline_path.write_text( "PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n\n" "PMID- 2\nTI - Study Two\nAB - abs2\nDP - 2021\n", encoding="utf-8", ) df = rp.process_paths([medline_path]) assert len(df) == 2 def test_process_paths_returns_empty_df_for_missing_file(tmp_path: Path): df = rp.process_paths([tmp_path / "does_not_exist.txt"]) assert df.empty def test_process_paths_writes_excel_when_output_path_given(tmp_path: Path): medline_path = tmp_path / "pubmed.txt" medline_path.write_text("PMID- 1\nTI - Study One\nAB - abs1\nDP - 2020\n", encoding="utf-8") output_path = tmp_path / "out.xlsx" df = rp.parse_references([medline_path], output_path) assert output_path.exists() reloaded = pd.read_excel(output_path, engine="openpyxl") assert len(reloaded) == len(df) == 1