"""Tests for the Excel -> JSON pipeline (src/data/processor.py). The processor rewrites data/processed/majors.json, which the whole app reads, and several of its rules are documented responses to real bugs. These cover the pure functions; the integration guarantee is that regenerating on the current spreadsheets produces no diff. """ import pytest from data.processor import ( extract_governorate_from_branch, merge_data, parse_holland_codes, parse_languages, parse_secondary_tracks, ) # --- Holland codes --------------------------------------------------------- def test_well_formed_triplet_passes_through(): codes = parse_holland_codes("S", "I", "R", "SIR") assert codes == {"primary": "S", "secondary": "I", "tertiary": "R", "triplet": "SIR"} def test_repeated_letter_is_repaired_from_column_20(): """الطب is recorded S/I/S, which leaves one slot unmatchable.""" codes = parse_holland_codes("S", "I", "S", "SIR") assert codes["triplet"] == "SIR" def test_blank_tertiary_is_not_overwritten_by_column_20(capsys): """Column 20 disagrees with 17-19 on seven rows; only the repeated-letter shape may fall back to it.""" codes = parse_holland_codes("S", "I", "", "ECS") assert codes["primary"] == "S" and codes["secondary"] == "I" assert codes["triplet"] != "ECS" assert "WARNING" in capsys.readouterr().out def test_unrepairable_codes_keep_the_individual_columns(capsys): codes = parse_holland_codes("S", "I", "S", "") assert codes["triplet"] == "SIS" assert "WARNING" in capsys.readouterr().out # --- Tracks and languages -------------------------------------------------- def test_economics_track_is_not_duplicated(): """Both "اقتصاد" and "اقتصاد واجتماع" map to the same track.""" assert parse_secondary_tracks("اقتصاد واجتماع") == ["اقتصاد واجتماع"] def test_multiple_tracks_are_parsed(): tracks = parse_secondary_tracks("علوم عامة - علوم حياة") assert set(tracks) == {"علوم عامة", "علوم حياة"} def test_parse_languages_reads_presence_per_column(): langs = parse_languages({"arabic": "x", "english": None, "french": "y", "other_lang": None}) assert langs == ["العربية", "الفرنسية"] # --- Branch address -> governorate ---------------------------------------- def test_district_name_resolves_to_its_governorate(): assert extract_governorate_from_branch("طرابلس") == "الشمال" def test_longer_keyword_wins(): assert extract_governorate_from_branch("البقاع الغربي") == "البقاع" def test_locality_beats_a_governorate_name_in_the_same_address(): """"الحدث – بيروت" is a Mount Lebanon campus; first-match ordering used to file it under Beirut.""" assert extract_governorate_from_branch("الحدث - قرب بيروت") == "جبل لبنان" def test_districts_missing_from_the_old_hand_map_now_resolve(): for district, gov in [("جزين", "الجنوب"), ("بشري", "الشمال"), ("مرجعيون", "النبطية"), ("راشيا", "البقاع")]: assert extract_governorate_from_branch(f"فرع {district}") == gov def test_unknown_address_returns_none(): assert extract_governorate_from_branch("عنوان غير معروف") is None # --- merge_data ------------------------------------------------------------ def _major(**kw): base = dict(id=1, name_ar="تجريبي", holland_codes={"primary": "R", "secondary": "I", "tertiary": "A", "triplet": "RIA"}, teaching_languages=["العربية"], locations=[]) base.update(kw) return base def test_merge_never_overwrites_holland_codes(): majors = merge_data([_major()], {1: {"holland_primary": "S", "holland_secondary": "E", "holland_tertiary": "C", "holland_triplet": "SEC"}}) assert majors[0]["holland_codes"]["triplet"] == "RIA" def test_merge_adds_branches_only_when_the_major_has_none(): majors = merge_data([_major(locations=[])], {1: {"branches": ["فرع صيدا"]}}) assert majors[0]["locations"][0]["governorate"] == "الجنوب" def test_merge_leaves_existing_branches_alone(): existing = [{"governorate": "بيروت", "district": "بيروت", "branch": "فرع بيروت"}] majors = merge_data([_major(locations=list(existing))], {1: {"branches": ["فرع صيدا"]}}) assert majors[0]["locations"] == existing def test_language_code_is_case_insensitive(): majors = merge_data([_major(teaching_languages=[])], {1: {"teaching_lang_code": "ARA/ENG"}}) assert majors[0]["teaching_languages"] == ["العربية", "الانكليزية"] def test_supplementary_language_code_replaces_the_main_sheet(): """The supplementary sheet exists to correct the main one, so "Eng" there means English only. Merging the two instead changes 6 majors' languages — verified by regenerating — so replacement is deliberate, not an oversight.""" majors = merge_data([_major(teaching_languages=["العربية", "أرمني"])], {1: {"teaching_lang_code": "Eng"}}) assert majors[0]["teaching_languages"] == ["الانكليزية"] def test_unrecognised_language_code_warns_and_keeps_the_main_sheet(capsys): majors = merge_data([_major(teaching_languages=["العربية"])], {1: {"teaching_lang_code": "XYZ"}}) assert majors[0]["teaching_languages"] == ["العربية"] assert "WARNING" in capsys.readouterr().out # --- Dataset integrity ----------------------------------------------------- def test_committed_major_ids_are_unique(): from data.loader import load_majors ids = [m["id"] for m in load_majors()] assert len(ids) == len(set(ids)) def test_committed_locations_index_covers_every_major(): """locations.json is keyed by id; a collision would silently drop a major. The id-46 duplicate is reassigned, so this must stay 1:1.""" import json from config.settings import LOCATIONS_JSON_FILE from data.loader import load_majors index = json.loads(LOCATIONS_JSON_FILE.read_text(encoding="utf-8"))["major_locations"] assert len(index) == len(load_majors())