Download tests/test_processor.py from ya02/kalim: direct link, hf CLI and curl.
- Browser
- Download file 6.4 kB
-
https://huggingface.co/spaces/ya02/kalim/resolve/main/tests/test_processor.py
- Command line
-
hf download hf://spaces/ya02/kalim/tests/test_processor.py
-
curl -L -o test_processor.py https://huggingface.co/spaces/ya02/kalim/resolve/main/tests/test_processor.py
6.4 kB
| """Tests for the Excel -> JSON pipeline (src/data/processor.py). | |
| The processor rewrites data/processed/majors.json, which the whole app reads, | |
| and several of its rules are documented responses to real bugs. These cover | |
| the pure functions; the integration guarantee is that regenerating on the | |
| current spreadsheets produces no diff. | |
| """ | |
| import pytest | |
| from data.processor import ( | |
| extract_governorate_from_branch, | |
| merge_data, | |
| parse_holland_codes, | |
| parse_languages, | |
| parse_secondary_tracks, | |
| ) | |
| # --- Holland codes --------------------------------------------------------- | |
| def test_well_formed_triplet_passes_through(): | |
| codes = parse_holland_codes("S", "I", "R", "SIR") | |
| assert codes == {"primary": "S", "secondary": "I", "tertiary": "R", "triplet": "SIR"} | |
| def test_repeated_letter_is_repaired_from_column_20(): | |
| """الطب is recorded S/I/S, which leaves one slot unmatchable.""" | |
| codes = parse_holland_codes("S", "I", "S", "SIR") | |
| assert codes["triplet"] == "SIR" | |
| def test_blank_tertiary_is_not_overwritten_by_column_20(capsys): | |
| """Column 20 disagrees with 17-19 on seven rows; only the repeated-letter | |
| shape may fall back to it.""" | |
| codes = parse_holland_codes("S", "I", "", "ECS") | |
| assert codes["primary"] == "S" and codes["secondary"] == "I" | |
| assert codes["triplet"] != "ECS" | |
| assert "WARNING" in capsys.readouterr().out | |
| def test_unrepairable_codes_keep_the_individual_columns(capsys): | |
| codes = parse_holland_codes("S", "I", "S", "") | |
| assert codes["triplet"] == "SIS" | |
| assert "WARNING" in capsys.readouterr().out | |
| # --- Tracks and languages -------------------------------------------------- | |
| def test_economics_track_is_not_duplicated(): | |
| """Both "اقتصاد" and "اقتصاد واجتماع" map to the same track.""" | |
| assert parse_secondary_tracks("اقتصاد واجتماع") == ["اقتصاد واجتماع"] | |
| def test_multiple_tracks_are_parsed(): | |
| tracks = parse_secondary_tracks("علوم عامة - علوم حياة") | |
| assert set(tracks) == {"علوم عامة", "علوم حياة"} | |
| def test_parse_languages_reads_presence_per_column(): | |
| langs = parse_languages({"arabic": "x", "english": None, "french": "y", "other_lang": None}) | |
| assert langs == ["العربية", "الفرنسية"] | |
| # --- Branch address -> governorate ---------------------------------------- | |
| def test_district_name_resolves_to_its_governorate(): | |
| assert extract_governorate_from_branch("طرابلس") == "الشمال" | |
| def test_longer_keyword_wins(): | |
| assert extract_governorate_from_branch("البقاع الغربي") == "البقاع" | |
| def test_locality_beats_a_governorate_name_in_the_same_address(): | |
| """"الحدث – بيروت" is a Mount Lebanon campus; first-match ordering used to | |
| file it under Beirut.""" | |
| assert extract_governorate_from_branch("الحدث - قرب بيروت") == "جبل لبنان" | |
| def test_districts_missing_from_the_old_hand_map_now_resolve(): | |
| for district, gov in [("جزين", "الجنوب"), ("بشري", "الشمال"), | |
| ("مرجعيون", "النبطية"), ("راشيا", "البقاع")]: | |
| assert extract_governorate_from_branch(f"فرع {district}") == gov | |
| def test_unknown_address_returns_none(): | |
| assert extract_governorate_from_branch("عنوان غير معروف") is None | |
| # --- merge_data ------------------------------------------------------------ | |
| def _major(**kw): | |
| base = dict(id=1, name_ar="تجريبي", | |
| holland_codes={"primary": "R", "secondary": "I", | |
| "tertiary": "A", "triplet": "RIA"}, | |
| teaching_languages=["العربية"], locations=[]) | |
| base.update(kw) | |
| return base | |
| def test_merge_never_overwrites_holland_codes(): | |
| majors = merge_data([_major()], {1: {"holland_primary": "S", "holland_secondary": "E", | |
| "holland_tertiary": "C", "holland_triplet": "SEC"}}) | |
| assert majors[0]["holland_codes"]["triplet"] == "RIA" | |
| def test_merge_adds_branches_only_when_the_major_has_none(): | |
| majors = merge_data([_major(locations=[])], | |
| {1: {"branches": ["فرع صيدا"]}}) | |
| assert majors[0]["locations"][0]["governorate"] == "الجنوب" | |
| def test_merge_leaves_existing_branches_alone(): | |
| existing = [{"governorate": "بيروت", "district": "بيروت", "branch": "فرع بيروت"}] | |
| majors = merge_data([_major(locations=list(existing))], | |
| {1: {"branches": ["فرع صيدا"]}}) | |
| assert majors[0]["locations"] == existing | |
| def test_language_code_is_case_insensitive(): | |
| majors = merge_data([_major(teaching_languages=[])], {1: {"teaching_lang_code": "ARA/ENG"}}) | |
| assert majors[0]["teaching_languages"] == ["العربية", "الانكليزية"] | |
| def test_supplementary_language_code_replaces_the_main_sheet(): | |
| """The supplementary sheet exists to correct the main one, so "Eng" there | |
| means English only. Merging the two instead changes 6 majors' languages — | |
| verified by regenerating — so replacement is deliberate, not an oversight.""" | |
| majors = merge_data([_major(teaching_languages=["العربية", "أرمني"])], | |
| {1: {"teaching_lang_code": "Eng"}}) | |
| assert majors[0]["teaching_languages"] == ["الانكليزية"] | |
| def test_unrecognised_language_code_warns_and_keeps_the_main_sheet(capsys): | |
| majors = merge_data([_major(teaching_languages=["العربية"])], | |
| {1: {"teaching_lang_code": "XYZ"}}) | |
| assert majors[0]["teaching_languages"] == ["العربية"] | |
| assert "WARNING" in capsys.readouterr().out | |
| # --- Dataset integrity ----------------------------------------------------- | |
| def test_committed_major_ids_are_unique(): | |
| from data.loader import load_majors | |
| ids = [m["id"] for m in load_majors()] | |
| assert len(ids) == len(set(ids)) | |
| def test_committed_locations_index_covers_every_major(): | |
| """locations.json is keyed by id; a collision would silently drop a major. | |
| The id-46 duplicate is reassigned, so this must stay 1:1.""" | |
| import json | |
| from config.settings import LOCATIONS_JSON_FILE | |
| from data.loader import load_majors | |
| index = json.loads(LOCATIONS_JSON_FILE.read_text(encoding="utf-8"))["major_locations"] | |
| assert len(index) == len(load_majors()) | |