File size: 6,398 Bytes
bae15d1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 | """Tests for the Excel -> JSON pipeline (src/data/processor.py).
The processor rewrites data/processed/majors.json, which the whole app reads,
and several of its rules are documented responses to real bugs. These cover
the pure functions; the integration guarantee is that regenerating on the
current spreadsheets produces no diff.
"""
import pytest
from data.processor import (
extract_governorate_from_branch,
merge_data,
parse_holland_codes,
parse_languages,
parse_secondary_tracks,
)
# --- Holland codes ---------------------------------------------------------
def test_well_formed_triplet_passes_through():
codes = parse_holland_codes("S", "I", "R", "SIR")
assert codes == {"primary": "S", "secondary": "I", "tertiary": "R", "triplet": "SIR"}
def test_repeated_letter_is_repaired_from_column_20():
"""الطب is recorded S/I/S, which leaves one slot unmatchable."""
codes = parse_holland_codes("S", "I", "S", "SIR")
assert codes["triplet"] == "SIR"
def test_blank_tertiary_is_not_overwritten_by_column_20(capsys):
"""Column 20 disagrees with 17-19 on seven rows; only the repeated-letter
shape may fall back to it."""
codes = parse_holland_codes("S", "I", "", "ECS")
assert codes["primary"] == "S" and codes["secondary"] == "I"
assert codes["triplet"] != "ECS"
assert "WARNING" in capsys.readouterr().out
def test_unrepairable_codes_keep_the_individual_columns(capsys):
codes = parse_holland_codes("S", "I", "S", "")
assert codes["triplet"] == "SIS"
assert "WARNING" in capsys.readouterr().out
# --- Tracks and languages --------------------------------------------------
def test_economics_track_is_not_duplicated():
"""Both "اقتصاد" and "اقتصاد واجتماع" map to the same track."""
assert parse_secondary_tracks("اقتصاد واجتماع") == ["اقتصاد واجتماع"]
def test_multiple_tracks_are_parsed():
tracks = parse_secondary_tracks("علوم عامة - علوم حياة")
assert set(tracks) == {"علوم عامة", "علوم حياة"}
def test_parse_languages_reads_presence_per_column():
langs = parse_languages({"arabic": "x", "english": None, "french": "y", "other_lang": None})
assert langs == ["العربية", "الفرنسية"]
# --- Branch address -> governorate ----------------------------------------
def test_district_name_resolves_to_its_governorate():
assert extract_governorate_from_branch("طرابلس") == "الشمال"
def test_longer_keyword_wins():
assert extract_governorate_from_branch("البقاع الغربي") == "البقاع"
def test_locality_beats_a_governorate_name_in_the_same_address():
""""الحدث – بيروت" is a Mount Lebanon campus; first-match ordering used to
file it under Beirut."""
assert extract_governorate_from_branch("الحدث - قرب بيروت") == "جبل لبنان"
def test_districts_missing_from_the_old_hand_map_now_resolve():
for district, gov in [("جزين", "الجنوب"), ("بشري", "الشمال"),
("مرجعيون", "النبطية"), ("راشيا", "البقاع")]:
assert extract_governorate_from_branch(f"فرع {district}") == gov
def test_unknown_address_returns_none():
assert extract_governorate_from_branch("عنوان غير معروف") is None
# --- merge_data ------------------------------------------------------------
def _major(**kw):
base = dict(id=1, name_ar="تجريبي",
holland_codes={"primary": "R", "secondary": "I",
"tertiary": "A", "triplet": "RIA"},
teaching_languages=["العربية"], locations=[])
base.update(kw)
return base
def test_merge_never_overwrites_holland_codes():
majors = merge_data([_major()], {1: {"holland_primary": "S", "holland_secondary": "E",
"holland_tertiary": "C", "holland_triplet": "SEC"}})
assert majors[0]["holland_codes"]["triplet"] == "RIA"
def test_merge_adds_branches_only_when_the_major_has_none():
majors = merge_data([_major(locations=[])],
{1: {"branches": ["فرع صيدا"]}})
assert majors[0]["locations"][0]["governorate"] == "الجنوب"
def test_merge_leaves_existing_branches_alone():
existing = [{"governorate": "بيروت", "district": "بيروت", "branch": "فرع بيروت"}]
majors = merge_data([_major(locations=list(existing))],
{1: {"branches": ["فرع صيدا"]}})
assert majors[0]["locations"] == existing
def test_language_code_is_case_insensitive():
majors = merge_data([_major(teaching_languages=[])], {1: {"teaching_lang_code": "ARA/ENG"}})
assert majors[0]["teaching_languages"] == ["العربية", "الانكليزية"]
def test_supplementary_language_code_replaces_the_main_sheet():
"""The supplementary sheet exists to correct the main one, so "Eng" there
means English only. Merging the two instead changes 6 majors' languages —
verified by regenerating — so replacement is deliberate, not an oversight."""
majors = merge_data([_major(teaching_languages=["العربية", "أرمني"])],
{1: {"teaching_lang_code": "Eng"}})
assert majors[0]["teaching_languages"] == ["الانكليزية"]
def test_unrecognised_language_code_warns_and_keeps_the_main_sheet(capsys):
majors = merge_data([_major(teaching_languages=["العربية"])],
{1: {"teaching_lang_code": "XYZ"}})
assert majors[0]["teaching_languages"] == ["العربية"]
assert "WARNING" in capsys.readouterr().out
# --- Dataset integrity -----------------------------------------------------
def test_committed_major_ids_are_unique():
from data.loader import load_majors
ids = [m["id"] for m in load_majors()]
assert len(ids) == len(set(ids))
def test_committed_locations_index_covers_every_major():
"""locations.json is keyed by id; a collision would silently drop a major.
The id-46 duplicate is reassigned, so this must stay 1:1."""
import json
from config.settings import LOCATIONS_JSON_FILE
from data.loader import load_majors
index = json.loads(LOCATIONS_JSON_FILE.read_text(encoding="utf-8"))["major_locations"]
assert len(index) == len(load_majors())
|