kalim / tests /test_processor.py
YasserHaidar
Kalim Space deploy (e803a10)
bae15d1
Raw History Blame Contribute Delete
6.4 kB
"""Tests for the Excel -> JSON pipeline (src/data/processor.py).
The processor rewrites data/processed/majors.json, which the whole app reads,
and several of its rules are documented responses to real bugs. These cover
the pure functions; the integration guarantee is that regenerating on the
current spreadsheets produces no diff.
"""
import pytest
from data.processor import (
extract_governorate_from_branch,
merge_data,
parse_holland_codes,
parse_languages,
parse_secondary_tracks,
)
# --- Holland codes ---------------------------------------------------------
def test_well_formed_triplet_passes_through():
codes = parse_holland_codes("S", "I", "R", "SIR")
assert codes == {"primary": "S", "secondary": "I", "tertiary": "R", "triplet": "SIR"}
def test_repeated_letter_is_repaired_from_column_20():
"""الطب is recorded S/I/S, which leaves one slot unmatchable."""
codes = parse_holland_codes("S", "I", "S", "SIR")
assert codes["triplet"] == "SIR"
def test_blank_tertiary_is_not_overwritten_by_column_20(capsys):
"""Column 20 disagrees with 17-19 on seven rows; only the repeated-letter
shape may fall back to it."""
codes = parse_holland_codes("S", "I", "", "ECS")
assert codes["primary"] == "S" and codes["secondary"] == "I"
assert codes["triplet"] != "ECS"
assert "WARNING" in capsys.readouterr().out
def test_unrepairable_codes_keep_the_individual_columns(capsys):
codes = parse_holland_codes("S", "I", "S", "")
assert codes["triplet"] == "SIS"
assert "WARNING" in capsys.readouterr().out
# --- Tracks and languages --------------------------------------------------
def test_economics_track_is_not_duplicated():
"""Both "اقتصاد" and "اقتصاد واجتماع" map to the same track."""
assert parse_secondary_tracks("اقتصاد واجتماع") == ["اقتصاد واجتماع"]
def test_multiple_tracks_are_parsed():
tracks = parse_secondary_tracks("علوم عامة - علوم حياة")
assert set(tracks) == {"علوم عامة", "علوم حياة"}
def test_parse_languages_reads_presence_per_column():
langs = parse_languages({"arabic": "x", "english": None, "french": "y", "other_lang": None})
assert langs == ["العربية", "الفرنسية"]
# --- Branch address -> governorate ----------------------------------------
def test_district_name_resolves_to_its_governorate():
assert extract_governorate_from_branch("طرابلس") == "الشمال"
def test_longer_keyword_wins():
assert extract_governorate_from_branch("البقاع الغربي") == "البقاع"
def test_locality_beats_a_governorate_name_in_the_same_address():
""""الحدث – بيروت" is a Mount Lebanon campus; first-match ordering used to
file it under Beirut."""
assert extract_governorate_from_branch("الحدث - قرب بيروت") == "جبل لبنان"
def test_districts_missing_from_the_old_hand_map_now_resolve():
for district, gov in [("جزين", "الجنوب"), ("بشري", "الشمال"),
("مرجعيون", "النبطية"), ("راشيا", "البقاع")]:
assert extract_governorate_from_branch(f"فرع {district}") == gov
def test_unknown_address_returns_none():
assert extract_governorate_from_branch("عنوان غير معروف") is None
# --- merge_data ------------------------------------------------------------
def _major(**kw):
base = dict(id=1, name_ar="تجريبي",
holland_codes={"primary": "R", "secondary": "I",
"tertiary": "A", "triplet": "RIA"},
teaching_languages=["العربية"], locations=[])
base.update(kw)
return base
def test_merge_never_overwrites_holland_codes():
majors = merge_data([_major()], {1: {"holland_primary": "S", "holland_secondary": "E",
"holland_tertiary": "C", "holland_triplet": "SEC"}})
assert majors[0]["holland_codes"]["triplet"] == "RIA"
def test_merge_adds_branches_only_when_the_major_has_none():
majors = merge_data([_major(locations=[])],
{1: {"branches": ["فرع صيدا"]}})
assert majors[0]["locations"][0]["governorate"] == "الجنوب"
def test_merge_leaves_existing_branches_alone():
existing = [{"governorate": "بيروت", "district": "بيروت", "branch": "فرع بيروت"}]
majors = merge_data([_major(locations=list(existing))],
{1: {"branches": ["فرع صيدا"]}})
assert majors[0]["locations"] == existing
def test_language_code_is_case_insensitive():
majors = merge_data([_major(teaching_languages=[])], {1: {"teaching_lang_code": "ARA/ENG"}})
assert majors[0]["teaching_languages"] == ["العربية", "الانكليزية"]
def test_supplementary_language_code_replaces_the_main_sheet():
"""The supplementary sheet exists to correct the main one, so "Eng" there
means English only. Merging the two instead changes 6 majors' languages —
verified by regenerating — so replacement is deliberate, not an oversight."""
majors = merge_data([_major(teaching_languages=["العربية", "أرمني"])],
{1: {"teaching_lang_code": "Eng"}})
assert majors[0]["teaching_languages"] == ["الانكليزية"]
def test_unrecognised_language_code_warns_and_keeps_the_main_sheet(capsys):
majors = merge_data([_major(teaching_languages=["العربية"])],
{1: {"teaching_lang_code": "XYZ"}})
assert majors[0]["teaching_languages"] == ["العربية"]
assert "WARNING" in capsys.readouterr().out
# --- Dataset integrity -----------------------------------------------------
def test_committed_major_ids_are_unique():
from data.loader import load_majors
ids = [m["id"] for m in load_majors()]
assert len(ids) == len(set(ids))
def test_committed_locations_index_covers_every_major():
"""locations.json is keyed by id; a collision would silently drop a major.
The id-46 duplicate is reassigned, so this must stay 1:1."""
import json
from config.settings import LOCATIONS_JSON_FILE
from data.loader import load_majors
index = json.loads(LOCATIONS_JSON_FILE.read_text(encoding="utf-8"))["major_locations"]
assert len(index) == len(load_majors())