File size: 6,398 Bytes
bae15d1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
"""Tests for the Excel -> JSON pipeline (src/data/processor.py).

The processor rewrites data/processed/majors.json, which the whole app reads,
and several of its rules are documented responses to real bugs. These cover
the pure functions; the integration guarantee is that regenerating on the
current spreadsheets produces no diff.
"""
import pytest

from data.processor import (
    extract_governorate_from_branch,
    merge_data,
    parse_holland_codes,
    parse_languages,
    parse_secondary_tracks,
)


# --- Holland codes ---------------------------------------------------------

def test_well_formed_triplet_passes_through():
    codes = parse_holland_codes("S", "I", "R", "SIR")
    assert codes == {"primary": "S", "secondary": "I", "tertiary": "R", "triplet": "SIR"}


def test_repeated_letter_is_repaired_from_column_20():
    """الطب is recorded S/I/S, which leaves one slot unmatchable."""
    codes = parse_holland_codes("S", "I", "S", "SIR")
    assert codes["triplet"] == "SIR"


def test_blank_tertiary_is_not_overwritten_by_column_20(capsys):
    """Column 20 disagrees with 17-19 on seven rows; only the repeated-letter
    shape may fall back to it."""
    codes = parse_holland_codes("S", "I", "", "ECS")
    assert codes["primary"] == "S" and codes["secondary"] == "I"
    assert codes["triplet"] != "ECS"
    assert "WARNING" in capsys.readouterr().out


def test_unrepairable_codes_keep_the_individual_columns(capsys):
    codes = parse_holland_codes("S", "I", "S", "")
    assert codes["triplet"] == "SIS"
    assert "WARNING" in capsys.readouterr().out


# --- Tracks and languages --------------------------------------------------

def test_economics_track_is_not_duplicated():
    """Both "اقتصاد" and "اقتصاد واجتماع" map to the same track."""
    assert parse_secondary_tracks("اقتصاد واجتماع") == ["اقتصاد واجتماع"]


def test_multiple_tracks_are_parsed():
    tracks = parse_secondary_tracks("علوم عامة - علوم حياة")
    assert set(tracks) == {"علوم عامة", "علوم حياة"}


def test_parse_languages_reads_presence_per_column():
    langs = parse_languages({"arabic": "x", "english": None, "french": "y", "other_lang": None})
    assert langs == ["العربية", "الفرنسية"]


# --- Branch address -> governorate ----------------------------------------

def test_district_name_resolves_to_its_governorate():
    assert extract_governorate_from_branch("طرابلس") == "الشمال"


def test_longer_keyword_wins():
    assert extract_governorate_from_branch("البقاع الغربي") == "البقاع"


def test_locality_beats_a_governorate_name_in_the_same_address():
    """"الحدث – بيروت" is a Mount Lebanon campus; first-match ordering used to
    file it under Beirut."""
    assert extract_governorate_from_branch("الحدث - قرب بيروت") == "جبل لبنان"


def test_districts_missing_from_the_old_hand_map_now_resolve():
    for district, gov in [("جزين", "الجنوب"), ("بشري", "الشمال"),
                          ("مرجعيون", "النبطية"), ("راشيا", "البقاع")]:
        assert extract_governorate_from_branch(f"فرع {district}") == gov


def test_unknown_address_returns_none():
    assert extract_governorate_from_branch("عنوان غير معروف") is None


# --- merge_data ------------------------------------------------------------

def _major(**kw):
    base = dict(id=1, name_ar="تجريبي",
                holland_codes={"primary": "R", "secondary": "I",
                               "tertiary": "A", "triplet": "RIA"},
                teaching_languages=["العربية"], locations=[])
    base.update(kw)
    return base


def test_merge_never_overwrites_holland_codes():
    majors = merge_data([_major()], {1: {"holland_primary": "S", "holland_secondary": "E",
                                         "holland_tertiary": "C", "holland_triplet": "SEC"}})
    assert majors[0]["holland_codes"]["triplet"] == "RIA"


def test_merge_adds_branches_only_when_the_major_has_none():
    majors = merge_data([_major(locations=[])],
                        {1: {"branches": ["فرع صيدا"]}})
    assert majors[0]["locations"][0]["governorate"] == "الجنوب"


def test_merge_leaves_existing_branches_alone():
    existing = [{"governorate": "بيروت", "district": "بيروت", "branch": "فرع بيروت"}]
    majors = merge_data([_major(locations=list(existing))],
                        {1: {"branches": ["فرع صيدا"]}})
    assert majors[0]["locations"] == existing


def test_language_code_is_case_insensitive():
    majors = merge_data([_major(teaching_languages=[])], {1: {"teaching_lang_code": "ARA/ENG"}})
    assert majors[0]["teaching_languages"] == ["العربية", "الانكليزية"]


def test_supplementary_language_code_replaces_the_main_sheet():
    """The supplementary sheet exists to correct the main one, so "Eng" there
    means English only. Merging the two instead changes 6 majors' languages —
    verified by regenerating — so replacement is deliberate, not an oversight."""
    majors = merge_data([_major(teaching_languages=["العربية", "أرمني"])],
                        {1: {"teaching_lang_code": "Eng"}})
    assert majors[0]["teaching_languages"] == ["الانكليزية"]


def test_unrecognised_language_code_warns_and_keeps_the_main_sheet(capsys):
    majors = merge_data([_major(teaching_languages=["العربية"])],
                        {1: {"teaching_lang_code": "XYZ"}})
    assert majors[0]["teaching_languages"] == ["العربية"]
    assert "WARNING" in capsys.readouterr().out


# --- Dataset integrity -----------------------------------------------------

def test_committed_major_ids_are_unique():
    from data.loader import load_majors

    ids = [m["id"] for m in load_majors()]
    assert len(ids) == len(set(ids))


def test_committed_locations_index_covers_every_major():
    """locations.json is keyed by id; a collision would silently drop a major.
    The id-46 duplicate is reassigned, so this must stay 1:1."""
    import json

    from config.settings import LOCATIONS_JSON_FILE
    from data.loader import load_majors

    index = json.loads(LOCATIONS_JSON_FILE.read_text(encoding="utf-8"))["major_locations"]
    assert len(index) == len(load_majors())