File size: 4,379 Bytes
bae15d1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
"""
Data Loader for Kalim Chatbot
Utilities for loading and accessing the processed knowledge base
"""
import json
from typing import Dict, List, Optional, Any
from functools import lru_cache

from config.settings import (
    MAJORS_JSON_FILE,
    LOCATIONS_JSON_FILE,
    HOLLAND_MAPPING_FILE,
    HOLLAND_CODES,
    SECONDARY_TRACKS,
    GOVERNORATES
)


@lru_cache(maxsize=1)
def load_majors() -> List[Dict[str, Any]]:
    """Load majors data from JSON file."""
    if not MAJORS_JSON_FILE.exists():
        raise FileNotFoundError(
            f"Majors file not found: {MAJORS_JSON_FILE}\n"
            "Run 'python -m src.data.processor' first to generate the data."
        )

    with open(MAJORS_JSON_FILE, 'r', encoding='utf-8') as f:
        data = json.load(f)

    return data.get('majors', [])

@lru_cache(maxsize=1)
def load_holland_mapping() -> Dict[str, Any]:
    """Load Holland RIASEC mapping from JSON file."""
    if not HOLLAND_MAPPING_FILE.exists():
        # Return default if file doesn't exist
        return {
            "codes": HOLLAND_CODES,
            "code_list": ["R", "I", "A", "S", "E", "C"],
            "dropdown_options": [
                {"code": "R", "label": "R - الواقعي (Realistic)"},
                {"code": "I", "label": "I - الباحث (Investigative)"},
                {"code": "A", "label": "A - الفني (Artistic)"},
                {"code": "S", "label": "S - الاجتماعي (Social)"},
                {"code": "E", "label": "E - المغامر (Enterprising)"},
                {"code": "C", "label": "C - التقليدي (Conventional)"},
            ]
        }

    with open(HOLLAND_MAPPING_FILE, 'r', encoding='utf-8') as f:
        return json.load(f)


def get_major_by_id(major_id: int) -> Optional[Dict[str, Any]]:
    """Get a specific major by ID."""
    majors = load_majors()
    for major in majors:
        if major['id'] == major_id:
            return major
    return None

def get_governorates() -> List[str]:
    """Get list of all governorates."""
    return list(GOVERNORATES.keys())


def get_districts(governorate: str) -> List[str]:
    """Get districts for a specific governorate."""
    return GOVERNORATES.get(governorate, [])


@lru_cache(maxsize=1)
def get_dataset_governorates() -> List[str]:
    """Governorates that actually occur in majors.json, in GOVERNORATES order.

    The UI must offer only these. `GOVERNORATES` follows Lebanon's newer
    eight-governorate division, in which عكار and بعلبك-الهرمل are governorates
    of their own; the dataset follows the older six, filing حلبا under
    الشمال/عكار and دورس under البقاع/بعلبك. Offering the config's list let a
    student from Akkar finish the whole wizard and be told no major matched —
    a dead end that reads as "the university has nothing for you" when the
    branch exists one level down.
    """
    present = {
        loc.get('governorate')
        for major in load_majors()
        for loc in major.get('locations', [])
        if loc.get('governorate')
    }
    return [g for g in GOVERNORATES if g != "الكل" and g in present]


@lru_cache(maxsize=1)
def _dataset_districts_by_gov() -> Dict[str, List[str]]:
    by_gov: Dict[str, set] = {}
    for major in load_majors():
        for loc in major.get('locations', []):
            gov, district = loc.get('governorate'), loc.get('district')
            if gov and district:
                by_gov.setdefault(gov, set()).add(district)
    return {gov: sorted(names) for gov, names in by_gov.items()}


def get_dataset_districts(governorate: str) -> List[str]:
    """Districts of this governorate present in majors.json.

    Kept in the GOVERNORATES reference order where possible, with anything the
    config map does not list appended — that is how عكار (a district of الشمال
    in the data, a governorate in the config) still reaches the dropdown.
    """
    present = set(_dataset_districts_by_gov().get(governorate, []))
    ordered = [d for d in GOVERNORATES.get(governorate, []) if d in present]
    return ordered + sorted(present - set(ordered))

# Export commonly used functions
__all__ = [
    'load_majors',
    'load_holland_mapping',
    'get_major_by_id',
    'get_governorates',
    'get_districts',
    'get_dataset_governorates',
    'get_dataset_districts',
]