Download src/core/engine.py from ya02/kalim: direct link, hf CLI and curl.
- Browser
- Download file 41.9 kB
-
https://huggingface.co/spaces/ya02/kalim/resolve/main/src/core/engine.py
- Command line
-
hf download hf://spaces/ya02/kalim/src/core/engine.py
-
curl -L -o engine.py https://huggingface.co/spaces/ya02/kalim/resolve/main/src/core/engine.py
41.9 kB
| """Kalim advisor engine — every piece of chat/recommendation logic that does | |
| not depend on a specific UI toolkit. | |
| `app.py` (Streamlit) and `app_gradio.py` (Gradio, for Hugging Face Spaces) | |
| are both thin shells over this module. Behaviour that the client validated — | |
| result phrasing, the why-not-in-results reasons, reply sanitisation — lives | |
| here exactly once, so the two front ends cannot drift apart the way the old | |
| `src/chains/` fork did. | |
| Nothing in this file may import streamlit or gradio. State (the student's | |
| selections and current results) is passed in as plain arguments; each front | |
| end owns its own session storage. | |
| """ | |
| import logging | |
| import os | |
| import re | |
| from dataclasses import dataclass | |
| from functools import lru_cache | |
| from typing import Callable, Dict, List, Optional, Tuple | |
| from config.settings import HOLLAND_CODES | |
| from config.prompts import FOLLOWUP_PROMPT | |
| from data.loader import load_majors | |
| from rag.retriever import KalimRetriever, StudentProfile | |
| logger = logging.getLogger("kalim") | |
| # How many cards to reveal at a time. The retriever returns every major above | |
| # the score threshold — often 60+ — and the student pages through them. Capping | |
| # the search itself at 10 was the reason testers reported that majors visible in | |
| # the source spreadsheet never appeared in the chatbot. | |
| PAGE_SIZE = 10 | |
| MAX_RESULTS = 100 | |
| # Abuse controls for the public deployments, which are unauthenticated URLs | |
| # whose chat path ends in a paid provider call. The length cap bounds what a | |
| # visitor can push into the prompt; the per-session turn budget is enforced by | |
| # each shell (session storage is shell-owned) — beyond it the shells pass | |
| # llm=None and the student still gets the deterministic answer_without_llm | |
| # reply. Per-session only: neither host offers IP-level limiting. | |
| MAX_QUESTION_CHARS = 500 | |
| MAX_LLM_TURNS_PER_SESSION = 30 | |
| _QUESTION_TOO_LONG = "سؤالك طويل جداً. رجاءً اختصره إلى أقل من 500 حرف وأعد إرساله." | |
| # "Show me more majors like me" triggers. Exported so a front end can grow its | |
| # own visible window after the reply lists the next page. | |
| # | |
| # Bare comparatives (أكثر، اكثر، غيرها، إضافية) used to be in this list and | |
| # hijacked ordinary questions — "أي تخصص أكثر طلباً؟" was answered with a | |
| # pagination table instead of an answer. Same bug class as TABLE_KEYWORDS | |
| # below. Only phrases that unambiguously ask for more results belong here. | |
| MORE_KEYWORDS = [ | |
| 'المزيد', 'مزيد من', 'تشبهني', | |
| 'تخصصات إضافية', 'تخصصات اضافية', 'تخصصات أخرى', 'تخصصات اخرى', | |
| 'تخصصات غيرها', 'اعرض أكثر', 'اعرض اكثر', 'show more', 'more', | |
| ] | |
| # Only an explicit request for a table/list gets one. This used to also fire | |
| # on "اعطيني" and "المطابقة", so "اعطيني تخصصات تناسب شخصيتي" — an ordinary | |
| # request for advice — was answered with a bare table instead of an answer. | |
| TABLE_KEYWORDS = ['جدول', 'قائمة', 'table', 'list'] | |
| def get_retriever() -> KalimRetriever: | |
| """One retriever per process, so the majors list is parsed once.""" | |
| return KalimRetriever() | |
| # --- LLM provider selection ------------------------------------------------- | |
| # An explicit cloud key wins, otherwise fall back to a local Ollama model. | |
| # The secret getter is injected: Streamlit layers st.secrets over the | |
| # environment, Gradio/HF Spaces use the environment alone. | |
| def build_llm(get_secret: Callable[[str], str] = os.getenv): | |
| """Construct the best available LLM client, or None. | |
| Backed by LangChain (llm.langchain_llm) rather than calling each | |
| provider's SDK directly — this is the thesis's LangChain design | |
| proposition, implemented in the code path a student's question | |
| actually travels through. | |
| LangChain is not installed everywhere this engine runs: requirements.txt | |
| (the Hugging Face Space) carries it, requirements-deploy.txt (Streamlit | |
| Community Cloud, Docker) does not have to. An ImportError therefore falls | |
| back to the provider SDKs in llm.cloud_llm instead of taking the chat | |
| down — the behaviour requirements.txt already claims in its comment | |
| ("cloud_llm.py is kept only as a documented fallback"). | |
| """ | |
| try: | |
| from llm.langchain_llm import build_langchain_llm | |
| except ImportError: | |
| logger.warning( | |
| "LangChain is not installed; falling back to the provider SDKs in " | |
| "llm.cloud_llm. Install langchain-core and the integration package " | |
| "for your provider to use the LangChain path." | |
| ) | |
| return _build_llm_via_sdks(get_secret) | |
| return build_langchain_llm(get_secret=get_secret) | |
| def _build_llm_via_sdks(get_secret: Callable[[str], str]): | |
| """Pre-LangChain provider selection, kept as the no-LangChain fallback.""" | |
| from llm.cloud_llm import PROVIDER_PRECEDENCE | |
| for env_var, provider in PROVIDER_PRECEDENCE: | |
| if not get_secret(env_var): | |
| continue | |
| try: | |
| from llm.cloud_llm import CloudLLM | |
| client = CloudLLM(provider, api_key=get_secret(env_var)) | |
| except Exception: | |
| logger.exception("Failed to initialise cloud provider %s", provider) | |
| continue | |
| if client.is_available: | |
| return client | |
| # Key present but no usable client — say so instead of silently | |
| # returning canned errors for every question. | |
| logger.error( | |
| "%s is configured but its client could not be created — either the " | |
| "provider SDK is not installed (pip install %s) or client creation " | |
| "failed; see the log above.", provider, provider, | |
| ) | |
| try: | |
| from llm.local_llm import get_llm as _get_local_llm | |
| client = _get_local_llm() | |
| except Exception: | |
| logger.exception("Failed to initialise the local Ollama client") | |
| return None | |
| return client if client is not None and client.is_available else None | |
| # --- Search ------------------------------------------------------------------ | |
| def run_search(profile: StudentProfile) -> List[Dict]: | |
| """Retrieve every matching major for the profile, best match first.""" | |
| return get_retriever().retrieve_with_expansion( | |
| profile, top_k=MAX_RESULTS, min_results=PAGE_SIZE, score_threshold=50 | |
| ) | |
| def results_intro(results: List[Dict], holland_codes: List[str], interactive_cards: bool = True) -> str: | |
| """Chat message summarising a fresh result set. | |
| `interactive_cards=True` is the Streamlit wording (expander cards plus the | |
| filter row above them); False suits a front end that renders the results as | |
| a numbered table in the chat, like the Gradio app. | |
| """ | |
| if not results: | |
| return "لم أجد تخصصات تطابق معاييرك. جرّب تغيير الفلاتر من الشريط الجانبي." | |
| student_codes = ' - '.join(holland_codes) | |
| shown = min(len(results), PAGE_SIZE) | |
| intro = f"وجدت **{len(results)} تخصصاً** يتطابق مع خياراتك ({student_codes})" | |
| intro += f"، وأعرض لك أفضل {shown} منها.\n\n" if len(results) > shown else ".\n\n" | |
| # The retriever widens the search by re-scoring under every ordering of the | |
| # student's codes when too few majors clear the threshold. Say so, since | |
| # the percentages shown still reflect the order the student gave. | |
| if any(r.get('expanded') for r in results): | |
| intro += ( | |
| "🔎 لقلة النتائج المطابقة تماماً، وسّعت البحث ليشمل تخصصات قريبة من " | |
| "أنماط شخصيتك بترتيب مختلف؛ النسب المعروضة تعكس ترتيبك أنت.\n\n" | |
| ) | |
| if not interactive_cards: | |
| return ( | |
| intro | |
| + ("يمكنك الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد.\n\n" if len(results) > shown else "") | |
| + "💡 اضغط على أي تخصص في القائمة أدناه لعرض تفاصيله الكاملة، " | |
| + "أو اكتب رقمه أو اسمه في مربع الكتابة لتسألني عنه." | |
| ) | |
| return ( | |
| intro | |
| + "يمكنك تصفية النتائج أدناه بحسب رمز الشخصية، الكلية، أو نسبة المطابقة" | |
| + ("، أو الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد" if len(results) > shown else "") | |
| + ".\n\n💡 اضغط على أي بطاقة لعرض تفاصيل التخصص، أو اكتب رقمه في الأسفل لتسألني عنه." | |
| ) | |
| def results_table(results: List[Dict], start: int = 1, title: str = "**جدول التخصصات المقترحة:**") -> str: | |
| """Markdown table of a result window, numbered from `start`.""" | |
| table = "| # | التخصص | رمز هولاند | النسبة |\n|---|--------|------------|--------|\n" | |
| for i, r in enumerate(results, start): | |
| major = r.get('major', {}) | |
| table += ( | |
| f"| {i} | {major.get('name_ar', '')} " | |
| f"| {major.get('holland_codes', {}).get('triplet', '')} " | |
| f"| {r.get('holland_score', 0)}% |\n" | |
| ) | |
| return f"{title}\n\n{table}" | |
| def kalim_recommendation(results: List[Dict], holland_codes: List[str], | |
| track: Optional[str], governorates: List[str]) -> str: | |
| """Kalim's personalized recommendation based on Holland, Track, and Location.""" | |
| if not results: | |
| return "" | |
| top = results[0] | |
| major = top.get('major', {}) | |
| score = top.get('holland_score', 0) | |
| holland = major.get('holland_codes', {}).get('triplet', '') | |
| student_codes = '-'.join(holland_codes) | |
| track = track or "الكل" | |
| location = ', '.join(governorates) if governorates else "الكل" | |
| rec = f"🎯 **توصية كليم:**\n\n" | |
| rec += f"بناءً على:\n" | |
| rec += f"• شخصيتك المهنية: **{student_codes}**\n" | |
| rec += f"• فرعك الثانوي: **{track}**\n" | |
| rec += f"• موقعك: **{location}**\n\n" | |
| rec += f"أنصحك بتخصص **{major.get('name_ar', '')}**\n\n" | |
| rec += f"✅ توافق هولاند: **{score}%** (التخصص: {holland})\n" | |
| # Only claim what was checked. The track line used to print | |
| # unconditionally — even when the student skipped the track step — and the | |
| # location line printed the student's whole selection, though the location | |
| # filter is a union: one matching branch is enough to be in the results. | |
| if track != "الكل": | |
| rec += f"✅ يقبل فرعك الثانوي ({track})\n" | |
| if governorates: | |
| major_govs = {loc.get('governorate') for loc in major.get('locations', []) | |
| if loc.get('governorate')} | |
| available_here = [g for g in governorates if g in major_govs] | |
| if available_here: | |
| rec += f"✅ متوفر في {', '.join(available_here)}\n" | |
| rec += f"\n📚 {major.get('faculty_ar', '')}" | |
| return rec | |
| def compare_top3(results: List[Dict]) -> str: | |
| """Short comparison of the three best matches.""" | |
| if not results: | |
| return "" | |
| comparison = "**مقارنة بين أفضل 3 تخصصات:**\n\n" | |
| for i, r in enumerate(results[:3], 1): | |
| major = r.get('major', {}) | |
| comparison += f"**{i}. {major.get('name_ar', '')}**\n" | |
| comparison += f"- المطابقة: {r.get('holland_score', 0)}%\n" | |
| comparison += f"- الكلية: {major.get('faculty_ar', '')}\n\n" | |
| return comparison | |
| # --- Major formatting --------------------------------------------------------- | |
| def format_major_details(major: Dict) -> str: | |
| """Format major details from the dataset for display.""" | |
| parts = [f"**{major.get('name_ar', '')}**"] | |
| if major.get('name_en'): | |
| parts.append(f"*{major['name_en']}*") | |
| parts.append("") | |
| parts.append(f"📚 **الكلية:** {major.get('faculty_ar', 'غير متوفر')}") | |
| if major.get('holland_interpretation'): | |
| parts.append(f"🧠 **تفسير هولاند:** {major['holland_interpretation']}") | |
| parts.append(f"📝 **الوصف:** {major.get('description', 'غير متوفر')}") | |
| parts.append(f"💼 **فرص العمل:** {major.get('career_opportunities', 'غير متوفر')}") | |
| parts.append(f"📋 **متطلبات القبول:** {major.get('admission_requirements', 'غير متوفر')}") | |
| parts.append(f"🗣️ **لغات التدريس:** {', '.join(major.get('teaching_languages', []))}") | |
| if major.get('entry_mechanism'): | |
| parts.append(f"🔑 **آلية الدخول:** {major['entry_mechanism']}") | |
| if major.get('locations'): | |
| branches = [loc.get('branch', '') for loc in major['locations'] if loc.get('branch')] | |
| if branches: | |
| parts.append(f"📍 **الفروع:** {' | '.join(branches)}") | |
| return "\n\n".join(parts) | |
| def major_context_lines(major: Dict, score=None) -> str: | |
| """Compact plain-text rendering of a major, used to build LLM context.""" | |
| lines = [f"{major.get('name_ar', '')} ({major.get('name_en', '')})"] | |
| lines.append(f" الكلية: {major.get('faculty_ar', '')}") | |
| lines.append(f" رمز هولاند: {major.get('holland_codes', {}).get('triplet', '')}") | |
| if score is not None: | |
| lines.append(f" نسبة المطابقة: {score}%") | |
| for label, key in ( | |
| ("الوصف", 'description'), | |
| ("فرص العمل", 'career_opportunities'), | |
| ("متطلبات القبول", 'admission_requirements'), | |
| ("آلية الدخول", 'entry_mechanism'), | |
| ): | |
| if major.get(key): | |
| lines.append(f" {label}: {major[key]}") | |
| lines.append(f" لغات التدريس: {', '.join(major.get('teaching_languages', []))}") | |
| return "\n".join(lines) | |
| # --- Reply sanitisation -------------------------------------------------------- | |
| # Scripts that must never reach a student. Qwen and other multilingual models | |
| # occasionally leak CJK characters into Arabic output, and private-use or | |
| # replacement code points show up when generation goes wrong. | |
| _UNWANTED_SCRIPTS = re.compile( | |
| '[' | |
| ' -〿' # CJK punctuation | |
| '-ヿ' # hiragana, katakana | |
| 'ㇰ-ㇿ' # katakana extensions | |
| '㐀-䶿' # CJK unified ideographs extension A | |
| '一-鿿' # CJK unified ideographs | |
| '가-' # hangul syllables | |
| '豈-' # CJK compatibility ideographs | |
| '-' # private use area | |
| '�' # replacement character | |
| ']+' | |
| ) | |
| _ARABIC_LETTER = re.compile(r'[-ۿ]') | |
| _LATIN_LETTER = re.compile(r'[A-Za-z]') | |
| # Reasoning models wrap their chain of thought in <think> tags, or announce it | |
| # in a preamble. cloud_llm asks the provider to suppress it, but that switch is | |
| # model-specific and silently absent on some, so strip it here as well. | |
| _THINK_BLOCK = re.compile(r'<think>.*?</think>|<thinking>.*?</thinking>', re.DOTALL | re.IGNORECASE) | |
| _OPEN_THINK = re.compile(r'<think(?:ing)?>', re.IGNORECASE) | |
| _CLOSE_THINK = re.compile(r'</think(?:ing)?>', re.IGNORECASE) | |
| _THINK_PREAMBLE = re.compile( | |
| r"^\s*(?:here'?s?\s+(?:a|my|the)\s+thinking\s+process|" | |
| r"let me think|thinking process|reasoning)\s*:?.*?(?=\n\s*\n|\Z)", | |
| re.IGNORECASE | re.DOTALL, | |
| ) | |
| def strip_reasoning(text: str) -> str: | |
| """Remove any chain-of-thought the model exposed despite being asked not to.""" | |
| cleaned = _THINK_BLOCK.sub('', text) | |
| # Some providers strip the opening tag and leave the closing one: whatever | |
| # follows the last closing tag is the answer. | |
| if _CLOSE_THINK.search(cleaned): | |
| cleaned = _CLOSE_THINK.split(cleaned)[-1] | |
| # An unclosed opening tag means the answer never arrived; dropping it leaves | |
| # an empty reply, which the caller replaces with the data-driven fallback. | |
| cleaned = _OPEN_THINK.split(cleaned)[0] | |
| cleaned = _THINK_PREAMBLE.sub('', cleaned) | |
| return cleaned.strip() | |
| def sanitize_reply(text: str, fallback: str) -> str: | |
| """Drop stray foreign-script characters, or reject the reply outright. | |
| A few leaked characters are cleaned silently. A reply that is largely the | |
| wrong script is treated as a failed generation and replaced by the | |
| data-driven fallback, because a student should never be shown one. | |
| """ | |
| text = strip_reasoning(text or "") | |
| if not text.strip(): | |
| logger.warning("LLM returned an empty reply") | |
| return fallback | |
| cleaned = _UNWANTED_SCRIPTS.sub('', text) | |
| removed = len(text) - len(cleaned) | |
| if removed > max(4, len(text) * 0.02): | |
| logger.warning("LLM reply contained %d foreign-script characters; discarded", removed) | |
| return fallback | |
| # The app is Arabic-only. Latin is fine for a major's English name, but a | |
| # reply that is mostly Latin means the model answered in the wrong language. | |
| arabic = len(_ARABIC_LETTER.findall(cleaned)) | |
| latin = len(_LATIN_LETTER.findall(cleaned)) | |
| if arabic + latin > 40 and arabic < (arabic + latin) * 0.4: | |
| logger.warning("LLM replied mostly in Latin script (%d Arabic vs %d Latin)", arabic, latin) | |
| return fallback | |
| # Tidy whitespace left behind by removed characters. | |
| return re.sub(r'[ \t]{2,}', ' ', cleaned).strip() | |
| # --- Holland (RIASEC) questions ------------------------------------------------ | |
| # Students ask what a personality type *is*, not only which majors match it. | |
| # The authoritative Arabic definitions live in config.settings, so they are fed | |
| # to the model as context: left to its own knowledge a smaller model will | |
| # happily define الواقعي as التقليدي. | |
| _HOLLAND_BY_NAME = {info['ar']: code for code, info in HOLLAND_CODES.items()} | |
| def holland_types_mentioned(message: str) -> List[str]: | |
| """RIASEC codes the question is about, by Arabic name or standalone letter.""" | |
| found = [] | |
| for name, code in _HOLLAND_BY_NAME.items(): | |
| if name in message and code not in found: | |
| found.append(code) | |
| for code in HOLLAND_CODES: | |
| if code not in found and re.search(rf'(?<![A-Za-z]){code}(?![A-Za-z])', message): | |
| found.append(code) | |
| return found | |
| # "What does this type mean?" — a definition request, not a request for majors. | |
| _DEFINITION_CUES = ( | |
| 'ما هي', 'ما هو', 'ماهي', 'ماهو', 'شو هي', 'شو هو', 'شو يعني', | |
| 'ماذا يعني', 'ماذا تعني', 'معنى', 'يعني', 'تعني', 'عرّف', 'عرف', | |
| 'اشرح', 'وضح', 'وضّح', 'ما معنى', | |
| ) | |
| def is_definition_question(message: str) -> bool: | |
| return any(cue in message for cue in _DEFINITION_CUES) | |
| def holland_reference(codes: List[str]) -> str: | |
| """Context block describing the given RIASEC codes, from the dataset's own | |
| definitions rather than the model's memory.""" | |
| lines = [ | |
| f"{code} — {HOLLAND_CODES[code]['ar']} ({HOLLAND_CODES[code]['en']}): " | |
| f"{HOLLAND_CODES[code]['description_ar']}" | |
| for code in codes if code in HOLLAND_CODES | |
| ] | |
| if not lines: | |
| return "" | |
| return "[مرجع أنماط هولاند RIASEC — تعريفات معتمدة]\n" + "\n".join(lines) | |
| def explain_holland_types(codes: List[str], visible: Optional[List[Dict]]) -> str: | |
| """Deterministic answer to "what is the <type> personality?".""" | |
| parts = [] | |
| for code in codes: | |
| info = HOLLAND_CODES[code] | |
| parts.append(f"**{info['ar']} ({code} — {info['en']})**\n{info['description_ar']}") | |
| answer = "\n\n".join(parts) | |
| if visible: | |
| matching = [ | |
| r for r in visible | |
| if r.get('major', {}).get('holland_codes', {}).get('primary') in codes | |
| ] | |
| if matching: | |
| names = "، ".join(r['major'].get('name_ar', '') for r in matching[:5]) | |
| answer += f"\n\nمن بين التخصصات المقترحة لك، الأقرب إلى هذا النمط: {names}." | |
| return answer | |
| # --- Prompt boundary and reply validation ------------------------------------- | |
| # The student's message is interpolated into FOLLOWUP_PROMPT. Left unfenced, a | |
| # message can reproduce the template's own section headers and impersonate the | |
| # database block, getting Kalim to state invented admission requirements in an | |
| # authoritative voice. Two defences, both here so both front ends inherit them. | |
| _PROMPT_STRUCTURE_MARKERS = ( | |
| "معلومات الطالب:", "معلومات من قاعدة البيانات", "سؤال الطالب", | |
| "<<<سؤال_الطالب", "سؤال_الطالب>>>", | |
| "[التخصص الذي سأل عنه الطالب", "[التخصصات المقترحة للطالب", | |
| ) | |
| _PERCENT_RE = re.compile(r'(\d+(?:[.,]\d+)?)\s*[%٪]') | |
| def fence_question(message: str) -> str: | |
| """Drop lines that impersonate the prompt's own structure.""" | |
| kept = [ | |
| line for line in message.splitlines() | |
| if not any(marker in line for marker in _PROMPT_STRUCTURE_MARKERS) | |
| ] | |
| return "\n".join(kept).strip() | |
| def _percent_keys(text: str) -> set: | |
| """Percentages in `text`, as exact strings plus rounded integers. | |
| Rounding is tolerated so a model that says "81%" for a context value of | |
| 81.4% is not treated as inventing a number. | |
| """ | |
| keys = set() | |
| for raw in _PERCENT_RE.findall(text): | |
| keys.add(raw) | |
| try: | |
| keys.add(str(round(float(raw.replace(',', '.'))))) | |
| except ValueError: | |
| pass | |
| return keys | |
| def reply_consistent_with_context(reply: str, context: str) -> bool: | |
| """Whether the reply stays inside the data it was given. | |
| Checks the two claim types a student would act on: match percentages and | |
| major names. Anything asserted that is absent from the retrieved context | |
| means the model invented it (or was steered into doing so), and the caller | |
| shows the deterministic answer instead. Conservative by design — the cost | |
| of a false negative is a plainer reply, the cost of a false positive is a | |
| confidently wrong admission requirement. | |
| """ | |
| allowed_percents = _percent_keys(context) | |
| for value in _percent_keys(reply): | |
| if value not in allowed_percents: | |
| logger.warning("LLM reply asserted a percentage absent from context") | |
| return False | |
| for major in load_majors(): | |
| name = major.get('name_ar') | |
| if name and name in reply and name not in context: | |
| logger.warning("LLM reply named a major absent from context: %s", name) | |
| return False | |
| return True | |
| # --- Free chat ------------------------------------------------------------------- | |
| def profile_summary(holland_codes: List[str], track: Optional[str], | |
| languages: List[str], governorates: List[str]) -> str: | |
| """The student's answers, written out for the model.""" | |
| lines = [ | |
| "- نمط الشخصية (هولاند): " | |
| + (", ".join(f"{c} ({HOLLAND_CODES[c]['ar']})" for c in holland_codes if c in HOLLAND_CODES) or "غير محدد") | |
| ] | |
| lines.append(f"- الفرع الثانوي: {track or 'لم يحدده'}") | |
| lines.append(f"- لغات التدريس المفضلة: {', '.join(languages) or 'كل اللغات'}") | |
| lines.append(f"- المحافظات المفضلة: {', '.join(governorates) or 'كل المحافظات'}") | |
| return "\n".join(lines) | |
| def why_not_in_results(major: Dict, holland_codes: List[str], track: Optional[str], | |
| languages: List[str], governorates: List[str]) -> List[str]: | |
| """Which of the student's own criteria this major fails. | |
| Turns "this major is not in your results" into a specific, checkable reason, | |
| so the answer can be useful instead of merely apologetic. | |
| """ | |
| reasons = [] | |
| if track and major.get('allowed_tracks') and track not in major['allowed_tracks']: | |
| reasons.append( | |
| f"لا يقبل فرع {track}، بل {', '.join(major['allowed_tracks'])}" | |
| ) | |
| if languages and major.get('teaching_languages') and not set(languages) & set(major['teaching_languages']): | |
| reasons.append(f"لغة التدريس فيه {', '.join(major['teaching_languages'])}") | |
| if governorates: | |
| available = {loc.get('governorate') for loc in major.get('locations', []) if loc.get('governorate')} | |
| if available and not available & set(governorates): | |
| reasons.append(f"غير متوفر في {', '.join(governorates)} بل في {', '.join(sorted(available))}") | |
| if holland_codes: | |
| score = get_retriever().calculate_holland_score( | |
| holland_codes, major.get('holland_codes', {}) | |
| ) | |
| if score < 50: | |
| # Worded so the number cannot be lifted out and re-presented as a | |
| # strong match; smaller models did exactly that with "منخفضة (34.9%)". | |
| reasons.append(f"توافقه مع نمط شخصيته ضعيف: {score} من 100 فقط") | |
| return reasons | |
| # Arabic-Indic (٠-٩) and Eastern Arabic-Indic (۰-۹) digits, mapped to ASCII so | |
| # "رقم ٣" and "رقم 3" resolve identically — an Arabic keyboard produces the | |
| # former, and the app's own examples use it. | |
| _DIGIT_MAP = str.maketrans( | |
| "٠١٢٣٤٥٦٧٨٩" "۰۱۲۳۴۵۶۷۸۹", | |
| "0123456789" "0123456789", | |
| ) | |
| # A number only counts as a card reference when the student marked it as one, | |
| # or the whole message is the number. Without this, "أريد 5 تخصصات" silently | |
| # answered about result #5. | |
| _ORDINAL_REF = re.compile(r'(?:رقم|الرقم|#)\s*(\d+)') | |
| _NUMBER_ONLY = re.compile(r'\s*(\d+)\s*[؟?.!]?\s*') | |
| def majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]: | |
| """Majors the student referred to, as (major, result_or_None) pairs. | |
| result is the entry from the student's shortlist when the major is in it, | |
| otherwise None — the caller uses that to explain the difference. | |
| """ | |
| found = [] | |
| seen_ids = set() | |
| def add(major, result): | |
| marker = (major.get('id'), major.get('name_ar')) | |
| if marker not in seen_ids: | |
| seen_ids.add(marker) | |
| found.append((major, result)) | |
| # Reference by position ("رقم 3" / "رقم ٣"), with Arabic-Indic digits | |
| # normalised first. A bare number is not an index — see _ORDINAL_REF. | |
| normalized = message.translate(_DIGIT_MAP) | |
| match = _ORDINAL_REF.search(normalized) or _NUMBER_ONLY.fullmatch(normalized) | |
| if match: | |
| index = int(match.group(1)) | |
| if 1 <= index <= len(visible): | |
| result = visible[index - 1] | |
| add(result.get('major', {}), result) | |
| # Reference by name. Arabic has no case, so match the message as written. | |
| # Longest names first so a specific major beats a shorter name nested in it. | |
| by_name = {r['major'].get('name_ar'): r for r in visible if r.get('major')} | |
| for major in sorted(load_majors(), key=lambda m: -len(m.get('name_ar', ''))): | |
| if len(found) >= 3: | |
| break | |
| name = major.get('name_ar') | |
| if name and name in message: | |
| add(major, by_name.get(name)) | |
| return found | |
| # --- Semantic fallback (embedding-based RAG) --------------------------------- | |
| # Activates rag.retriever.KalimRetriever.retrieve_by_query — and therefore | |
| # rag.vectorstore / rag.embeddings — which existed in the codebase but had no | |
| # caller. Used only when the student's question does not name a major | |
| # literally: majors_mentioned (exact name / card number) is tried first and | |
| # always wins when it finds something, since an exact reference is more | |
| # reliable than a similarity score. | |
| SEMANTIC_SCORE_THRESHOLD = 0.35 # tune against real queries before shipping | |
| SEMANTIC_MAX_RESULTS = 3 | |
| def semantic_majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]: | |
| """Majors found by embedding similarity rather than an exact name/number. | |
| Returns the same (major, result_or_None) shape as majors_mentioned, so | |
| callers do not need to branch on which path found the major. | |
| """ | |
| if len(message.strip()) < 8: | |
| return [] | |
| try: | |
| hits = get_retriever().retrieve_by_query(message, top_k=SEMANTIC_MAX_RESULTS) | |
| except Exception: | |
| # Missing index, missing optional deps (chromadb/sentence-transformers), | |
| # or an unreachable embedding model must never break the chat — the | |
| # caller already has answer_without_llm as a safe fallback. | |
| logger.exception("Semantic search failed; continuing without it") | |
| return [] | |
| by_id = {r['major'].get('id'): r for r in visible if r.get('major')} | |
| found = [] | |
| for hit in hits: | |
| if hit.get('semantic_score', 0) < SEMANTIC_SCORE_THRESHOLD: | |
| continue | |
| major = hit.get('major') | |
| if not major: | |
| continue | |
| result = by_id.get(major.get('id')) | |
| found.append((major, result)) | |
| return found | |
| # --- Deterministic major answers ---------------------------------------------- | |
| # Used whenever no model is configured or its reply was rejected. It has to | |
| # read like an answer, not a record: dumping every field of a major was the | |
| # behaviour students saw as a non-answer. | |
| # Which field a question is actually about. Cheap keyword cues, deliberately | |
| # narrow — anything unmatched falls through to the general summary. | |
| _FIELD_CUES = ( | |
| ('admission_requirements', "شروط القبول", | |
| ('شرط', 'شروط', 'قبول', 'معدل', 'أقبل', 'اقبل', 'بقدر ادخل', 'بقدر أدخل')), | |
| ('career_opportunities', "فرص العمل", | |
| ('فرص', 'عمل', 'وظيف', 'مهن', 'شغل', 'مستقبل', 'راتب')), | |
| ('entry_mechanism', "آلية الدخول", | |
| ('آلية', 'الية', 'مباراة', 'دخول', 'تسجيل', 'امتحان')), | |
| ('teaching_languages', "لغات التدريس", | |
| ('لغة', 'لغات', 'بالعربي', 'بالانكليزي', 'بالفرنسي')), | |
| ('allowed_tracks', "الفروع المقبولة", | |
| ('فرعي', 'الفرع', 'فروع', 'ثانوي', 'بكالوريا')), | |
| ) | |
| def _asked_fields(message: str) -> List[Tuple[str, str]]: | |
| """(field, label) pairs the question is about, in cue order.""" | |
| return [(field, label) for field, label, cues in _FIELD_CUES | |
| if any(cue in message for cue in cues)] | |
| def _first_sentence(text: Optional[str], limit: int = 220) -> str: | |
| text = (text or "").strip() | |
| if not text: | |
| return "" | |
| head = text.split('.')[0].strip() | |
| if len(head) < 30 and '.' in text: # too short to stand alone | |
| head = '.'.join(text.split('.')[:2]).strip() | |
| if len(head) > limit: | |
| head = head[:limit].rsplit(' ', 1)[0] + "…" | |
| return head | |
| def _major_facts_line(major: Dict) -> str: | |
| bits = [] | |
| if major.get('faculty_ar'): | |
| bits.append(f"📚 {major['faculty_ar']}") | |
| if major.get('entry_mechanism'): | |
| bits.append(f"🔑 {major['entry_mechanism']}") | |
| govs = sorted({loc.get('governorate') for loc in major.get('locations', []) | |
| if loc.get('governorate')}) | |
| if govs: | |
| bits.append("📍 " + "، ".join(govs)) | |
| return " · ".join(bits) | |
| def describe_major(major: Dict, result: Optional[Dict], visible: Optional[List[Dict]], | |
| holland_codes: List[str], track: Optional[str], | |
| languages: List[str], governorates: List[str], | |
| message: str = "") -> str: | |
| """A readable answer about one major, built only from its record.""" | |
| name = major.get('name_ar', '') | |
| parts = [] | |
| if result: | |
| parts.append( | |
| f"**{name}** ضمن التخصصات المقترحة لك، بنسبة توافق " | |
| f"**{result.get('holland_score', 0)}%**." | |
| ) | |
| reasons = result.get('match_reasons') or [] | |
| if reasons: | |
| parts.append("لماذا يناسبك:\n" + "\n".join(f"- {reason}" for reason in reasons)) | |
| else: | |
| reasons = why_not_in_results(major, holland_codes, track, languages, governorates) | |
| parts.append( | |
| f"**{name}** خارج نتائجك الحالية لأنه {reasons[0]}." | |
| if reasons else f"**{name}** ليس ضمن التخصصات المقترحة لك حالياً." | |
| ) | |
| # Answer the specific thing asked, when the question named one. | |
| asked = _asked_fields(message) | |
| for field, label in asked: | |
| value = major.get(field) | |
| if isinstance(value, list): | |
| value = "، ".join(value) | |
| if value: | |
| parts.append(f"**{label}:** {value}") | |
| if not asked: | |
| summary = _first_sentence(major.get('description')) | |
| if summary: | |
| parts.append(summary) | |
| facts = _major_facts_line(major) | |
| if facts: | |
| parts.append(facts) | |
| if not result and visible: | |
| best = visible[0] | |
| parts.append( | |
| f"أقرب بديل ضمن نتائجك: **{best.get('major', {}).get('name_ar', '')}** " | |
| f"({best.get('holland_score', 0)}% توافق)." | |
| ) | |
| parts.append( | |
| "اضغط على بطاقة التخصص في القائمة لعرض كل التفاصيل، أو اسألني عن أي جانب آخر." | |
| if result else "اسألني عن شروط قبوله أو فرص العمل فيه إن أردت." | |
| ) | |
| return "\n\n".join(parts) | |
| def answer_without_llm(mentioned: List[Tuple[Dict, Optional[Dict]]], visible: Optional[List[Dict]], | |
| holland_codes: List[str], track: Optional[str], | |
| languages: List[str], governorates: List[str], | |
| message: str = "") -> str: | |
| """Deterministic reply used when no model is configured or its output failed. | |
| This must still answer the most common question — "which majors suit me?" — | |
| rather than telling the student to name a major, which was the behaviour | |
| that made the assistant look broken whenever the LLM was unavailable. | |
| """ | |
| if not mentioned: | |
| # "ما هي الشخصية الواقعية؟" is a question about a type, not a request | |
| # for a shortlist — answering it with five majors reads as a non-answer. | |
| asked_types = holland_types_mentioned(message) if message else [] | |
| if asked_types: | |
| return explain_holland_types(asked_types, visible) | |
| if visible: | |
| lines = [ | |
| f"**{i}. {r['major'].get('name_ar', '')}** — {r['major'].get('faculty_ar', '')} " | |
| f"({r.get('holland_score', 0)}% توافق)" | |
| for i, r in enumerate(visible[:5], 1) | |
| ] | |
| codes = ' - '.join(holland_codes) | |
| return ( | |
| f"بناءً على نمط شخصيتك ({codes}) وفرعك الثانوي، هذه أقرب التخصصات إليك:\n\n" | |
| + "\n\n".join(lines) | |
| + "\n\nاكتب رقم أي تخصص أو اسمه لأعطيك تفاصيله الكاملة." | |
| ) | |
| return "لم أجد نتائج بعد. اختر معايير البحث من الشريط الجانبي لأقترح عليك تخصصات مناسبة." | |
| major, result = mentioned[0] | |
| return describe_major(major, result, visible, holland_codes, track, | |
| languages, governorates, message=message) | |
| class ChatTurn: | |
| """One free-chat exchange: what to say, and the window now on screen. | |
| `visible` is the shell's new display window. The engine owns the paging | |
| decision so the two front ends cannot drift: the Gradio shell used to | |
| re-run the MORE_KEYWORDS test itself (missing the engine's extra guards) | |
| while the Streamlit shell never widened at all, so a follow-up "رقم 12" | |
| resolved against a stale list. | |
| """ | |
| reply: str | |
| visible: List[Dict] | |
| def free_chat_reply(message: str, visible: List[Dict], all_results: List[Dict], | |
| holland_codes: List[str], track: Optional[str], | |
| languages: List[str], governorates: List[str], llm) -> ChatTurn: | |
| """Answer a free-form question about the student's results. | |
| `visible` must be the filtered/sorted subset actually on screen, so numbered | |
| references resolve to the same cards the student is looking at; `all_results` | |
| is the pool the same filters produced, so a paged-in extra page cannot | |
| contain majors the student filtered away. | |
| Returns a ChatTurn — the shell must store `turn.visible` as its new window. | |
| """ | |
| if len(message) > MAX_QUESTION_CHARS: | |
| return ChatTurn(_QUESTION_TOO_LONG, visible) | |
| # Which majors the question is about. Resolved first so that a question | |
| # naming a specific major is never swallowed by the paging branch below. | |
| mentioned = majors_mentioned(message, visible) | |
| # An exact name/number reference always wins over a similarity guess. | |
| # Semantic search only runs when nothing was named literally, and only | |
| # for the kind of free-text description embeddings are good at — not a | |
| # short control phrase like "المزيد" or a bare "رقم 3", which the length | |
| # guard inside semantic_majors_mentioned already screens out. | |
| if not mentioned: | |
| mentioned = semantic_majors_mentioned(message, visible) | |
| # "Show me more majors like me". Testers reported the bot kept repeating the | |
| # same shortlist, because the search itself was capped. The full ranked list | |
| # is available now, so answer from the part the student has not seen yet. | |
| if not mentioned and any(kw in message for kw in MORE_KEYWORDS) and len(all_results) > len(visible): | |
| seen = {(r['major'].get('id'), r['major'].get('name_ar')) for r in visible} | |
| extra = [r for r in all_results if (r['major'].get('id'), r['major'].get('name_ar')) not in seen] | |
| if extra: | |
| page = extra[:PAGE_SIZE] | |
| reply = ( | |
| results_table( | |
| page, start=len(visible) + 1, | |
| title=f"نعم، هناك **{len(extra)} تخصصاً إضافياً** يتطابق مع شخصيتك. إليك التالية منها:", | |
| ) | |
| + "\n💡 اضغط \"عرض تخصصات إضافية\" أعلى مربع الكتابة لعرضها كبطاقات كاملة." | |
| ) | |
| # The listed page is now on screen: hand it back so the shell's | |
| # numbering and the student's next "رقم N" agree. | |
| return ChatTurn(reply, list(visible) + page) | |
| if any(kw in message for kw in TABLE_KEYWORDS) and visible: | |
| return ChatTurn(results_table(visible), visible) | |
| # "ما هي الشخصية الواقعية؟" — we hold the authoritative definition, so | |
| # answer it from the data rather than hoping the model relays it. Same | |
| # principle as the two branches above: never ask a model for something the | |
| # dataset already answers exactly. | |
| asked_types = holland_types_mentioned(message) | |
| if asked_types and not mentioned and is_definition_question(message): | |
| return ChatTurn(explain_holland_types(asked_types, visible), visible) | |
| # Everything below is retrieval: how the mentioned majors relate to this | |
| # student, then let the model phrase the answer. Answering straight from | |
| # these lookups is what produced the data-dump replies testers found | |
| # unfriendly. | |
| fallback = answer_without_llm(mentioned, visible, holland_codes, track, | |
| languages, governorates, message=message) | |
| if llm is None: | |
| return ChatTurn(fallback, visible) | |
| context_blocks = [] | |
| for major, result in mentioned: | |
| header = f"[التخصص الذي سأل عنه الطالب — {'ضمن نتائجه' if result else 'خارج نتائجه'}]" | |
| block = f"{header}\n{major_context_lines(major, result.get('holland_score') if result else None)}" | |
| if not result: | |
| reasons = why_not_in_results(major, holland_codes, track, languages, governorates) | |
| if reasons: | |
| block += "\n سبب عدم ظهوره ضمن نتائج الطالب: " + "؛ ".join(reasons) | |
| context_blocks.append(block) | |
| # Always include the shortlist so the model can suggest a real alternative | |
| # rather than inventing one. | |
| if visible: | |
| shortlist = "\n".join( | |
| f"{i}. {major_context_lines(r.get('major', {}), r.get('holland_score', 0))}" | |
| for i, r in enumerate(visible[:7], 1) | |
| ) | |
| context_blocks.append(f"[التخصصات المقترحة للطالب حالياً]\n{shortlist}") | |
| # Ground personality-type talk in the dataset's own definitions: the codes | |
| # the question names, or failing that the student's own. | |
| reference = holland_reference(holland_types_mentioned(message) or holland_codes) | |
| if reference: | |
| context_blocks.append(reference) | |
| context = "\n\n".join(context_blocks) | |
| prompt = FOLLOWUP_PROMPT.format( | |
| student_profile=profile_summary(holland_codes, track, languages, governorates), | |
| majors_data=context or "لا توجد بيانات مطابقة لسؤال الطالب.", | |
| question=fence_question(message), | |
| ) | |
| try: | |
| reply = llm.generate(prompt) | |
| except Exception: | |
| logger.exception("LLM generation failed") | |
| return ChatTurn(fallback, visible) | |
| # Never show raw model output: a leaked script or a wrong-language answer | |
| # is replaced by the data-driven reply, which is always correct if plainer. | |
| reply = sanitize_reply(reply, fallback) | |
| # The fallback is built from this same data, so it needs no checking — and | |
| # checking it would reject its own shortlist names. | |
| if reply is not fallback and not reply_consistent_with_context(reply, context): | |
| return ChatTurn(fallback, visible) | |
| return ChatTurn(reply, visible) | |