kalim / src /core /engine.py
YasserHaidar
Repair the edits the web editor mangled, and the paths they broke
6e08541
Raw History Blame Contribute Delete
41.9 kB
"""Kalim advisor engine — every piece of chat/recommendation logic that does
not depend on a specific UI toolkit.
`app.py` (Streamlit) and `app_gradio.py` (Gradio, for Hugging Face Spaces)
are both thin shells over this module. Behaviour that the client validated —
result phrasing, the why-not-in-results reasons, reply sanitisation — lives
here exactly once, so the two front ends cannot drift apart the way the old
`src/chains/` fork did.
Nothing in this file may import streamlit or gradio. State (the student's
selections and current results) is passed in as plain arguments; each front
end owns its own session storage.
"""
import logging
import os
import re
from dataclasses import dataclass
from functools import lru_cache
from typing import Callable, Dict, List, Optional, Tuple
from config.settings import HOLLAND_CODES
from config.prompts import FOLLOWUP_PROMPT
from data.loader import load_majors
from rag.retriever import KalimRetriever, StudentProfile
logger = logging.getLogger("kalim")
# How many cards to reveal at a time. The retriever returns every major above
# the score threshold — often 60+ — and the student pages through them. Capping
# the search itself at 10 was the reason testers reported that majors visible in
# the source spreadsheet never appeared in the chatbot.
PAGE_SIZE = 10
MAX_RESULTS = 100
# Abuse controls for the public deployments, which are unauthenticated URLs
# whose chat path ends in a paid provider call. The length cap bounds what a
# visitor can push into the prompt; the per-session turn budget is enforced by
# each shell (session storage is shell-owned) — beyond it the shells pass
# llm=None and the student still gets the deterministic answer_without_llm
# reply. Per-session only: neither host offers IP-level limiting.
MAX_QUESTION_CHARS = 500
MAX_LLM_TURNS_PER_SESSION = 30
_QUESTION_TOO_LONG = "سؤالك طويل جداً. رجاءً اختصره إلى أقل من 500 حرف وأعد إرساله."
# "Show me more majors like me" triggers. Exported so a front end can grow its
# own visible window after the reply lists the next page.
#
# Bare comparatives (أكثر، اكثر، غيرها، إضافية) used to be in this list and
# hijacked ordinary questions — "أي تخصص أكثر طلباً؟" was answered with a
# pagination table instead of an answer. Same bug class as TABLE_KEYWORDS
# below. Only phrases that unambiguously ask for more results belong here.
MORE_KEYWORDS = [
'المزيد', 'مزيد من', 'تشبهني',
'تخصصات إضافية', 'تخصصات اضافية', 'تخصصات أخرى', 'تخصصات اخرى',
'تخصصات غيرها', 'اعرض أكثر', 'اعرض اكثر', 'show more', 'more',
]
# Only an explicit request for a table/list gets one. This used to also fire
# on "اعطيني" and "المطابقة", so "اعطيني تخصصات تناسب شخصيتي" — an ordinary
# request for advice — was answered with a bare table instead of an answer.
TABLE_KEYWORDS = ['جدول', 'قائمة', 'table', 'list']
@lru_cache(maxsize=1)
def get_retriever() -> KalimRetriever:
"""One retriever per process, so the majors list is parsed once."""
return KalimRetriever()
# --- LLM provider selection -------------------------------------------------
# An explicit cloud key wins, otherwise fall back to a local Ollama model.
# The secret getter is injected: Streamlit layers st.secrets over the
# environment, Gradio/HF Spaces use the environment alone.
def build_llm(get_secret: Callable[[str], str] = os.getenv):
"""Construct the best available LLM client, or None.
Backed by LangChain (llm.langchain_llm) rather than calling each
provider's SDK directly — this is the thesis's LangChain design
proposition, implemented in the code path a student's question
actually travels through.
LangChain is not installed everywhere this engine runs: requirements.txt
(the Hugging Face Space) carries it, requirements-deploy.txt (Streamlit
Community Cloud, Docker) does not have to. An ImportError therefore falls
back to the provider SDKs in llm.cloud_llm instead of taking the chat
down — the behaviour requirements.txt already claims in its comment
("cloud_llm.py is kept only as a documented fallback").
"""
try:
from llm.langchain_llm import build_langchain_llm
except ImportError:
logger.warning(
"LangChain is not installed; falling back to the provider SDKs in "
"llm.cloud_llm. Install langchain-core and the integration package "
"for your provider to use the LangChain path."
)
return _build_llm_via_sdks(get_secret)
return build_langchain_llm(get_secret=get_secret)
def _build_llm_via_sdks(get_secret: Callable[[str], str]):
"""Pre-LangChain provider selection, kept as the no-LangChain fallback."""
from llm.cloud_llm import PROVIDER_PRECEDENCE
for env_var, provider in PROVIDER_PRECEDENCE:
if not get_secret(env_var):
continue
try:
from llm.cloud_llm import CloudLLM
client = CloudLLM(provider, api_key=get_secret(env_var))
except Exception:
logger.exception("Failed to initialise cloud provider %s", provider)
continue
if client.is_available:
return client
# Key present but no usable client — say so instead of silently
# returning canned errors for every question.
logger.error(
"%s is configured but its client could not be created — either the "
"provider SDK is not installed (pip install %s) or client creation "
"failed; see the log above.", provider, provider,
)
try:
from llm.local_llm import get_llm as _get_local_llm
client = _get_local_llm()
except Exception:
logger.exception("Failed to initialise the local Ollama client")
return None
return client if client is not None and client.is_available else None
# --- Search ------------------------------------------------------------------
def run_search(profile: StudentProfile) -> List[Dict]:
"""Retrieve every matching major for the profile, best match first."""
return get_retriever().retrieve_with_expansion(
profile, top_k=MAX_RESULTS, min_results=PAGE_SIZE, score_threshold=50
)
def results_intro(results: List[Dict], holland_codes: List[str], interactive_cards: bool = True) -> str:
"""Chat message summarising a fresh result set.
`interactive_cards=True` is the Streamlit wording (expander cards plus the
filter row above them); False suits a front end that renders the results as
a numbered table in the chat, like the Gradio app.
"""
if not results:
return "لم أجد تخصصات تطابق معاييرك. جرّب تغيير الفلاتر من الشريط الجانبي."
student_codes = ' - '.join(holland_codes)
shown = min(len(results), PAGE_SIZE)
intro = f"وجدت **{len(results)} تخصصاً** يتطابق مع خياراتك ({student_codes})"
intro += f"، وأعرض لك أفضل {shown} منها.\n\n" if len(results) > shown else ".\n\n"
# The retriever widens the search by re-scoring under every ordering of the
# student's codes when too few majors clear the threshold. Say so, since
# the percentages shown still reflect the order the student gave.
if any(r.get('expanded') for r in results):
intro += (
"🔎 لقلة النتائج المطابقة تماماً، وسّعت البحث ليشمل تخصصات قريبة من "
"أنماط شخصيتك بترتيب مختلف؛ النسب المعروضة تعكس ترتيبك أنت.\n\n"
)
if not interactive_cards:
return (
intro
+ ("يمكنك الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد.\n\n" if len(results) > shown else "")
+ "💡 اضغط على أي تخصص في القائمة أدناه لعرض تفاصيله الكاملة، "
+ "أو اكتب رقمه أو اسمه في مربع الكتابة لتسألني عنه."
)
return (
intro
+ "يمكنك تصفية النتائج أدناه بحسب رمز الشخصية، الكلية، أو نسبة المطابقة"
+ ("، أو الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد" if len(results) > shown else "")
+ ".\n\n💡 اضغط على أي بطاقة لعرض تفاصيل التخصص، أو اكتب رقمه في الأسفل لتسألني عنه."
)
def results_table(results: List[Dict], start: int = 1, title: str = "**جدول التخصصات المقترحة:**") -> str:
"""Markdown table of a result window, numbered from `start`."""
table = "| # | التخصص | رمز هولاند | النسبة |\n|---|--------|------------|--------|\n"
for i, r in enumerate(results, start):
major = r.get('major', {})
table += (
f"| {i} | {major.get('name_ar', '')} "
f"| {major.get('holland_codes', {}).get('triplet', '')} "
f"| {r.get('holland_score', 0)}% |\n"
)
return f"{title}\n\n{table}"
def kalim_recommendation(results: List[Dict], holland_codes: List[str],
track: Optional[str], governorates: List[str]) -> str:
"""Kalim's personalized recommendation based on Holland, Track, and Location."""
if not results:
return ""
top = results[0]
major = top.get('major', {})
score = top.get('holland_score', 0)
holland = major.get('holland_codes', {}).get('triplet', '')
student_codes = '-'.join(holland_codes)
track = track or "الكل"
location = ', '.join(governorates) if governorates else "الكل"
rec = f"🎯 **توصية كليم:**\n\n"
rec += f"بناءً على:\n"
rec += f"• شخصيتك المهنية: **{student_codes}**\n"
rec += f"• فرعك الثانوي: **{track}**\n"
rec += f"• موقعك: **{location}**\n\n"
rec += f"أنصحك بتخصص **{major.get('name_ar', '')}**\n\n"
rec += f"✅ توافق هولاند: **{score}%** (التخصص: {holland})\n"
# Only claim what was checked. The track line used to print
# unconditionally — even when the student skipped the track step — and the
# location line printed the student's whole selection, though the location
# filter is a union: one matching branch is enough to be in the results.
if track != "الكل":
rec += f"✅ يقبل فرعك الثانوي ({track})\n"
if governorates:
major_govs = {loc.get('governorate') for loc in major.get('locations', [])
if loc.get('governorate')}
available_here = [g for g in governorates if g in major_govs]
if available_here:
rec += f"✅ متوفر في {', '.join(available_here)}\n"
rec += f"\n📚 {major.get('faculty_ar', '')}"
return rec
def compare_top3(results: List[Dict]) -> str:
"""Short comparison of the three best matches."""
if not results:
return ""
comparison = "**مقارنة بين أفضل 3 تخصصات:**\n\n"
for i, r in enumerate(results[:3], 1):
major = r.get('major', {})
comparison += f"**{i}. {major.get('name_ar', '')}**\n"
comparison += f"- المطابقة: {r.get('holland_score', 0)}%\n"
comparison += f"- الكلية: {major.get('faculty_ar', '')}\n\n"
return comparison
# --- Major formatting ---------------------------------------------------------
def format_major_details(major: Dict) -> str:
"""Format major details from the dataset for display."""
parts = [f"**{major.get('name_ar', '')}**"]
if major.get('name_en'):
parts.append(f"*{major['name_en']}*")
parts.append("")
parts.append(f"📚 **الكلية:** {major.get('faculty_ar', 'غير متوفر')}")
if major.get('holland_interpretation'):
parts.append(f"🧠 **تفسير هولاند:** {major['holland_interpretation']}")
parts.append(f"📝 **الوصف:** {major.get('description', 'غير متوفر')}")
parts.append(f"💼 **فرص العمل:** {major.get('career_opportunities', 'غير متوفر')}")
parts.append(f"📋 **متطلبات القبول:** {major.get('admission_requirements', 'غير متوفر')}")
parts.append(f"🗣️ **لغات التدريس:** {', '.join(major.get('teaching_languages', []))}")
if major.get('entry_mechanism'):
parts.append(f"🔑 **آلية الدخول:** {major['entry_mechanism']}")
if major.get('locations'):
branches = [loc.get('branch', '') for loc in major['locations'] if loc.get('branch')]
if branches:
parts.append(f"📍 **الفروع:** {' | '.join(branches)}")
return "\n\n".join(parts)
def major_context_lines(major: Dict, score=None) -> str:
"""Compact plain-text rendering of a major, used to build LLM context."""
lines = [f"{major.get('name_ar', '')} ({major.get('name_en', '')})"]
lines.append(f" الكلية: {major.get('faculty_ar', '')}")
lines.append(f" رمز هولاند: {major.get('holland_codes', {}).get('triplet', '')}")
if score is not None:
lines.append(f" نسبة المطابقة: {score}%")
for label, key in (
("الوصف", 'description'),
("فرص العمل", 'career_opportunities'),
("متطلبات القبول", 'admission_requirements'),
("آلية الدخول", 'entry_mechanism'),
):
if major.get(key):
lines.append(f" {label}: {major[key]}")
lines.append(f" لغات التدريس: {', '.join(major.get('teaching_languages', []))}")
return "\n".join(lines)
# --- Reply sanitisation --------------------------------------------------------
# Scripts that must never reach a student. Qwen and other multilingual models
# occasionally leak CJK characters into Arabic output, and private-use or
# replacement code points show up when generation goes wrong.
_UNWANTED_SCRIPTS = re.compile(
'['
' -〿' # CJK punctuation
'぀-ヿ' # hiragana, katakana
'ㇰ-ㇿ' # katakana extensions
'㐀-䶿' # CJK unified ideographs extension A
'一-鿿' # CJK unified ideographs
'가-힯' # hangul syllables
'豈-﫿' # CJK compatibility ideographs
'-' # private use area
'�' # replacement character
']+'
)
_ARABIC_LETTER = re.compile(r'[؀-ۿ]')
_LATIN_LETTER = re.compile(r'[A-Za-z]')
# Reasoning models wrap their chain of thought in <think> tags, or announce it
# in a preamble. cloud_llm asks the provider to suppress it, but that switch is
# model-specific and silently absent on some, so strip it here as well.
_THINK_BLOCK = re.compile(r'<think>.*?</think>|<thinking>.*?</thinking>', re.DOTALL | re.IGNORECASE)
_OPEN_THINK = re.compile(r'<think(?:ing)?>', re.IGNORECASE)
_CLOSE_THINK = re.compile(r'</think(?:ing)?>', re.IGNORECASE)
_THINK_PREAMBLE = re.compile(
r"^\s*(?:here'?s?\s+(?:a|my|the)\s+thinking\s+process|"
r"let me think|thinking process|reasoning)\s*:?.*?(?=\n\s*\n|\Z)",
re.IGNORECASE | re.DOTALL,
)
def strip_reasoning(text: str) -> str:
"""Remove any chain-of-thought the model exposed despite being asked not to."""
cleaned = _THINK_BLOCK.sub('', text)
# Some providers strip the opening tag and leave the closing one: whatever
# follows the last closing tag is the answer.
if _CLOSE_THINK.search(cleaned):
cleaned = _CLOSE_THINK.split(cleaned)[-1]
# An unclosed opening tag means the answer never arrived; dropping it leaves
# an empty reply, which the caller replaces with the data-driven fallback.
cleaned = _OPEN_THINK.split(cleaned)[0]
cleaned = _THINK_PREAMBLE.sub('', cleaned)
return cleaned.strip()
def sanitize_reply(text: str, fallback: str) -> str:
"""Drop stray foreign-script characters, or reject the reply outright.
A few leaked characters are cleaned silently. A reply that is largely the
wrong script is treated as a failed generation and replaced by the
data-driven fallback, because a student should never be shown one.
"""
text = strip_reasoning(text or "")
if not text.strip():
logger.warning("LLM returned an empty reply")
return fallback
cleaned = _UNWANTED_SCRIPTS.sub('', text)
removed = len(text) - len(cleaned)
if removed > max(4, len(text) * 0.02):
logger.warning("LLM reply contained %d foreign-script characters; discarded", removed)
return fallback
# The app is Arabic-only. Latin is fine for a major's English name, but a
# reply that is mostly Latin means the model answered in the wrong language.
arabic = len(_ARABIC_LETTER.findall(cleaned))
latin = len(_LATIN_LETTER.findall(cleaned))
if arabic + latin > 40 and arabic < (arabic + latin) * 0.4:
logger.warning("LLM replied mostly in Latin script (%d Arabic vs %d Latin)", arabic, latin)
return fallback
# Tidy whitespace left behind by removed characters.
return re.sub(r'[ \t]{2,}', ' ', cleaned).strip()
# --- Holland (RIASEC) questions ------------------------------------------------
# Students ask what a personality type *is*, not only which majors match it.
# The authoritative Arabic definitions live in config.settings, so they are fed
# to the model as context: left to its own knowledge a smaller model will
# happily define الواقعي as التقليدي.
_HOLLAND_BY_NAME = {info['ar']: code for code, info in HOLLAND_CODES.items()}
def holland_types_mentioned(message: str) -> List[str]:
"""RIASEC codes the question is about, by Arabic name or standalone letter."""
found = []
for name, code in _HOLLAND_BY_NAME.items():
if name in message and code not in found:
found.append(code)
for code in HOLLAND_CODES:
if code not in found and re.search(rf'(?<![A-Za-z]){code}(?![A-Za-z])', message):
found.append(code)
return found
# "What does this type mean?" — a definition request, not a request for majors.
_DEFINITION_CUES = (
'ما هي', 'ما هو', 'ماهي', 'ماهو', 'شو هي', 'شو هو', 'شو يعني',
'ماذا يعني', 'ماذا تعني', 'معنى', 'يعني', 'تعني', 'عرّف', 'عرف',
'اشرح', 'وضح', 'وضّح', 'ما معنى',
)
def is_definition_question(message: str) -> bool:
return any(cue in message for cue in _DEFINITION_CUES)
def holland_reference(codes: List[str]) -> str:
"""Context block describing the given RIASEC codes, from the dataset's own
definitions rather than the model's memory."""
lines = [
f"{code} — {HOLLAND_CODES[code]['ar']} ({HOLLAND_CODES[code]['en']}): "
f"{HOLLAND_CODES[code]['description_ar']}"
for code in codes if code in HOLLAND_CODES
]
if not lines:
return ""
return "[مرجع أنماط هولاند RIASEC — تعريفات معتمدة]\n" + "\n".join(lines)
def explain_holland_types(codes: List[str], visible: Optional[List[Dict]]) -> str:
"""Deterministic answer to "what is the <type> personality?"."""
parts = []
for code in codes:
info = HOLLAND_CODES[code]
parts.append(f"**{info['ar']} ({code} — {info['en']})**\n{info['description_ar']}")
answer = "\n\n".join(parts)
if visible:
matching = [
r for r in visible
if r.get('major', {}).get('holland_codes', {}).get('primary') in codes
]
if matching:
names = "، ".join(r['major'].get('name_ar', '') for r in matching[:5])
answer += f"\n\nمن بين التخصصات المقترحة لك، الأقرب إلى هذا النمط: {names}."
return answer
# --- Prompt boundary and reply validation -------------------------------------
# The student's message is interpolated into FOLLOWUP_PROMPT. Left unfenced, a
# message can reproduce the template's own section headers and impersonate the
# database block, getting Kalim to state invented admission requirements in an
# authoritative voice. Two defences, both here so both front ends inherit them.
_PROMPT_STRUCTURE_MARKERS = (
"معلومات الطالب:", "معلومات من قاعدة البيانات", "سؤال الطالب",
"<<<سؤال_الطالب", "سؤال_الطالب>>>",
"[التخصص الذي سأل عنه الطالب", "[التخصصات المقترحة للطالب",
)
_PERCENT_RE = re.compile(r'(\d+(?:[.,]\d+)?)\s*[%٪]')
def fence_question(message: str) -> str:
"""Drop lines that impersonate the prompt's own structure."""
kept = [
line for line in message.splitlines()
if not any(marker in line for marker in _PROMPT_STRUCTURE_MARKERS)
]
return "\n".join(kept).strip()
def _percent_keys(text: str) -> set:
"""Percentages in `text`, as exact strings plus rounded integers.
Rounding is tolerated so a model that says "81%" for a context value of
81.4% is not treated as inventing a number.
"""
keys = set()
for raw in _PERCENT_RE.findall(text):
keys.add(raw)
try:
keys.add(str(round(float(raw.replace(',', '.')))))
except ValueError:
pass
return keys
def reply_consistent_with_context(reply: str, context: str) -> bool:
"""Whether the reply stays inside the data it was given.
Checks the two claim types a student would act on: match percentages and
major names. Anything asserted that is absent from the retrieved context
means the model invented it (or was steered into doing so), and the caller
shows the deterministic answer instead. Conservative by design — the cost
of a false negative is a plainer reply, the cost of a false positive is a
confidently wrong admission requirement.
"""
allowed_percents = _percent_keys(context)
for value in _percent_keys(reply):
if value not in allowed_percents:
logger.warning("LLM reply asserted a percentage absent from context")
return False
for major in load_majors():
name = major.get('name_ar')
if name and name in reply and name not in context:
logger.warning("LLM reply named a major absent from context: %s", name)
return False
return True
# --- Free chat -------------------------------------------------------------------
def profile_summary(holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str]) -> str:
"""The student's answers, written out for the model."""
lines = [
"- نمط الشخصية (هولاند): "
+ (", ".join(f"{c} ({HOLLAND_CODES[c]['ar']})" for c in holland_codes if c in HOLLAND_CODES) or "غير محدد")
]
lines.append(f"- الفرع الثانوي: {track or 'لم يحدده'}")
lines.append(f"- لغات التدريس المفضلة: {', '.join(languages) or 'كل اللغات'}")
lines.append(f"- المحافظات المفضلة: {', '.join(governorates) or 'كل المحافظات'}")
return "\n".join(lines)
def why_not_in_results(major: Dict, holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str]) -> List[str]:
"""Which of the student's own criteria this major fails.
Turns "this major is not in your results" into a specific, checkable reason,
so the answer can be useful instead of merely apologetic.
"""
reasons = []
if track and major.get('allowed_tracks') and track not in major['allowed_tracks']:
reasons.append(
f"لا يقبل فرع {track}، بل {', '.join(major['allowed_tracks'])}"
)
if languages and major.get('teaching_languages') and not set(languages) & set(major['teaching_languages']):
reasons.append(f"لغة التدريس فيه {', '.join(major['teaching_languages'])}")
if governorates:
available = {loc.get('governorate') for loc in major.get('locations', []) if loc.get('governorate')}
if available and not available & set(governorates):
reasons.append(f"غير متوفر في {', '.join(governorates)} بل في {', '.join(sorted(available))}")
if holland_codes:
score = get_retriever().calculate_holland_score(
holland_codes, major.get('holland_codes', {})
)
if score < 50:
# Worded so the number cannot be lifted out and re-presented as a
# strong match; smaller models did exactly that with "منخفضة (34.9%)".
reasons.append(f"توافقه مع نمط شخصيته ضعيف: {score} من 100 فقط")
return reasons
# Arabic-Indic (٠-٩) and Eastern Arabic-Indic (۰-۹) digits, mapped to ASCII so
# "رقم ٣" and "رقم 3" resolve identically — an Arabic keyboard produces the
# former, and the app's own examples use it.
_DIGIT_MAP = str.maketrans(
"٠١٢٣٤٥٦٧٨٩" "۰۱۲۳۴۵۶۷۸۹",
"0123456789" "0123456789",
)
# A number only counts as a card reference when the student marked it as one,
# or the whole message is the number. Without this, "أريد 5 تخصصات" silently
# answered about result #5.
_ORDINAL_REF = re.compile(r'(?:رقم|الرقم|#)\s*(\d+)')
_NUMBER_ONLY = re.compile(r'\s*(\d+)\s*[؟?.!]?\s*')
def majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]:
"""Majors the student referred to, as (major, result_or_None) pairs.
result is the entry from the student's shortlist when the major is in it,
otherwise None — the caller uses that to explain the difference.
"""
found = []
seen_ids = set()
def add(major, result):
marker = (major.get('id'), major.get('name_ar'))
if marker not in seen_ids:
seen_ids.add(marker)
found.append((major, result))
# Reference by position ("رقم 3" / "رقم ٣"), with Arabic-Indic digits
# normalised first. A bare number is not an index — see _ORDINAL_REF.
normalized = message.translate(_DIGIT_MAP)
match = _ORDINAL_REF.search(normalized) or _NUMBER_ONLY.fullmatch(normalized)
if match:
index = int(match.group(1))
if 1 <= index <= len(visible):
result = visible[index - 1]
add(result.get('major', {}), result)
# Reference by name. Arabic has no case, so match the message as written.
# Longest names first so a specific major beats a shorter name nested in it.
by_name = {r['major'].get('name_ar'): r for r in visible if r.get('major')}
for major in sorted(load_majors(), key=lambda m: -len(m.get('name_ar', ''))):
if len(found) >= 3:
break
name = major.get('name_ar')
if name and name in message:
add(major, by_name.get(name))
return found
# --- Semantic fallback (embedding-based RAG) ---------------------------------
# Activates rag.retriever.KalimRetriever.retrieve_by_query — and therefore
# rag.vectorstore / rag.embeddings — which existed in the codebase but had no
# caller. Used only when the student's question does not name a major
# literally: majors_mentioned (exact name / card number) is tried first and
# always wins when it finds something, since an exact reference is more
# reliable than a similarity score.
SEMANTIC_SCORE_THRESHOLD = 0.35 # tune against real queries before shipping
SEMANTIC_MAX_RESULTS = 3
def semantic_majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]:
"""Majors found by embedding similarity rather than an exact name/number.
Returns the same (major, result_or_None) shape as majors_mentioned, so
callers do not need to branch on which path found the major.
"""
if len(message.strip()) < 8:
return []
try:
hits = get_retriever().retrieve_by_query(message, top_k=SEMANTIC_MAX_RESULTS)
except Exception:
# Missing index, missing optional deps (chromadb/sentence-transformers),
# or an unreachable embedding model must never break the chat — the
# caller already has answer_without_llm as a safe fallback.
logger.exception("Semantic search failed; continuing without it")
return []
by_id = {r['major'].get('id'): r for r in visible if r.get('major')}
found = []
for hit in hits:
if hit.get('semantic_score', 0) < SEMANTIC_SCORE_THRESHOLD:
continue
major = hit.get('major')
if not major:
continue
result = by_id.get(major.get('id'))
found.append((major, result))
return found
# --- Deterministic major answers ----------------------------------------------
# Used whenever no model is configured or its reply was rejected. It has to
# read like an answer, not a record: dumping every field of a major was the
# behaviour students saw as a non-answer.
# Which field a question is actually about. Cheap keyword cues, deliberately
# narrow — anything unmatched falls through to the general summary.
_FIELD_CUES = (
('admission_requirements', "شروط القبول",
('شرط', 'شروط', 'قبول', 'معدل', 'أقبل', 'اقبل', 'بقدر ادخل', 'بقدر أدخل')),
('career_opportunities', "فرص العمل",
('فرص', 'عمل', 'وظيف', 'مهن', 'شغل', 'مستقبل', 'راتب')),
('entry_mechanism', "آلية الدخول",
('آلية', 'الية', 'مباراة', 'دخول', 'تسجيل', 'امتحان')),
('teaching_languages', "لغات التدريس",
('لغة', 'لغات', 'بالعربي', 'بالانكليزي', 'بالفرنسي')),
('allowed_tracks', "الفروع المقبولة",
('فرعي', 'الفرع', 'فروع', 'ثانوي', 'بكالوريا')),
)
def _asked_fields(message: str) -> List[Tuple[str, str]]:
"""(field, label) pairs the question is about, in cue order."""
return [(field, label) for field, label, cues in _FIELD_CUES
if any(cue in message for cue in cues)]
def _first_sentence(text: Optional[str], limit: int = 220) -> str:
text = (text or "").strip()
if not text:
return ""
head = text.split('.')[0].strip()
if len(head) < 30 and '.' in text: # too short to stand alone
head = '.'.join(text.split('.')[:2]).strip()
if len(head) > limit:
head = head[:limit].rsplit(' ', 1)[0] + "…"
return head
def _major_facts_line(major: Dict) -> str:
bits = []
if major.get('faculty_ar'):
bits.append(f"📚 {major['faculty_ar']}")
if major.get('entry_mechanism'):
bits.append(f"🔑 {major['entry_mechanism']}")
govs = sorted({loc.get('governorate') for loc in major.get('locations', [])
if loc.get('governorate')})
if govs:
bits.append("📍 " + "، ".join(govs))
return " · ".join(bits)
def describe_major(major: Dict, result: Optional[Dict], visible: Optional[List[Dict]],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str],
message: str = "") -> str:
"""A readable answer about one major, built only from its record."""
name = major.get('name_ar', '')
parts = []
if result:
parts.append(
f"**{name}** ضمن التخصصات المقترحة لك، بنسبة توافق "
f"**{result.get('holland_score', 0)}%**."
)
reasons = result.get('match_reasons') or []
if reasons:
parts.append("لماذا يناسبك:\n" + "\n".join(f"- {reason}" for reason in reasons))
else:
reasons = why_not_in_results(major, holland_codes, track, languages, governorates)
parts.append(
f"**{name}** خارج نتائجك الحالية لأنه {reasons[0]}."
if reasons else f"**{name}** ليس ضمن التخصصات المقترحة لك حالياً."
)
# Answer the specific thing asked, when the question named one.
asked = _asked_fields(message)
for field, label in asked:
value = major.get(field)
if isinstance(value, list):
value = "، ".join(value)
if value:
parts.append(f"**{label}:** {value}")
if not asked:
summary = _first_sentence(major.get('description'))
if summary:
parts.append(summary)
facts = _major_facts_line(major)
if facts:
parts.append(facts)
if not result and visible:
best = visible[0]
parts.append(
f"أقرب بديل ضمن نتائجك: **{best.get('major', {}).get('name_ar', '')}** "
f"({best.get('holland_score', 0)}% توافق)."
)
parts.append(
"اضغط على بطاقة التخصص في القائمة لعرض كل التفاصيل، أو اسألني عن أي جانب آخر."
if result else "اسألني عن شروط قبوله أو فرص العمل فيه إن أردت."
)
return "\n\n".join(parts)
def answer_without_llm(mentioned: List[Tuple[Dict, Optional[Dict]]], visible: Optional[List[Dict]],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str],
message: str = "") -> str:
"""Deterministic reply used when no model is configured or its output failed.
This must still answer the most common question — "which majors suit me?" —
rather than telling the student to name a major, which was the behaviour
that made the assistant look broken whenever the LLM was unavailable.
"""
if not mentioned:
# "ما هي الشخصية الواقعية؟" is a question about a type, not a request
# for a shortlist — answering it with five majors reads as a non-answer.
asked_types = holland_types_mentioned(message) if message else []
if asked_types:
return explain_holland_types(asked_types, visible)
if visible:
lines = [
f"**{i}. {r['major'].get('name_ar', '')}** — {r['major'].get('faculty_ar', '')} "
f"({r.get('holland_score', 0)}% توافق)"
for i, r in enumerate(visible[:5], 1)
]
codes = ' - '.join(holland_codes)
return (
f"بناءً على نمط شخصيتك ({codes}) وفرعك الثانوي، هذه أقرب التخصصات إليك:\n\n"
+ "\n\n".join(lines)
+ "\n\nاكتب رقم أي تخصص أو اسمه لأعطيك تفاصيله الكاملة."
)
return "لم أجد نتائج بعد. اختر معايير البحث من الشريط الجانبي لأقترح عليك تخصصات مناسبة."
major, result = mentioned[0]
return describe_major(major, result, visible, holland_codes, track,
languages, governorates, message=message)
@dataclass
class ChatTurn:
"""One free-chat exchange: what to say, and the window now on screen.
`visible` is the shell's new display window. The engine owns the paging
decision so the two front ends cannot drift: the Gradio shell used to
re-run the MORE_KEYWORDS test itself (missing the engine's extra guards)
while the Streamlit shell never widened at all, so a follow-up "رقم 12"
resolved against a stale list.
"""
reply: str
visible: List[Dict]
def free_chat_reply(message: str, visible: List[Dict], all_results: List[Dict],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str], llm) -> ChatTurn:
"""Answer a free-form question about the student's results.
`visible` must be the filtered/sorted subset actually on screen, so numbered
references resolve to the same cards the student is looking at; `all_results`
is the pool the same filters produced, so a paged-in extra page cannot
contain majors the student filtered away.
Returns a ChatTurn — the shell must store `turn.visible` as its new window.
"""
if len(message) > MAX_QUESTION_CHARS:
return ChatTurn(_QUESTION_TOO_LONG, visible)
# Which majors the question is about. Resolved first so that a question
# naming a specific major is never swallowed by the paging branch below.
mentioned = majors_mentioned(message, visible)
# An exact name/number reference always wins over a similarity guess.
# Semantic search only runs when nothing was named literally, and only
# for the kind of free-text description embeddings are good at — not a
# short control phrase like "المزيد" or a bare "رقم 3", which the length
# guard inside semantic_majors_mentioned already screens out.
if not mentioned:
mentioned = semantic_majors_mentioned(message, visible)
# "Show me more majors like me". Testers reported the bot kept repeating the
# same shortlist, because the search itself was capped. The full ranked list
# is available now, so answer from the part the student has not seen yet.
if not mentioned and any(kw in message for kw in MORE_KEYWORDS) and len(all_results) > len(visible):
seen = {(r['major'].get('id'), r['major'].get('name_ar')) for r in visible}
extra = [r for r in all_results if (r['major'].get('id'), r['major'].get('name_ar')) not in seen]
if extra:
page = extra[:PAGE_SIZE]
reply = (
results_table(
page, start=len(visible) + 1,
title=f"نعم، هناك **{len(extra)} تخصصاً إضافياً** يتطابق مع شخصيتك. إليك التالية منها:",
)
+ "\n💡 اضغط \"عرض تخصصات إضافية\" أعلى مربع الكتابة لعرضها كبطاقات كاملة."
)
# The listed page is now on screen: hand it back so the shell's
# numbering and the student's next "رقم N" agree.
return ChatTurn(reply, list(visible) + page)
if any(kw in message for kw in TABLE_KEYWORDS) and visible:
return ChatTurn(results_table(visible), visible)
# "ما هي الشخصية الواقعية؟" — we hold the authoritative definition, so
# answer it from the data rather than hoping the model relays it. Same
# principle as the two branches above: never ask a model for something the
# dataset already answers exactly.
asked_types = holland_types_mentioned(message)
if asked_types and not mentioned and is_definition_question(message):
return ChatTurn(explain_holland_types(asked_types, visible), visible)
# Everything below is retrieval: how the mentioned majors relate to this
# student, then let the model phrase the answer. Answering straight from
# these lookups is what produced the data-dump replies testers found
# unfriendly.
fallback = answer_without_llm(mentioned, visible, holland_codes, track,
languages, governorates, message=message)
if llm is None:
return ChatTurn(fallback, visible)
context_blocks = []
for major, result in mentioned:
header = f"[التخصص الذي سأل عنه الطالب — {'ضمن نتائجه' if result else 'خارج نتائجه'}]"
block = f"{header}\n{major_context_lines(major, result.get('holland_score') if result else None)}"
if not result:
reasons = why_not_in_results(major, holland_codes, track, languages, governorates)
if reasons:
block += "\n سبب عدم ظهوره ضمن نتائج الطالب: " + "؛ ".join(reasons)
context_blocks.append(block)
# Always include the shortlist so the model can suggest a real alternative
# rather than inventing one.
if visible:
shortlist = "\n".join(
f"{i}. {major_context_lines(r.get('major', {}), r.get('holland_score', 0))}"
for i, r in enumerate(visible[:7], 1)
)
context_blocks.append(f"[التخصصات المقترحة للطالب حالياً]\n{shortlist}")
# Ground personality-type talk in the dataset's own definitions: the codes
# the question names, or failing that the student's own.
reference = holland_reference(holland_types_mentioned(message) or holland_codes)
if reference:
context_blocks.append(reference)
context = "\n\n".join(context_blocks)
prompt = FOLLOWUP_PROMPT.format(
student_profile=profile_summary(holland_codes, track, languages, governorates),
majors_data=context or "لا توجد بيانات مطابقة لسؤال الطالب.",
question=fence_question(message),
)
try:
reply = llm.generate(prompt)
except Exception:
logger.exception("LLM generation failed")
return ChatTurn(fallback, visible)
# Never show raw model output: a leaked script or a wrong-language answer
# is replaced by the data-driven reply, which is always correct if plainer.
reply = sanitize_reply(reply, fallback)
# The fallback is built from this same data, so it needs no checking — and
# checking it would reject its own shortlist names.
if reply is not fallback and not reply_consistent_with_context(reply, context):
return ChatTurn(fallback, visible)
return ChatTurn(reply, visible)