File size: 41,885 Bytes
bae15d1 d09d4f9 bae15d1 d09d4f9 6e08541 d09d4f9 6e08541 bae15d1 d09d4f9 bae15d1 6e08541 bae15d1 d09d4f9 6e08541 bae15d1 d09d4f9 bae15d1 d09d4f9 bae15d1 6e08541 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 | """Kalim advisor engine — every piece of chat/recommendation logic that does
not depend on a specific UI toolkit.
`app.py` (Streamlit) and `app_gradio.py` (Gradio, for Hugging Face Spaces)
are both thin shells over this module. Behaviour that the client validated —
result phrasing, the why-not-in-results reasons, reply sanitisation — lives
here exactly once, so the two front ends cannot drift apart the way the old
`src/chains/` fork did.
Nothing in this file may import streamlit or gradio. State (the student's
selections and current results) is passed in as plain arguments; each front
end owns its own session storage.
"""
import logging
import os
import re
from dataclasses import dataclass
from functools import lru_cache
from typing import Callable, Dict, List, Optional, Tuple
from config.settings import HOLLAND_CODES
from config.prompts import FOLLOWUP_PROMPT
from data.loader import load_majors
from rag.retriever import KalimRetriever, StudentProfile
logger = logging.getLogger("kalim")
# How many cards to reveal at a time. The retriever returns every major above
# the score threshold — often 60+ — and the student pages through them. Capping
# the search itself at 10 was the reason testers reported that majors visible in
# the source spreadsheet never appeared in the chatbot.
PAGE_SIZE = 10
MAX_RESULTS = 100
# Abuse controls for the public deployments, which are unauthenticated URLs
# whose chat path ends in a paid provider call. The length cap bounds what a
# visitor can push into the prompt; the per-session turn budget is enforced by
# each shell (session storage is shell-owned) — beyond it the shells pass
# llm=None and the student still gets the deterministic answer_without_llm
# reply. Per-session only: neither host offers IP-level limiting.
MAX_QUESTION_CHARS = 500
MAX_LLM_TURNS_PER_SESSION = 30
_QUESTION_TOO_LONG = "سؤالك طويل جداً. رجاءً اختصره إلى أقل من 500 حرف وأعد إرساله."
# "Show me more majors like me" triggers. Exported so a front end can grow its
# own visible window after the reply lists the next page.
#
# Bare comparatives (أكثر، اكثر، غيرها، إضافية) used to be in this list and
# hijacked ordinary questions — "أي تخصص أكثر طلباً؟" was answered with a
# pagination table instead of an answer. Same bug class as TABLE_KEYWORDS
# below. Only phrases that unambiguously ask for more results belong here.
MORE_KEYWORDS = [
'المزيد', 'مزيد من', 'تشبهني',
'تخصصات إضافية', 'تخصصات اضافية', 'تخصصات أخرى', 'تخصصات اخرى',
'تخصصات غيرها', 'اعرض أكثر', 'اعرض اكثر', 'show more', 'more',
]
# Only an explicit request for a table/list gets one. This used to also fire
# on "اعطيني" and "المطابقة", so "اعطيني تخصصات تناسب شخصيتي" — an ordinary
# request for advice — was answered with a bare table instead of an answer.
TABLE_KEYWORDS = ['جدول', 'قائمة', 'table', 'list']
@lru_cache(maxsize=1)
def get_retriever() -> KalimRetriever:
"""One retriever per process, so the majors list is parsed once."""
return KalimRetriever()
# --- LLM provider selection -------------------------------------------------
# An explicit cloud key wins, otherwise fall back to a local Ollama model.
# The secret getter is injected: Streamlit layers st.secrets over the
# environment, Gradio/HF Spaces use the environment alone.
def build_llm(get_secret: Callable[[str], str] = os.getenv):
"""Construct the best available LLM client, or None.
Backed by LangChain (llm.langchain_llm) rather than calling each
provider's SDK directly — this is the thesis's LangChain design
proposition, implemented in the code path a student's question
actually travels through.
LangChain is not installed everywhere this engine runs: requirements.txt
(the Hugging Face Space) carries it, requirements-deploy.txt (Streamlit
Community Cloud, Docker) does not have to. An ImportError therefore falls
back to the provider SDKs in llm.cloud_llm instead of taking the chat
down — the behaviour requirements.txt already claims in its comment
("cloud_llm.py is kept only as a documented fallback").
"""
try:
from llm.langchain_llm import build_langchain_llm
except ImportError:
logger.warning(
"LangChain is not installed; falling back to the provider SDKs in "
"llm.cloud_llm. Install langchain-core and the integration package "
"for your provider to use the LangChain path."
)
return _build_llm_via_sdks(get_secret)
return build_langchain_llm(get_secret=get_secret)
def _build_llm_via_sdks(get_secret: Callable[[str], str]):
"""Pre-LangChain provider selection, kept as the no-LangChain fallback."""
from llm.cloud_llm import PROVIDER_PRECEDENCE
for env_var, provider in PROVIDER_PRECEDENCE:
if not get_secret(env_var):
continue
try:
from llm.cloud_llm import CloudLLM
client = CloudLLM(provider, api_key=get_secret(env_var))
except Exception:
logger.exception("Failed to initialise cloud provider %s", provider)
continue
if client.is_available:
return client
# Key present but no usable client — say so instead of silently
# returning canned errors for every question.
logger.error(
"%s is configured but its client could not be created — either the "
"provider SDK is not installed (pip install %s) or client creation "
"failed; see the log above.", provider, provider,
)
try:
from llm.local_llm import get_llm as _get_local_llm
client = _get_local_llm()
except Exception:
logger.exception("Failed to initialise the local Ollama client")
return None
return client if client is not None and client.is_available else None
# --- Search ------------------------------------------------------------------
def run_search(profile: StudentProfile) -> List[Dict]:
"""Retrieve every matching major for the profile, best match first."""
return get_retriever().retrieve_with_expansion(
profile, top_k=MAX_RESULTS, min_results=PAGE_SIZE, score_threshold=50
)
def results_intro(results: List[Dict], holland_codes: List[str], interactive_cards: bool = True) -> str:
"""Chat message summarising a fresh result set.
`interactive_cards=True` is the Streamlit wording (expander cards plus the
filter row above them); False suits a front end that renders the results as
a numbered table in the chat, like the Gradio app.
"""
if not results:
return "لم أجد تخصصات تطابق معاييرك. جرّب تغيير الفلاتر من الشريط الجانبي."
student_codes = ' - '.join(holland_codes)
shown = min(len(results), PAGE_SIZE)
intro = f"وجدت **{len(results)} تخصصاً** يتطابق مع خياراتك ({student_codes})"
intro += f"، وأعرض لك أفضل {shown} منها.\n\n" if len(results) > shown else ".\n\n"
# The retriever widens the search by re-scoring under every ordering of the
# student's codes when too few majors clear the threshold. Say so, since
# the percentages shown still reflect the order the student gave.
if any(r.get('expanded') for r in results):
intro += (
"🔎 لقلة النتائج المطابقة تماماً، وسّعت البحث ليشمل تخصصات قريبة من "
"أنماط شخصيتك بترتيب مختلف؛ النسب المعروضة تعكس ترتيبك أنت.\n\n"
)
if not interactive_cards:
return (
intro
+ ("يمكنك الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد.\n\n" if len(results) > shown else "")
+ "💡 اضغط على أي تخصص في القائمة أدناه لعرض تفاصيله الكاملة، "
+ "أو اكتب رقمه أو اسمه في مربع الكتابة لتسألني عنه."
)
return (
intro
+ "يمكنك تصفية النتائج أدناه بحسب رمز الشخصية، الكلية، أو نسبة المطابقة"
+ ("، أو الضغط على \"عرض تخصصات إضافية\" لرؤية المزيد" if len(results) > shown else "")
+ ".\n\n💡 اضغط على أي بطاقة لعرض تفاصيل التخصص، أو اكتب رقمه في الأسفل لتسألني عنه."
)
def results_table(results: List[Dict], start: int = 1, title: str = "**جدول التخصصات المقترحة:**") -> str:
"""Markdown table of a result window, numbered from `start`."""
table = "| # | التخصص | رمز هولاند | النسبة |\n|---|--------|------------|--------|\n"
for i, r in enumerate(results, start):
major = r.get('major', {})
table += (
f"| {i} | {major.get('name_ar', '')} "
f"| {major.get('holland_codes', {}).get('triplet', '')} "
f"| {r.get('holland_score', 0)}% |\n"
)
return f"{title}\n\n{table}"
def kalim_recommendation(results: List[Dict], holland_codes: List[str],
track: Optional[str], governorates: List[str]) -> str:
"""Kalim's personalized recommendation based on Holland, Track, and Location."""
if not results:
return ""
top = results[0]
major = top.get('major', {})
score = top.get('holland_score', 0)
holland = major.get('holland_codes', {}).get('triplet', '')
student_codes = '-'.join(holland_codes)
track = track or "الكل"
location = ', '.join(governorates) if governorates else "الكل"
rec = f"🎯 **توصية كليم:**\n\n"
rec += f"بناءً على:\n"
rec += f"• شخصيتك المهنية: **{student_codes}**\n"
rec += f"• فرعك الثانوي: **{track}**\n"
rec += f"• موقعك: **{location}**\n\n"
rec += f"أنصحك بتخصص **{major.get('name_ar', '')}**\n\n"
rec += f"✅ توافق هولاند: **{score}%** (التخصص: {holland})\n"
# Only claim what was checked. The track line used to print
# unconditionally — even when the student skipped the track step — and the
# location line printed the student's whole selection, though the location
# filter is a union: one matching branch is enough to be in the results.
if track != "الكل":
rec += f"✅ يقبل فرعك الثانوي ({track})\n"
if governorates:
major_govs = {loc.get('governorate') for loc in major.get('locations', [])
if loc.get('governorate')}
available_here = [g for g in governorates if g in major_govs]
if available_here:
rec += f"✅ متوفر في {', '.join(available_here)}\n"
rec += f"\n📚 {major.get('faculty_ar', '')}"
return rec
def compare_top3(results: List[Dict]) -> str:
"""Short comparison of the three best matches."""
if not results:
return ""
comparison = "**مقارنة بين أفضل 3 تخصصات:**\n\n"
for i, r in enumerate(results[:3], 1):
major = r.get('major', {})
comparison += f"**{i}. {major.get('name_ar', '')}**\n"
comparison += f"- المطابقة: {r.get('holland_score', 0)}%\n"
comparison += f"- الكلية: {major.get('faculty_ar', '')}\n\n"
return comparison
# --- Major formatting ---------------------------------------------------------
def format_major_details(major: Dict) -> str:
"""Format major details from the dataset for display."""
parts = [f"**{major.get('name_ar', '')}**"]
if major.get('name_en'):
parts.append(f"*{major['name_en']}*")
parts.append("")
parts.append(f"📚 **الكلية:** {major.get('faculty_ar', 'غير متوفر')}")
if major.get('holland_interpretation'):
parts.append(f"🧠 **تفسير هولاند:** {major['holland_interpretation']}")
parts.append(f"📝 **الوصف:** {major.get('description', 'غير متوفر')}")
parts.append(f"💼 **فرص العمل:** {major.get('career_opportunities', 'غير متوفر')}")
parts.append(f"📋 **متطلبات القبول:** {major.get('admission_requirements', 'غير متوفر')}")
parts.append(f"🗣️ **لغات التدريس:** {', '.join(major.get('teaching_languages', []))}")
if major.get('entry_mechanism'):
parts.append(f"🔑 **آلية الدخول:** {major['entry_mechanism']}")
if major.get('locations'):
branches = [loc.get('branch', '') for loc in major['locations'] if loc.get('branch')]
if branches:
parts.append(f"📍 **الفروع:** {' | '.join(branches)}")
return "\n\n".join(parts)
def major_context_lines(major: Dict, score=None) -> str:
"""Compact plain-text rendering of a major, used to build LLM context."""
lines = [f"{major.get('name_ar', '')} ({major.get('name_en', '')})"]
lines.append(f" الكلية: {major.get('faculty_ar', '')}")
lines.append(f" رمز هولاند: {major.get('holland_codes', {}).get('triplet', '')}")
if score is not None:
lines.append(f" نسبة المطابقة: {score}%")
for label, key in (
("الوصف", 'description'),
("فرص العمل", 'career_opportunities'),
("متطلبات القبول", 'admission_requirements'),
("آلية الدخول", 'entry_mechanism'),
):
if major.get(key):
lines.append(f" {label}: {major[key]}")
lines.append(f" لغات التدريس: {', '.join(major.get('teaching_languages', []))}")
return "\n".join(lines)
# --- Reply sanitisation --------------------------------------------------------
# Scripts that must never reach a student. Qwen and other multilingual models
# occasionally leak CJK characters into Arabic output, and private-use or
# replacement code points show up when generation goes wrong.
_UNWANTED_SCRIPTS = re.compile(
'['
' -〿' # CJK punctuation
'-ヿ' # hiragana, katakana
'ㇰ-ㇿ' # katakana extensions
'㐀-䶿' # CJK unified ideographs extension A
'一-鿿' # CJK unified ideographs
'가-' # hangul syllables
'豈-' # CJK compatibility ideographs
'-' # private use area
'�' # replacement character
']+'
)
_ARABIC_LETTER = re.compile(r'[-ۿ]')
_LATIN_LETTER = re.compile(r'[A-Za-z]')
# Reasoning models wrap their chain of thought in <think> tags, or announce it
# in a preamble. cloud_llm asks the provider to suppress it, but that switch is
# model-specific and silently absent on some, so strip it here as well.
_THINK_BLOCK = re.compile(r'<think>.*?</think>|<thinking>.*?</thinking>', re.DOTALL | re.IGNORECASE)
_OPEN_THINK = re.compile(r'<think(?:ing)?>', re.IGNORECASE)
_CLOSE_THINK = re.compile(r'</think(?:ing)?>', re.IGNORECASE)
_THINK_PREAMBLE = re.compile(
r"^\s*(?:here'?s?\s+(?:a|my|the)\s+thinking\s+process|"
r"let me think|thinking process|reasoning)\s*:?.*?(?=\n\s*\n|\Z)",
re.IGNORECASE | re.DOTALL,
)
def strip_reasoning(text: str) -> str:
"""Remove any chain-of-thought the model exposed despite being asked not to."""
cleaned = _THINK_BLOCK.sub('', text)
# Some providers strip the opening tag and leave the closing one: whatever
# follows the last closing tag is the answer.
if _CLOSE_THINK.search(cleaned):
cleaned = _CLOSE_THINK.split(cleaned)[-1]
# An unclosed opening tag means the answer never arrived; dropping it leaves
# an empty reply, which the caller replaces with the data-driven fallback.
cleaned = _OPEN_THINK.split(cleaned)[0]
cleaned = _THINK_PREAMBLE.sub('', cleaned)
return cleaned.strip()
def sanitize_reply(text: str, fallback: str) -> str:
"""Drop stray foreign-script characters, or reject the reply outright.
A few leaked characters are cleaned silently. A reply that is largely the
wrong script is treated as a failed generation and replaced by the
data-driven fallback, because a student should never be shown one.
"""
text = strip_reasoning(text or "")
if not text.strip():
logger.warning("LLM returned an empty reply")
return fallback
cleaned = _UNWANTED_SCRIPTS.sub('', text)
removed = len(text) - len(cleaned)
if removed > max(4, len(text) * 0.02):
logger.warning("LLM reply contained %d foreign-script characters; discarded", removed)
return fallback
# The app is Arabic-only. Latin is fine for a major's English name, but a
# reply that is mostly Latin means the model answered in the wrong language.
arabic = len(_ARABIC_LETTER.findall(cleaned))
latin = len(_LATIN_LETTER.findall(cleaned))
if arabic + latin > 40 and arabic < (arabic + latin) * 0.4:
logger.warning("LLM replied mostly in Latin script (%d Arabic vs %d Latin)", arabic, latin)
return fallback
# Tidy whitespace left behind by removed characters.
return re.sub(r'[ \t]{2,}', ' ', cleaned).strip()
# --- Holland (RIASEC) questions ------------------------------------------------
# Students ask what a personality type *is*, not only which majors match it.
# The authoritative Arabic definitions live in config.settings, so they are fed
# to the model as context: left to its own knowledge a smaller model will
# happily define الواقعي as التقليدي.
_HOLLAND_BY_NAME = {info['ar']: code for code, info in HOLLAND_CODES.items()}
def holland_types_mentioned(message: str) -> List[str]:
"""RIASEC codes the question is about, by Arabic name or standalone letter."""
found = []
for name, code in _HOLLAND_BY_NAME.items():
if name in message and code not in found:
found.append(code)
for code in HOLLAND_CODES:
if code not in found and re.search(rf'(?<![A-Za-z]){code}(?![A-Za-z])', message):
found.append(code)
return found
# "What does this type mean?" — a definition request, not a request for majors.
_DEFINITION_CUES = (
'ما هي', 'ما هو', 'ماهي', 'ماهو', 'شو هي', 'شو هو', 'شو يعني',
'ماذا يعني', 'ماذا تعني', 'معنى', 'يعني', 'تعني', 'عرّف', 'عرف',
'اشرح', 'وضح', 'وضّح', 'ما معنى',
)
def is_definition_question(message: str) -> bool:
return any(cue in message for cue in _DEFINITION_CUES)
def holland_reference(codes: List[str]) -> str:
"""Context block describing the given RIASEC codes, from the dataset's own
definitions rather than the model's memory."""
lines = [
f"{code} — {HOLLAND_CODES[code]['ar']} ({HOLLAND_CODES[code]['en']}): "
f"{HOLLAND_CODES[code]['description_ar']}"
for code in codes if code in HOLLAND_CODES
]
if not lines:
return ""
return "[مرجع أنماط هولاند RIASEC — تعريفات معتمدة]\n" + "\n".join(lines)
def explain_holland_types(codes: List[str], visible: Optional[List[Dict]]) -> str:
"""Deterministic answer to "what is the <type> personality?"."""
parts = []
for code in codes:
info = HOLLAND_CODES[code]
parts.append(f"**{info['ar']} ({code} — {info['en']})**\n{info['description_ar']}")
answer = "\n\n".join(parts)
if visible:
matching = [
r for r in visible
if r.get('major', {}).get('holland_codes', {}).get('primary') in codes
]
if matching:
names = "، ".join(r['major'].get('name_ar', '') for r in matching[:5])
answer += f"\n\nمن بين التخصصات المقترحة لك، الأقرب إلى هذا النمط: {names}."
return answer
# --- Prompt boundary and reply validation -------------------------------------
# The student's message is interpolated into FOLLOWUP_PROMPT. Left unfenced, a
# message can reproduce the template's own section headers and impersonate the
# database block, getting Kalim to state invented admission requirements in an
# authoritative voice. Two defences, both here so both front ends inherit them.
_PROMPT_STRUCTURE_MARKERS = (
"معلومات الطالب:", "معلومات من قاعدة البيانات", "سؤال الطالب",
"<<<سؤال_الطالب", "سؤال_الطالب>>>",
"[التخصص الذي سأل عنه الطالب", "[التخصصات المقترحة للطالب",
)
_PERCENT_RE = re.compile(r'(\d+(?:[.,]\d+)?)\s*[%٪]')
def fence_question(message: str) -> str:
"""Drop lines that impersonate the prompt's own structure."""
kept = [
line for line in message.splitlines()
if not any(marker in line for marker in _PROMPT_STRUCTURE_MARKERS)
]
return "\n".join(kept).strip()
def _percent_keys(text: str) -> set:
"""Percentages in `text`, as exact strings plus rounded integers.
Rounding is tolerated so a model that says "81%" for a context value of
81.4% is not treated as inventing a number.
"""
keys = set()
for raw in _PERCENT_RE.findall(text):
keys.add(raw)
try:
keys.add(str(round(float(raw.replace(',', '.')))))
except ValueError:
pass
return keys
def reply_consistent_with_context(reply: str, context: str) -> bool:
"""Whether the reply stays inside the data it was given.
Checks the two claim types a student would act on: match percentages and
major names. Anything asserted that is absent from the retrieved context
means the model invented it (or was steered into doing so), and the caller
shows the deterministic answer instead. Conservative by design — the cost
of a false negative is a plainer reply, the cost of a false positive is a
confidently wrong admission requirement.
"""
allowed_percents = _percent_keys(context)
for value in _percent_keys(reply):
if value not in allowed_percents:
logger.warning("LLM reply asserted a percentage absent from context")
return False
for major in load_majors():
name = major.get('name_ar')
if name and name in reply and name not in context:
logger.warning("LLM reply named a major absent from context: %s", name)
return False
return True
# --- Free chat -------------------------------------------------------------------
def profile_summary(holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str]) -> str:
"""The student's answers, written out for the model."""
lines = [
"- نمط الشخصية (هولاند): "
+ (", ".join(f"{c} ({HOLLAND_CODES[c]['ar']})" for c in holland_codes if c in HOLLAND_CODES) or "غير محدد")
]
lines.append(f"- الفرع الثانوي: {track or 'لم يحدده'}")
lines.append(f"- لغات التدريس المفضلة: {', '.join(languages) or 'كل اللغات'}")
lines.append(f"- المحافظات المفضلة: {', '.join(governorates) or 'كل المحافظات'}")
return "\n".join(lines)
def why_not_in_results(major: Dict, holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str]) -> List[str]:
"""Which of the student's own criteria this major fails.
Turns "this major is not in your results" into a specific, checkable reason,
so the answer can be useful instead of merely apologetic.
"""
reasons = []
if track and major.get('allowed_tracks') and track not in major['allowed_tracks']:
reasons.append(
f"لا يقبل فرع {track}، بل {', '.join(major['allowed_tracks'])}"
)
if languages and major.get('teaching_languages') and not set(languages) & set(major['teaching_languages']):
reasons.append(f"لغة التدريس فيه {', '.join(major['teaching_languages'])}")
if governorates:
available = {loc.get('governorate') for loc in major.get('locations', []) if loc.get('governorate')}
if available and not available & set(governorates):
reasons.append(f"غير متوفر في {', '.join(governorates)} بل في {', '.join(sorted(available))}")
if holland_codes:
score = get_retriever().calculate_holland_score(
holland_codes, major.get('holland_codes', {})
)
if score < 50:
# Worded so the number cannot be lifted out and re-presented as a
# strong match; smaller models did exactly that with "منخفضة (34.9%)".
reasons.append(f"توافقه مع نمط شخصيته ضعيف: {score} من 100 فقط")
return reasons
# Arabic-Indic (٠-٩) and Eastern Arabic-Indic (۰-۹) digits, mapped to ASCII so
# "رقم ٣" and "رقم 3" resolve identically — an Arabic keyboard produces the
# former, and the app's own examples use it.
_DIGIT_MAP = str.maketrans(
"٠١٢٣٤٥٦٧٨٩" "۰۱۲۳۴۵۶۷۸۹",
"0123456789" "0123456789",
)
# A number only counts as a card reference when the student marked it as one,
# or the whole message is the number. Without this, "أريد 5 تخصصات" silently
# answered about result #5.
_ORDINAL_REF = re.compile(r'(?:رقم|الرقم|#)\s*(\d+)')
_NUMBER_ONLY = re.compile(r'\s*(\d+)\s*[؟?.!]?\s*')
def majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]:
"""Majors the student referred to, as (major, result_or_None) pairs.
result is the entry from the student's shortlist when the major is in it,
otherwise None — the caller uses that to explain the difference.
"""
found = []
seen_ids = set()
def add(major, result):
marker = (major.get('id'), major.get('name_ar'))
if marker not in seen_ids:
seen_ids.add(marker)
found.append((major, result))
# Reference by position ("رقم 3" / "رقم ٣"), with Arabic-Indic digits
# normalised first. A bare number is not an index — see _ORDINAL_REF.
normalized = message.translate(_DIGIT_MAP)
match = _ORDINAL_REF.search(normalized) or _NUMBER_ONLY.fullmatch(normalized)
if match:
index = int(match.group(1))
if 1 <= index <= len(visible):
result = visible[index - 1]
add(result.get('major', {}), result)
# Reference by name. Arabic has no case, so match the message as written.
# Longest names first so a specific major beats a shorter name nested in it.
by_name = {r['major'].get('name_ar'): r for r in visible if r.get('major')}
for major in sorted(load_majors(), key=lambda m: -len(m.get('name_ar', ''))):
if len(found) >= 3:
break
name = major.get('name_ar')
if name and name in message:
add(major, by_name.get(name))
return found
# --- Semantic fallback (embedding-based RAG) ---------------------------------
# Activates rag.retriever.KalimRetriever.retrieve_by_query — and therefore
# rag.vectorstore / rag.embeddings — which existed in the codebase but had no
# caller. Used only when the student's question does not name a major
# literally: majors_mentioned (exact name / card number) is tried first and
# always wins when it finds something, since an exact reference is more
# reliable than a similarity score.
SEMANTIC_SCORE_THRESHOLD = 0.35 # tune against real queries before shipping
SEMANTIC_MAX_RESULTS = 3
def semantic_majors_mentioned(message: str, visible: List[Dict]) -> List[Tuple[Dict, Optional[Dict]]]:
"""Majors found by embedding similarity rather than an exact name/number.
Returns the same (major, result_or_None) shape as majors_mentioned, so
callers do not need to branch on which path found the major.
"""
if len(message.strip()) < 8:
return []
try:
hits = get_retriever().retrieve_by_query(message, top_k=SEMANTIC_MAX_RESULTS)
except Exception:
# Missing index, missing optional deps (chromadb/sentence-transformers),
# or an unreachable embedding model must never break the chat — the
# caller already has answer_without_llm as a safe fallback.
logger.exception("Semantic search failed; continuing without it")
return []
by_id = {r['major'].get('id'): r for r in visible if r.get('major')}
found = []
for hit in hits:
if hit.get('semantic_score', 0) < SEMANTIC_SCORE_THRESHOLD:
continue
major = hit.get('major')
if not major:
continue
result = by_id.get(major.get('id'))
found.append((major, result))
return found
# --- Deterministic major answers ----------------------------------------------
# Used whenever no model is configured or its reply was rejected. It has to
# read like an answer, not a record: dumping every field of a major was the
# behaviour students saw as a non-answer.
# Which field a question is actually about. Cheap keyword cues, deliberately
# narrow — anything unmatched falls through to the general summary.
_FIELD_CUES = (
('admission_requirements', "شروط القبول",
('شرط', 'شروط', 'قبول', 'معدل', 'أقبل', 'اقبل', 'بقدر ادخل', 'بقدر أدخل')),
('career_opportunities', "فرص العمل",
('فرص', 'عمل', 'وظيف', 'مهن', 'شغل', 'مستقبل', 'راتب')),
('entry_mechanism', "آلية الدخول",
('آلية', 'الية', 'مباراة', 'دخول', 'تسجيل', 'امتحان')),
('teaching_languages', "لغات التدريس",
('لغة', 'لغات', 'بالعربي', 'بالانكليزي', 'بالفرنسي')),
('allowed_tracks', "الفروع المقبولة",
('فرعي', 'الفرع', 'فروع', 'ثانوي', 'بكالوريا')),
)
def _asked_fields(message: str) -> List[Tuple[str, str]]:
"""(field, label) pairs the question is about, in cue order."""
return [(field, label) for field, label, cues in _FIELD_CUES
if any(cue in message for cue in cues)]
def _first_sentence(text: Optional[str], limit: int = 220) -> str:
text = (text or "").strip()
if not text:
return ""
head = text.split('.')[0].strip()
if len(head) < 30 and '.' in text: # too short to stand alone
head = '.'.join(text.split('.')[:2]).strip()
if len(head) > limit:
head = head[:limit].rsplit(' ', 1)[0] + "…"
return head
def _major_facts_line(major: Dict) -> str:
bits = []
if major.get('faculty_ar'):
bits.append(f"📚 {major['faculty_ar']}")
if major.get('entry_mechanism'):
bits.append(f"🔑 {major['entry_mechanism']}")
govs = sorted({loc.get('governorate') for loc in major.get('locations', [])
if loc.get('governorate')})
if govs:
bits.append("📍 " + "، ".join(govs))
return " · ".join(bits)
def describe_major(major: Dict, result: Optional[Dict], visible: Optional[List[Dict]],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str],
message: str = "") -> str:
"""A readable answer about one major, built only from its record."""
name = major.get('name_ar', '')
parts = []
if result:
parts.append(
f"**{name}** ضمن التخصصات المقترحة لك، بنسبة توافق "
f"**{result.get('holland_score', 0)}%**."
)
reasons = result.get('match_reasons') or []
if reasons:
parts.append("لماذا يناسبك:\n" + "\n".join(f"- {reason}" for reason in reasons))
else:
reasons = why_not_in_results(major, holland_codes, track, languages, governorates)
parts.append(
f"**{name}** خارج نتائجك الحالية لأنه {reasons[0]}."
if reasons else f"**{name}** ليس ضمن التخصصات المقترحة لك حالياً."
)
# Answer the specific thing asked, when the question named one.
asked = _asked_fields(message)
for field, label in asked:
value = major.get(field)
if isinstance(value, list):
value = "، ".join(value)
if value:
parts.append(f"**{label}:** {value}")
if not asked:
summary = _first_sentence(major.get('description'))
if summary:
parts.append(summary)
facts = _major_facts_line(major)
if facts:
parts.append(facts)
if not result and visible:
best = visible[0]
parts.append(
f"أقرب بديل ضمن نتائجك: **{best.get('major', {}).get('name_ar', '')}** "
f"({best.get('holland_score', 0)}% توافق)."
)
parts.append(
"اضغط على بطاقة التخصص في القائمة لعرض كل التفاصيل، أو اسألني عن أي جانب آخر."
if result else "اسألني عن شروط قبوله أو فرص العمل فيه إن أردت."
)
return "\n\n".join(parts)
def answer_without_llm(mentioned: List[Tuple[Dict, Optional[Dict]]], visible: Optional[List[Dict]],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str],
message: str = "") -> str:
"""Deterministic reply used when no model is configured or its output failed.
This must still answer the most common question — "which majors suit me?" —
rather than telling the student to name a major, which was the behaviour
that made the assistant look broken whenever the LLM was unavailable.
"""
if not mentioned:
# "ما هي الشخصية الواقعية؟" is a question about a type, not a request
# for a shortlist — answering it with five majors reads as a non-answer.
asked_types = holland_types_mentioned(message) if message else []
if asked_types:
return explain_holland_types(asked_types, visible)
if visible:
lines = [
f"**{i}. {r['major'].get('name_ar', '')}** — {r['major'].get('faculty_ar', '')} "
f"({r.get('holland_score', 0)}% توافق)"
for i, r in enumerate(visible[:5], 1)
]
codes = ' - '.join(holland_codes)
return (
f"بناءً على نمط شخصيتك ({codes}) وفرعك الثانوي، هذه أقرب التخصصات إليك:\n\n"
+ "\n\n".join(lines)
+ "\n\nاكتب رقم أي تخصص أو اسمه لأعطيك تفاصيله الكاملة."
)
return "لم أجد نتائج بعد. اختر معايير البحث من الشريط الجانبي لأقترح عليك تخصصات مناسبة."
major, result = mentioned[0]
return describe_major(major, result, visible, holland_codes, track,
languages, governorates, message=message)
@dataclass
class ChatTurn:
"""One free-chat exchange: what to say, and the window now on screen.
`visible` is the shell's new display window. The engine owns the paging
decision so the two front ends cannot drift: the Gradio shell used to
re-run the MORE_KEYWORDS test itself (missing the engine's extra guards)
while the Streamlit shell never widened at all, so a follow-up "رقم 12"
resolved against a stale list.
"""
reply: str
visible: List[Dict]
def free_chat_reply(message: str, visible: List[Dict], all_results: List[Dict],
holland_codes: List[str], track: Optional[str],
languages: List[str], governorates: List[str], llm) -> ChatTurn:
"""Answer a free-form question about the student's results.
`visible` must be the filtered/sorted subset actually on screen, so numbered
references resolve to the same cards the student is looking at; `all_results`
is the pool the same filters produced, so a paged-in extra page cannot
contain majors the student filtered away.
Returns a ChatTurn — the shell must store `turn.visible` as its new window.
"""
if len(message) > MAX_QUESTION_CHARS:
return ChatTurn(_QUESTION_TOO_LONG, visible)
# Which majors the question is about. Resolved first so that a question
# naming a specific major is never swallowed by the paging branch below.
mentioned = majors_mentioned(message, visible)
# An exact name/number reference always wins over a similarity guess.
# Semantic search only runs when nothing was named literally, and only
# for the kind of free-text description embeddings are good at — not a
# short control phrase like "المزيد" or a bare "رقم 3", which the length
# guard inside semantic_majors_mentioned already screens out.
if not mentioned:
mentioned = semantic_majors_mentioned(message, visible)
# "Show me more majors like me". Testers reported the bot kept repeating the
# same shortlist, because the search itself was capped. The full ranked list
# is available now, so answer from the part the student has not seen yet.
if not mentioned and any(kw in message for kw in MORE_KEYWORDS) and len(all_results) > len(visible):
seen = {(r['major'].get('id'), r['major'].get('name_ar')) for r in visible}
extra = [r for r in all_results if (r['major'].get('id'), r['major'].get('name_ar')) not in seen]
if extra:
page = extra[:PAGE_SIZE]
reply = (
results_table(
page, start=len(visible) + 1,
title=f"نعم، هناك **{len(extra)} تخصصاً إضافياً** يتطابق مع شخصيتك. إليك التالية منها:",
)
+ "\n💡 اضغط \"عرض تخصصات إضافية\" أعلى مربع الكتابة لعرضها كبطاقات كاملة."
)
# The listed page is now on screen: hand it back so the shell's
# numbering and the student's next "رقم N" agree.
return ChatTurn(reply, list(visible) + page)
if any(kw in message for kw in TABLE_KEYWORDS) and visible:
return ChatTurn(results_table(visible), visible)
# "ما هي الشخصية الواقعية؟" — we hold the authoritative definition, so
# answer it from the data rather than hoping the model relays it. Same
# principle as the two branches above: never ask a model for something the
# dataset already answers exactly.
asked_types = holland_types_mentioned(message)
if asked_types and not mentioned and is_definition_question(message):
return ChatTurn(explain_holland_types(asked_types, visible), visible)
# Everything below is retrieval: how the mentioned majors relate to this
# student, then let the model phrase the answer. Answering straight from
# these lookups is what produced the data-dump replies testers found
# unfriendly.
fallback = answer_without_llm(mentioned, visible, holland_codes, track,
languages, governorates, message=message)
if llm is None:
return ChatTurn(fallback, visible)
context_blocks = []
for major, result in mentioned:
header = f"[التخصص الذي سأل عنه الطالب — {'ضمن نتائجه' if result else 'خارج نتائجه'}]"
block = f"{header}\n{major_context_lines(major, result.get('holland_score') if result else None)}"
if not result:
reasons = why_not_in_results(major, holland_codes, track, languages, governorates)
if reasons:
block += "\n سبب عدم ظهوره ضمن نتائج الطالب: " + "؛ ".join(reasons)
context_blocks.append(block)
# Always include the shortlist so the model can suggest a real alternative
# rather than inventing one.
if visible:
shortlist = "\n".join(
f"{i}. {major_context_lines(r.get('major', {}), r.get('holland_score', 0))}"
for i, r in enumerate(visible[:7], 1)
)
context_blocks.append(f"[التخصصات المقترحة للطالب حالياً]\n{shortlist}")
# Ground personality-type talk in the dataset's own definitions: the codes
# the question names, or failing that the student's own.
reference = holland_reference(holland_types_mentioned(message) or holland_codes)
if reference:
context_blocks.append(reference)
context = "\n\n".join(context_blocks)
prompt = FOLLOWUP_PROMPT.format(
student_profile=profile_summary(holland_codes, track, languages, governorates),
majors_data=context or "لا توجد بيانات مطابقة لسؤال الطالب.",
question=fence_question(message),
)
try:
reply = llm.generate(prompt)
except Exception:
logger.exception("LLM generation failed")
return ChatTurn(fallback, visible)
# Never show raw model output: a leaked script or a wrong-language answer
# is replaced by the data-driven reply, which is always correct if plainer.
reply = sanitize_reply(reply, fallback)
# The fallback is built from this same data, so it needs no checking — and
# checking it would reject its own shortlist names.
if reply is not fallback and not reply_consistent_with_context(reply, context):
return ChatTurn(fallback, visible)
return ChatTurn(reply, visible)
|