Download app.py from Amyneee/LegalBot: direct link, hf CLI and curl.
- Browser
- Download file 140 kB
-
https://huggingface.co/spaces/Amyneee/LegalBot/resolve/main/app.py
- Command line
-
hf download hf://spaces/Amyneee/LegalBot/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Amyneee/LegalBot/resolve/main/app.py
140 kB
| import re | |
| import json | |
| import os | |
| from dotenv import load_dotenv | |
| from pymongo import MongoClient | |
| from openai import OpenAI, AzureOpenAI | |
| import numpy as np | |
| from sklearn.metrics.pairwise import cosine_similarity | |
| from typing import List, Dict, Tuple, Optional, Any, Set | |
| import logging | |
| from functools import lru_cache | |
| from dataclasses import dataclass, field | |
| from datetime import datetime | |
| import hashlib | |
| from enum import Enum | |
| import time | |
| import math | |
| from collections import Counter | |
| import asyncio | |
| import chainlit as cl | |
| # Charger les variables d'environnement | |
| load_dotenv() | |
| logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') | |
| logger = logging.getLogger(__name__) | |
| # ============================================================ | |
| # CONFIGURATION PROFESSIONNELLE AVEC NOMS EXACTS | |
| # ============================================================ | |
| class Config: | |
| """Professional configuration for all Tunisian codes""" | |
| MONGO_URI = os.getenv("MONGO_URI") | |
| DB_CODE = os.getenv("DB_CODE") | |
| COLLECTIONS = { | |
| "CSP": os.getenv("CSP", "csp"), | |
| "DROITS_REELS": os.getenv("DROITS_REELS", "code_droits_reels"), | |
| "OBLIGATIONS_CONTRATS": os.getenv("OBLIGATIONS_CONTRATS", "code_des_obligations_et_des_contrats"), | |
| "PROCEDURE_CIVILE": os.getenv("PROCEDURE_CIVILE", "code_de_procédure_civile_et_commerciale"), | |
| "CODE_TRAVAIL": os.getenv("CODE_TRAVAIL", "code-de-travail"), | |
| "DROITS_PROCEDURES_FISCAUX": os.getenv("DROITS_PROCEDURES_FISCAUX", "code-des-droits-et-procedures-fiscaux"), | |
| "DROIT_INTERNATIONAL_PRIVE": os.getenv("DROIT_INTERNATIONAL_PRIVE", "code-du-droit-international-privé"), | |
| "CODE_PENAL": os.getenv("CODE_PENAL", "code-pénal"), | |
| "CODE_COMMERCE": os.getenv("CODE_COMMERCE", "code_de_commerce"), | |
| "PROCEDURES_PENALES": os.getenv("PROCEDURES_PENALES", "code_des_procedures_penales") | |
| } | |
| DB_JURIS = os.getenv("DB_JURIS") | |
| COL_JURIS = os.getenv("COL_JURIS") | |
| AZURE_ENDPOINT = os.getenv("AZURE_ENDPOINT") | |
| AZURE_API_KEY = os.getenv("AZURE_API_KEY") | |
| AZURE_API_VERSION = os.getenv("AZURE_API_VERSION") | |
| EMBEDDING_MODEL = os.getenv("EMBEDDING_MODEL") | |
| CHAT_MODEL = os.getenv("CHAT_MODEL") | |
| RELEVANCE_THRESHOLDS = { | |
| "CSP": {"HIGH": 0.78, "MEDIUM": 0.65, "MINIMUM": 0.55}, | |
| "DROITS_REELS": {"HIGH": 0.75, "MEDIUM": 0.62, "MINIMUM": 0.52}, | |
| "OBLIGATIONS_CONTRATS": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50}, | |
| "PROCEDURE_CIVILE": {"HIGH": 0.65, "MEDIUM": 0.55, "MINIMUM": 0.45}, | |
| "CODE_TRAVAIL": {"HIGH": 0.73, "MEDIUM": 0.61, "MINIMUM": 0.51}, | |
| "DROITS_PROCEDURES_FISCAUX": {"HIGH": 0.68, "MEDIUM": 0.56, "MINIMUM": 0.46}, | |
| "DROIT_INTERNATIONAL_PRIVE": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50}, | |
| "CODE_PENAL": {"HIGH": 0.76, "MEDIUM": 0.64, "MINIMUM": 0.54}, | |
| "CODE_COMMERCE": {"HIGH": 0.71, "MEDIUM": 0.59, "MINIMUM": 0.49}, | |
| "PROCEDURES_PENALES": {"HIGH": 0.74, "MEDIUM": 0.62, "MINIMUM": 0.52} | |
| } | |
| # Optimized jurisprudence thresholds | |
| JURIS_HIGH_RELEVANCE = 0.68 | |
| JURIS_MEDIUM_RELEVANCE = 0.52 | |
| JURIS_MINIMUM_RELEVANCE = 0.35 | |
| TEMP_ANALYSIS = 0.05 | |
| TEMP_CLASSIFICATION = 0.1 | |
| TEMP_GENERATION = 0.05 | |
| MAX_HIGH_ARTICLES = 12 | |
| MAX_MEDIUM_ARTICLES = 8 | |
| MAX_LOW_ARTICLES = 4 | |
| MAX_JURIS_DOCS = 20 # REDUCED FROM 60 | |
| MAX_JURIS_RETRIEVAL = 80 # REDUCED FROM 250 | |
| MAX_CROSS_CODE_ARTICLES = 6 | |
| MAX_LEXICAL_RESULTS = 30 | |
| # Optimized secondary code mapping | |
| SECONDARY_CODE_MAPPING = { | |
| "CSP": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "DROITS_REELS"], | |
| "DROITS_REELS": ["PROCEDURE_CIVILE", "CSP", "OBLIGATIONS_CONTRATS", "CODE_COMMERCE"], | |
| "OBLIGATIONS_CONTRATS": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "CODE_TRAVAIL", "CSP"], | |
| "PROCEDURE_CIVILE": ["CSP", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL"], | |
| "CODE_TRAVAIL": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CSP"], | |
| "DROITS_PROCEDURES_FISCAUX": ["CODE_COMMERCE", "PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"], | |
| "DROIT_INTERNATIONAL_PRIVE": ["CSP", "PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS"], | |
| "CODE_PENAL": ["PROCEDURES_PENALES", "PROCEDURE_CIVILE", "CSP"], | |
| "CODE_COMMERCE": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL", "DROITS_REELS"], | |
| "PROCEDURES_PENALES": ["CODE_PENAL", "PROCEDURE_CIVILE", "CSP"] | |
| } | |
| CODE_NAMES = { | |
| "CSP": { | |
| "fr": "Code du Statut Personnel", | |
| "ar": "مجلة الأحوال الشخصية", | |
| "keywords": ["mariage", "divorce", "héritage", "succession", "paternité", "نكاح", "طلاق", "إرث", "ميراث", "نسب", "نفقة", "حضانة", "زواج", "فرقة"] | |
| }, | |
| "DROITS_REELS": { | |
| "fr": "Code des Droits Réels", | |
| "ar": "مجلة الحقوق العينية", | |
| "keywords": ["propriété", "immobilier", "hypothèque", "servitude", "usufruit", "ملكية", "عقار", "رهن", "أراضي", "عقارية", "حيازة", "تملك"] | |
| }, | |
| "OBLIGATIONS_CONTRATS": { | |
| "fr": "Code des Obligations et des Contrats", | |
| "ar": "مجلة الالتزامات والعقود", | |
| "keywords": ["contrat", "obligation", "responsabilité", "délit", "عقد", "التزام", "مسؤولية", "خطأ", "تعويض", "إبرام", "فسخ", "إبطال"] | |
| }, | |
| "PROCEDURE_CIVILE": { | |
| "fr": "Code de Procédure Civile et Commerciale", | |
| "ar": "مجلة المرافعات المدنية والتجارية", | |
| "keywords": ["procédure", "appel", "recours", "jugement", "تعقيب", "نقض", "محكمة", "حكم", "قرار", "طعن", "إجراءات", "استئناف", "طلبات جديدة", "الطلبات الجديدة"] | |
| }, | |
| "CODE_TRAVAIL": { | |
| "fr": "Code du Travail", | |
| "ar": "مجلة الشغل", | |
| "keywords": ["travail", "emploi", "licenciement", "salaire", "contrat", "شغل", "عقد شغل", "فصل", "أجير", "أجرة", "إجازة", "تعويض"] | |
| }, | |
| "DROITS_PROCEDURES_FISCAUX": { | |
| "fr": "Code des Droits et Procédures Fiscaux", | |
| "ar": "مجلة الحقوق والإجراءات الضريبية", | |
| "keywords": ["fiscal", "impôt", "taxe", "droit fiscal", "procédure fiscale", "ضريبة", "جباية", "ديوان", "غرامة", "تحصيل", "تهرب"] | |
| }, | |
| "DROIT_INTERNATIONAL_PRIVE": { | |
| "fr": "Code de Droit International Privé", | |
| "ar": "مجلة القانون الدولي الخاص", | |
| "keywords": ["international", "conflit de lois", "nationalité", "étranger", "تنازع القوانين", "اختصاص دولي", "جنسية", "أجانب", "إقليمية"] | |
| }, | |
| "CODE_PENAL": { | |
| "fr": "Code Pénal", | |
| "ar": "المجلة الجزائية", | |
| "keywords": ["pénal", "crime", "délit", "contravention", "peine", "جناية", "جنحة", "مخالفة", "عقوبة", "سجن", "حبس", "جرم"] | |
| }, | |
| "CODE_COMMERCE": { | |
| "fr": "Code de Commerce", | |
| "ar": "مجلة التجارية", | |
| "keywords": ["commerce", "commerçant", "entreprise", "société", "faillite", "تاجر", "تجار", "شركة", "سجل تجاري", "إفلاس", "تسوية"] | |
| }, | |
| "PROCEDURES_PENALES": { | |
| "fr": "Code des Procédures Pénales", | |
| "ar": "مجلة الإجراءات الجزائية", | |
| "keywords": ["procédure pénale", "enquête", "instruction", "تحقيق", "تحقيق جزائي", "قاضي التحقيق", "إحالة", "نيابة", "محاكمة"] | |
| } | |
| } | |
| # Initialisation des clients | |
| mongo_client = MongoClient(Config.MONGO_URI) | |
| embedding_client = OpenAI(base_url=f"{Config.AZURE_ENDPOINT}openai/v1/", api_key=Config.AZURE_API_KEY) | |
| chat_client = AzureOpenAI(api_key=Config.AZURE_API_KEY, azure_endpoint=Config.AZURE_ENDPOINT, api_version=Config.AZURE_API_VERSION) | |
| # ============================================================ | |
| # TRANSLATION SERVICE | |
| # ============================================================ | |
| class TranslationService: | |
| """Service de traduction professionnel""" | |
| _translation_cache = {} | |
| def translate_text(text: str, source_lang: str, target_lang: str) -> str: | |
| if source_lang == target_lang or not text: | |
| return text | |
| cache_key = f"{source_lang}_{target_lang}_{hashlib.md5(text.encode()).hexdigest()}" | |
| if cache_key in TranslationService._translation_cache: | |
| return TranslationService._translation_cache[cache_key] | |
| try: | |
| if target_lang == "ar": | |
| system_content = "أنت مترجم قانوني محترف متخصص في الترجمة من الفرنسية إلى العربية. حافظ على الدقة القانونية والمصطلحات الفنية." | |
| prompt = f"""ترجم النص القانوني التالي بدقة مع الحفاظ على: | |
| 1. المعنى القانوني الدقيق | |
| 2. المصطلحات القانونية المتخصصة | |
| 3. الأرقام والمراجع القانونية | |
| 4. السياق القانوني التونسي | |
| النص الفرنسي: {text} | |
| الترجمة العربية (تجنب الإضافة أو الحذف، كن دقيقًا):""" | |
| else: | |
| system_content = "Tu es un traducteur juridique professionnel spécialisé en droit tunisien. Préserve la précision juridique et la terminologie technique." | |
| prompt = f"""Traduis ce texte juridique avec précision en préservant: | |
| 1. Le sens juridique exact | |
| 2. La terminologie juridique spécialisée | |
| 3. Les chiffres et références légales | |
| 4. Le contexte juridique tunisien | |
| Texte arabe: {text} | |
| Traduction française (sans ajout ni omission, sois précis):""" | |
| response = chat_client.chat.completions.create( | |
| model=Config.CHAT_MODEL, | |
| messages=[ | |
| {"role": "system", "content": system_content}, | |
| {"role": "user", "content": prompt} | |
| ], | |
| temperature=0.1, | |
| max_tokens=2000 | |
| ) | |
| translation = response.choices[0].message.content.strip() | |
| TranslationService._translation_cache[cache_key] = translation | |
| logger.info(f"✅ Traduction {source_lang} → {target_lang} effectuée") | |
| return translation | |
| except Exception as e: | |
| logger.error(f"Erreur de traduction: {e}") | |
| return text | |
| def translate_ar_to_fr(text: str) -> str: | |
| return TranslationService.translate_text(text, "ar", "fr") | |
| def translate_fr_to_ar(text: str) -> str: | |
| return TranslationService.translate_text(text, "fr", "ar") | |
| def detect_and_translate(query: str) -> Dict[str, str]: | |
| arabic_chars = len(re.findall(r'[\u0600-\u06FF]', query)) | |
| total_chars = len(query.replace(' ', '')) | |
| if total_chars > 0 and (arabic_chars / total_chars) > 0.2: | |
| translated = TranslationService.translate_ar_to_fr(query) | |
| return { | |
| "original_query": query, | |
| "translated_query": translated, | |
| "original_language": "ar", | |
| "search_language": "fr" | |
| } | |
| else: | |
| return { | |
| "original_query": query, | |
| "translated_query": query, | |
| "original_language": "fr", | |
| "search_language": "fr" | |
| } | |
| # ============================================================ | |
| # ENUMS & DATA STRUCTURES | |
| # ============================================================ | |
| class SourceType(Enum): | |
| STATUTE = "statute" | |
| JURISPRUDENCE = "jurisprudence" | |
| class LegalCode(Enum): | |
| CSP = "CSP" | |
| DROITS_REELS = "DROITS_REELS" | |
| OBLIGATIONS_CONTRATS = "OBLIGATIONS_CONTRATS" | |
| PROCEDURE_CIVILE = "PROCEDURE_CIVILE" | |
| CODE_TRAVAIL = "CODE_TRAVAIL" | |
| DROITS_PROCEDURES_FISCAUX = "DROITS_PROCEDURES_FISCAUX" | |
| DROIT_INTERNATIONAL_PRIVE = "DROIT_INTERNATIONAL_PRIVE" | |
| CODE_PENAL = "CODE_PENAL" | |
| CODE_COMMERCE = "CODE_COMMERCE" | |
| PROCEDURES_PENALES = "PROCEDURES_PENALES" | |
| def from_string(cls, value: str): | |
| try: | |
| normalized = value.upper().replace("-", "_").replace(" ", "_") | |
| return cls(normalized) | |
| except: | |
| return cls.CSP | |
| class RetrievalStatus(Enum): | |
| SUCCESSFULLY_RETRIEVED = "retrieved" | |
| CITED_NOT_RETRIEVED = "cited_not_retrieved" | |
| class ArticleMetadata: | |
| article_number: str | |
| normalized_number: str | |
| article_text_fr: str | |
| article_text_ar: str | |
| code_type: LegalCode | |
| code_name_fr: str | |
| code_name_ar: str | |
| chapter: Optional[str] = None | |
| section: Optional[str] = None | |
| pdf_source: Optional[str] = None | |
| class RelevanceScore: | |
| similarity: float | |
| topic_overlap: float | |
| entity_match: float | |
| article_match: float | |
| code_relevance: float | |
| combined_score: float | |
| relevance_level: str | |
| confidence: float | |
| class RetrievedSource: | |
| source_id: str | |
| source_type: SourceType | |
| article_metadata: Optional[ArticleMetadata] | |
| content: str | |
| relevance: RelevanceScore | |
| retrieval_timestamp: datetime | |
| retrieval_method: str | |
| summary: str = "" | |
| full_text: str = "" | |
| primary_code: bool = True | |
| tags: List[str] = field(default_factory=list) | |
| code_fr: str = "" | |
| code_ar: str = "" | |
| juridiction: str = "" | |
| date_decision: str = "" | |
| class CitedSource: | |
| article_number: str | |
| code_type: LegalCode | |
| citation_context: str | |
| citation_position: int | |
| retrieval_status: RetrievalStatus | |
| class ValidationResult: | |
| is_valid: bool | |
| confidence_score: float | |
| hallucinated_citations: List[str] | |
| missing_retrievals: List[str] | |
| validation_errors: List[str] | |
| validation_warnings: List[str] | |
| class QueryAnalysis: | |
| original_query: str | |
| translated_query: str | |
| original_language: str | |
| search_language: str | |
| extracted_topics: List[str] | |
| legal_entities: List[str] | |
| cited_articles: List[str] | |
| primary_legal_code: LegalCode | |
| secondary_codes: List[LegalCode] | |
| question_type: str | |
| complexity_score: float | |
| search_queries: List[str] | |
| requires_statutory_law: bool | |
| code_confidence: float | |
| class LegalAnswer: | |
| answer_text: str | |
| retrieved_sources: List[RetrievedSource] | |
| cited_sources: List[CitedSource] | |
| validation_result: ValidationResult | |
| query_analysis: QueryAnalysis | |
| processing_metadata: Dict[str, Any] | |
| confidence_score: float | |
| # ============================================================ | |
| # ARTICLE NUMBER NORMALIZER PROFESSIONNEL | |
| # ============================================================ | |
| class ArticleNumberNormalizer: | |
| def normalize(article_ref: str) -> str: | |
| """Normalise une référence d'article pour obtenir uniquement le numéro""" | |
| if not article_ref or not isinstance(article_ref, str): | |
| return "" | |
| cleaned = re.sub(r'[^\d\s]', '', article_ref.strip()) | |
| cleaned = re.sub(r'\s+', ' ', cleaned) | |
| match = re.search(r'(\d+)(?:\s*(?:bis|ter|quater))?', cleaned, re.IGNORECASE) | |
| if match: | |
| return match.group(0).strip().lower() | |
| return cleaned | |
| def extract_all_numbers(text: str) -> Set[str]: | |
| """Extrait uniquement les numéros d'articles valides avec mots-clés spécifiques""" | |
| if not text: | |
| return set() | |
| patterns = [ | |
| r'\b(?:Article|article|art\.|Art\.)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', | |
| r'\b(?:الفصل|فصل|المادة|مادة|المادّة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', | |
| r'\b(?:n°|N°|numéro|رقم)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', | |
| r'[\(\[]\s*(?:article|Article|art\.|الفصل|المادة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\s*[\)\]]' | |
| ] | |
| numbers = set() | |
| for pattern in patterns: | |
| matches = re.findall(pattern, text, re.IGNORECASE | re.UNICODE) | |
| for match in matches: | |
| if isinstance(match, tuple): | |
| num = match[0] | |
| else: | |
| num = match | |
| if num.isdigit(): | |
| num_int = int(num) | |
| if (1900 <= num_int <= 2100) or num_int > 9999: | |
| continue | |
| normalized = ArticleNumberNormalizer.normalize(num) | |
| if normalized and re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE): | |
| numbers.add(normalized) | |
| return numbers | |
| def is_valid_article_number(art_num: str) -> bool: | |
| """Vérifie si un numéro d'article est valide""" | |
| if not art_num or not isinstance(art_num, str): | |
| return False | |
| normalized = ArticleNumberNormalizer.normalize(art_num) | |
| if not normalized: | |
| return False | |
| if not re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE): | |
| return False | |
| match = re.search(r'(\d+)', normalized) | |
| if not match: | |
| return False | |
| num = int(match.group(1)) | |
| if 1900 <= num <= 2100: | |
| return False | |
| if num > 9999: | |
| return False | |
| return True | |
| # ============================================================ | |
| # LEXICAL SEARCH ENGINE | |
| # ============================================================ | |
| class LexicalSearchEngine: | |
| """Moteur de recherche lexicale avancée pour la jurisprudence""" | |
| def create_lexical_queries(query_analysis: QueryAnalysis) -> List[Dict]: | |
| """Crée des requêtes lexicales pour la recherche dans la jurisprudence""" | |
| queries = [] | |
| keywords_fr = [] | |
| keywords_ar = [] | |
| for topic in query_analysis.extracted_topics: | |
| if topic and len(topic) > 2: | |
| keywords_fr.append(topic.lower()) | |
| if not any(char in topic for char in 'اأإآبتثجحخدذرزسشصضطظعغفقكلمنهوي'): | |
| try: | |
| translated = TranslationService.translate_fr_to_ar(topic) | |
| keywords_ar.append(translated) | |
| except: | |
| pass | |
| query_terms = re.findall(r'\b\w+\b', query_analysis.translated_query.lower()) | |
| keywords_fr.extend([term for term in query_terms if len(term) > 3]) | |
| if query_analysis.original_language == "ar": | |
| arabic_terms = re.findall(r'[\u0600-\u06FF]+', query_analysis.original_query) | |
| keywords_ar.extend([term for term in arabic_terms if len(term) > 2]) | |
| keywords_fr = list(set([k for k in keywords_fr if len(k) > 2]))[:30] | |
| keywords_ar = list(set([k for k in keywords_ar if len(k) > 2]))[:30] | |
| if keywords_fr: | |
| queries.append({ | |
| "language": "fr", | |
| "keywords": keywords_fr, | |
| "search_fields": ["resume_fr", "faits_fr", "decision_fr", "tags_fr", "text_to_vector_fr.principe"], | |
| "boost_fields": { | |
| "resume_fr": 2.0, | |
| "faits_fr": 1.5, | |
| "text_to_vector_fr.principe": 2.5, | |
| "tags_fr": 3.0 | |
| } | |
| }) | |
| if keywords_ar: | |
| queries.append({ | |
| "language": "ar", | |
| "keywords": keywords_ar, | |
| "search_fields": ["resume_ar", "faits_ar", "decision_ar", "tags_ar", "text_to_vector_ar.principe"], | |
| "boost_fields": { | |
| "resume_ar": 2.0, | |
| "faits_ar": 1.5, | |
| "text_to_vector_ar.principe": 2.5, | |
| "tags_ar": 3.0 | |
| } | |
| }) | |
| all_codes = [query_analysis.primary_legal_code] + query_analysis.secondary_codes | |
| for code in all_codes: | |
| code_keywords = Config.CODE_NAMES.get(code.value, {}).get("keywords", []) | |
| if code_keywords: | |
| fr_keywords = [kw for kw in code_keywords if kw.isascii()] | |
| ar_keywords = [kw for kw in code_keywords if not kw.isascii()] | |
| if fr_keywords: | |
| queries.append({ | |
| "language": "fr", | |
| "keywords": fr_keywords[:15], | |
| "search_fields": ["tags_fr", "resume_fr", "code_fr"], | |
| "boost_fields": {"tags_fr": 3.0, "code_fr": 2.0}, | |
| "code_filter": code.value | |
| }) | |
| if ar_keywords: | |
| queries.append({ | |
| "language": "ar", | |
| "keywords": ar_keywords[:15], | |
| "search_fields": ["tags_ar", "resume_ar", "code_ar"], | |
| "boost_fields": {"tags_ar": 3.0, "code_ar": 2.0}, | |
| "code_filter": code.value | |
| }) | |
| if query_analysis.cited_articles: | |
| for article in query_analysis.cited_articles: | |
| if ArticleNumberNormalizer.is_valid_article_number(article): | |
| queries.append({ | |
| "language": "both", | |
| "keywords": [f"article {article}", f"الفصل {article}"], | |
| "search_fields": ["articles_cites.article_num", "text_to_vector_ar.principe", "text_to_vector_fr.principe"], | |
| "boost_fields": {"articles_cites.article_num": 5.0} | |
| }) | |
| return queries | |
| def search_lexical(collection, queries: List[Dict], limit: int = 100) -> List[Dict]: | |
| """Exécute une recherche lexicale dans la collection""" | |
| all_results = [] | |
| for query_config in queries: | |
| try: | |
| mongo_query = LexicalSearchEngine._build_mongo_query(query_config) | |
| results = list(collection.find(mongo_query).limit(limit)) | |
| for doc in results: | |
| lexical_score = LexicalSearchEngine._calculate_lexical_score(doc, query_config) | |
| doc["_lexical_score"] = lexical_score | |
| doc["_lexical_query"] = query_config | |
| doc["_search_method"] = "lexical_search" | |
| found = False | |
| for existing in all_results: | |
| if existing.get("_id") == doc.get("_id"): | |
| found = True | |
| if lexical_score > existing.get("_lexical_score", 0): | |
| existing.update(doc) | |
| break | |
| if not found: | |
| all_results.append(doc) | |
| except Exception as e: | |
| logger.error(f"Erreur recherche lexicale: {e}") | |
| continue | |
| all_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True) | |
| return all_results[:limit] | |
| def _build_mongo_query(query_config: Dict) -> Dict: | |
| """Construit une requête MongoDB pour la recherche lexicale""" | |
| language = query_config.get("language", "fr") | |
| keywords = query_config.get("keywords", []) | |
| search_fields = query_config.get("search_fields", []) | |
| code_filter = query_config.get("code_filter") | |
| if not keywords or not search_fields: | |
| return {} | |
| or_conditions = [] | |
| for keyword in keywords: | |
| if not keyword or len(keyword) < 2: | |
| continue | |
| escaped_keyword = re.escape(keyword) | |
| for field in search_fields: | |
| field_parts = field.split('.') | |
| if len(field_parts) > 1: | |
| nested_field = field_parts[0] | |
| nested_subfield = field_parts[1] | |
| condition = { | |
| f"{nested_field}.{nested_subfield}": { | |
| "$regex": escaped_keyword, | |
| "$options": "i" | |
| } | |
| } | |
| else: | |
| condition = { | |
| field: { | |
| "$regex": escaped_keyword, | |
| "$options": "i" | |
| } | |
| } | |
| or_conditions.append(condition) | |
| if not or_conditions: | |
| return {} | |
| mongo_query = {"$or": or_conditions} | |
| if code_filter: | |
| code_names = Config.CODE_NAMES.get(code_filter, {}) | |
| if code_names: | |
| code_fr = code_names.get("fr", "") | |
| code_ar = code_names.get("ar", "") | |
| code_conditions = [] | |
| if code_fr: | |
| code_conditions.append({"code_fr": {"$regex": code_fr, "$options": "i"}}) | |
| if code_ar: | |
| code_conditions.append({"code_ar": {"$regex": code_ar, "$options": "i"}}) | |
| if code_conditions: | |
| mongo_query["$and"] = [{"$or": code_conditions}] | |
| return mongo_query | |
| def _calculate_lexical_score(doc: Dict, query_config: Dict) -> float: | |
| """Calcule un score lexical basé sur la pertinence""" | |
| keywords = query_config.get("keywords", []) | |
| search_fields = query_config.get("search_fields", []) | |
| boost_fields = query_config.get("boost_fields", {}) | |
| if not keywords: | |
| return 0.0 | |
| total_score = 0.0 | |
| keyword_count = 0 | |
| for keyword in keywords: | |
| keyword_lower = keyword.lower() | |
| keyword_score = 0.0 | |
| for field in search_fields: | |
| field_value = LexicalSearchEngine._get_field_value(doc, field) | |
| if not field_value: | |
| continue | |
| if keyword_lower in field_value.lower(): | |
| base_score = 1.0 | |
| boost = boost_fields.get(field, 1.0) | |
| if re.search(rf'\b{re.escape(keyword_lower)}\b', field_value.lower()): | |
| base_score *= 1.5 | |
| keyword_score = max(keyword_score, base_score * boost) | |
| if keyword_score > 0: | |
| total_score += keyword_score | |
| keyword_count += 1 | |
| if keyword_count == 0: | |
| return 0.0 | |
| average_score = total_score / keyword_count | |
| coverage_bonus = keyword_count / len(keywords) * 0.5 | |
| tags_fr = doc.get("tags_fr", []) | |
| tags_ar = doc.get("tags_ar", []) | |
| all_tags = tags_fr + tags_ar | |
| tag_bonus = 0.0 | |
| for tag in all_tags: | |
| for keyword in keywords: | |
| if keyword.lower() in tag.lower(): | |
| tag_bonus += 0.2 | |
| final_score = min(1.0, average_score + coverage_bonus + tag_bonus) | |
| return final_score | |
| def _get_field_value(doc: Dict, field_path: str) -> str: | |
| """Obtient la valeur d'un champ, gère les champs imbriqués""" | |
| if not field_path: | |
| return "" | |
| parts = field_path.split('.') | |
| current = doc | |
| for part in parts: | |
| if isinstance(current, dict): | |
| current = current.get(part, {}) | |
| else: | |
| return "" | |
| if isinstance(current, str): | |
| return current | |
| elif isinstance(current, list): | |
| return " ".join([str(item) for item in current]) | |
| elif current: | |
| return str(current) | |
| return "" | |
| # ============================================================ | |
| # ADVANCED SEARCH ENGINES (BM25, TF-IDF, EXACT MATCH) | |
| # ============================================================ | |
| class BM25SearchEngine: | |
| """Moteur de recherche BM25 avancé""" | |
| def __init__(self): | |
| self.k1 = 1.5 | |
| self.b = 0.75 | |
| self.avgdl = 0 | |
| self.doc_freqs = {} | |
| self.idf = {} | |
| self.doc_lengths = [] | |
| self.corpus_size = 0 | |
| self.corpus = [] | |
| self.doc_ids = [] | |
| self.fields_weights = { | |
| "resume_ar": 2.0, "resume_fr": 2.0, | |
| "faits_ar": 1.5, "faits_fr": 1.5, | |
| "decision_ar": 1.8, "decision_fr": 1.8, | |
| "tags_ar": 3.0, "tags_fr": 3.0, | |
| "text_to_vector_ar.principe": 2.5, | |
| "text_to_vector_fr.principe": 2.5, | |
| "code_ar": 1.2, "code_fr": 1.2 | |
| } | |
| def preprocess_text(self, text: str, language: str = "ar") -> List[str]: | |
| """Prétraitement avancé du texte""" | |
| if not text: | |
| return [] | |
| text = text.lower() | |
| text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s]', ' ', text) | |
| text = re.sub(r'\s+', ' ', text).strip() | |
| if language == "ar": | |
| tokens = re.findall(r'[\u0600-\u06FF]+', text) | |
| arabic_stopwords = { | |
| 'في', 'من', 'إلى', 'على', 'أن', 'إن', 'ما', 'هو', 'هي', 'كان', | |
| 'يكون', 'كانت', 'ليس', 'لا', 'ولكن', 'أو', 'و', 'لكن', 'إذا', | |
| 'ذلك', 'هذا', 'هذه', 'تلك', 'التي', 'الذي', 'الذين', 'قد', 'حيث', | |
| 'عن', 'مع', 'بين', 'فيما', 'كل', 'بعض', 'أي', 'كل', 'مادة', 'فصل', | |
| 'المادة', 'الفصل', 'قانون', 'القانون', 'مجلة', 'المجلة' | |
| } | |
| tokens = [token for token in tokens if token not in arabic_stopwords and len(token) > 2] | |
| else: | |
| tokens = re.findall(r'\b\w+\b', text) | |
| french_stopwords = { | |
| 'le', 'la', 'les', 'de', 'des', 'du', 'et', 'est', 'une', 'un', | |
| 'dans', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'article', | |
| 'articles', 'code', 'loi', 'droit', 'juridique', 'tribunal', | |
| 'cour', 'jugement', 'décision', 'affaire', 'procédure' | |
| } | |
| tokens = [token for token in tokens if token not in french_stopwords and len(token) > 2] | |
| return tokens | |
| def build_index(self, documents: List[Dict], language: str = "ar"): | |
| """Construit l'index BM25""" | |
| self.corpus = [] | |
| self.doc_ids = [] | |
| self.doc_lengths = [] | |
| logger.info(f"🔨 Construction index BM25 pour {len(documents)} documents...") | |
| for doc in documents: | |
| doc_id = str(doc.get("_id", "")) | |
| if not doc_id: | |
| continue | |
| full_text_tokens = [] | |
| for field, weight in self.fields_weights.items(): | |
| field_parts = field.split('.') | |
| field_value = doc | |
| for part in field_parts: | |
| if isinstance(field_value, dict): | |
| field_value = field_value.get(part, "") | |
| else: | |
| field_value = "" | |
| break | |
| if field_value: | |
| if isinstance(field_value, list): | |
| field_value = " ".join(field_value) | |
| tokens = self.preprocess_text(str(field_value), language) | |
| for _ in range(int(weight)): | |
| full_text_tokens.extend(tokens) | |
| if full_text_tokens: | |
| self.corpus.append(full_text_tokens) | |
| self.doc_ids.append(doc_id) | |
| self.doc_lengths.append(len(full_text_tokens)) | |
| if not self.corpus: | |
| logger.warning("⚠️ Aucun document indexé pour BM25") | |
| return | |
| self.corpus_size = len(self.corpus) | |
| self.avgdl = sum(self.doc_lengths) / self.corpus_size | |
| self.doc_freqs = {} | |
| for doc_tokens in self.corpus: | |
| unique_tokens = set(doc_tokens) | |
| for token in unique_tokens: | |
| self.doc_freqs[token] = self.doc_freqs.get(token, 0) + 1 | |
| self.idf = {} | |
| for token, freq in self.doc_freqs.items(): | |
| self.idf[token] = math.log((self.corpus_size - freq + 0.5) / (freq + 0.5) + 1) | |
| logger.info(f"✅ Index BM25 construit: {self.corpus_size} documents, {len(self.idf)} tokens uniques") | |
| def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]: | |
| """Recherche BM25 avec la requête""" | |
| if not self.corpus: | |
| return [] | |
| query_tokens = self.preprocess_text(query, language) | |
| if not query_tokens: | |
| return [] | |
| scores = np.zeros(self.corpus_size) | |
| for i, doc_tokens in enumerate(self.corpus): | |
| doc_len = self.doc_lengths[i] | |
| for token in query_tokens: | |
| if token in self.idf: | |
| f = doc_tokens.count(token) | |
| idf_score = self.idf[token] | |
| numerator = f * (self.k1 + 1) | |
| denominator = f + self.k1 * (1 - self.b + self.b * doc_len / self.avgdl) | |
| scores[i] += idf_score * numerator / denominator if denominator != 0 else 0 | |
| sorted_indices = np.argsort(scores)[::-1][:top_k] | |
| results = [] | |
| for idx in sorted_indices: | |
| if scores[idx] > 0: | |
| results.append((self.doc_ids[idx], float(scores[idx]))) | |
| logger.info(f"🔍 Recherche BM25: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})") | |
| return results | |
| class TFIDFSearchEngine: | |
| """Moteur de recherche TF-IDF""" | |
| def __init__(self): | |
| self.vocabulary = {} | |
| self.idf = {} | |
| self.tfidf_matrix = None | |
| self.doc_ids = [] | |
| def build_index(self, documents: List[Dict], language: str = "ar"): | |
| """Construit l'index TF-IDF""" | |
| self.doc_ids = [] | |
| all_docs_tokens = [] | |
| logger.info(f"🔨 Construction index TF-IDF pour {len(documents)} documents...") | |
| for doc in documents: | |
| doc_id = str(doc.get("_id", "")) | |
| if not doc_id: | |
| continue | |
| text_parts = [] | |
| priority_fields = [ | |
| "resume_ar", "resume_fr", | |
| "text_to_vector_ar.principe", "text_to_vector_fr.principe", | |
| "tags_ar", "tags_fr", "faits_ar", "faits_fr" | |
| ] | |
| for field in priority_fields: | |
| field_parts = field.split('.') | |
| field_value = doc | |
| for part in field_parts: | |
| if isinstance(field_value, dict): | |
| field_value = field_value.get(part, "") | |
| else: | |
| field_value = "" | |
| break | |
| if field_value: | |
| if isinstance(field_value, list): | |
| field_value = " ".join(field_value) | |
| text_parts.append(str(field_value)) | |
| full_text = " ".join(text_parts) | |
| if language == "ar": | |
| tokens = re.findall(r'[\u0600-\u06FF]{3,}', full_text.lower()) | |
| else: | |
| tokens = re.findall(r'\b\w{3,}\b', full_text.lower()) | |
| if tokens: | |
| all_docs_tokens.append(tokens) | |
| self.doc_ids.append(doc_id) | |
| if not all_docs_tokens: | |
| return | |
| all_tokens = set() | |
| for doc_tokens in all_docs_tokens: | |
| all_tokens.update(doc_tokens) | |
| self.vocabulary = {token: idx for idx, token in enumerate(sorted(all_tokens))} | |
| tf_matrix = np.zeros((len(all_docs_tokens), len(self.vocabulary))) | |
| for i, doc_tokens in enumerate(all_docs_tokens): | |
| token_counts = Counter(doc_tokens) | |
| total_tokens = len(doc_tokens) | |
| for token, count in token_counts.items(): | |
| if token in self.vocabulary: | |
| idx = self.vocabulary[token] | |
| tf_matrix[i, idx] = count / total_tokens if total_tokens > 0 else 0 | |
| doc_count = len(all_docs_tokens) | |
| df = np.sum(tf_matrix > 0, axis=0) | |
| self.idf = np.log((doc_count + 1) / (df + 1)) + 1 | |
| self.tfidf_matrix = tf_matrix * self.idf | |
| logger.info(f"✅ Index TF-IDF construit: {len(self.doc_ids)} documents, {len(self.vocabulary)} tokens") | |
| def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]: | |
| """Recherche TF-IDF""" | |
| if self.tfidf_matrix is None or not self.doc_ids: | |
| return [] | |
| if language == "ar": | |
| query_tokens = re.findall(r'[\u0600-\u06FF]{3,}', query.lower()) | |
| else: | |
| query_tokens = re.findall(r'\b\w{3,}\b', query.lower()) | |
| if not query_tokens: | |
| return [] | |
| query_vector = np.zeros(len(self.vocabulary)) | |
| query_counts = Counter(query_tokens) | |
| total_tokens = len(query_tokens) | |
| for token, count in query_counts.items(): | |
| if token in self.vocabulary: | |
| idx = self.vocabulary[token] | |
| query_vector[idx] = count / total_tokens if total_tokens > 0 else 0 | |
| query_vector = query_vector * self.idf | |
| norm_docs = np.linalg.norm(self.tfidf_matrix, axis=1, keepdims=True) | |
| norm_query = np.linalg.norm(query_vector) | |
| similarities = np.dot(self.tfidf_matrix, query_vector) / (norm_docs.flatten() * norm_query + 1e-8) | |
| sorted_indices = np.argsort(similarities)[::-1][:top_k] | |
| results = [] | |
| for idx in sorted_indices: | |
| if similarities[idx] > 0: | |
| results.append((self.doc_ids[idx], float(similarities[idx]))) | |
| logger.info(f"🔍 Recherche TF-IDF: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})") | |
| return results | |
| class ExactMatchSearchEngine: | |
| """Moteur de recherche par correspondance exacte""" | |
| def __init__(self): | |
| self.inverted_index = {} | |
| self.doc_metadata = {} | |
| def build_index(self, documents: List[Dict], language: str = "ar"): | |
| """Construit un index inversé pour la recherche exacte""" | |
| self.inverted_index = {} | |
| self.doc_metadata = {} | |
| logger.info(f"🔨 Construction index exact pour {len(documents)} documents...") | |
| legal_keywords = { | |
| "ar": [ | |
| "كراء تجاري", "تجديد الكراء", "مؤسسات التعليم الخاص", | |
| "تعليم خاص", "الأكرية التجارية", "تأجير", "عقد كراء", | |
| "مدة الكراء", "حق التجديد", "المحلات التجارية", | |
| "المادة 532", "الفصل 532", "مجلة الالتزامات والعقود", | |
| "محكمة التعقيب", "المحكمة التجارية", "عقود التسويغ" | |
| ], | |
| "fr": [ | |
| "bail commercial", "renouvellement bail", "institutions enseignement privé", | |
| "enseignement privé", "baux commerciaux", "location", "contrat bail", | |
| "durée bail", "droit renouvellement", "locaux commerciaux", | |
| "article 532", "code obligations contrats", "cour cassation", | |
| "tribunal commerce", "contrats location" | |
| ] | |
| } | |
| keywords = legal_keywords.get(language, []) | |
| for doc in documents: | |
| doc_id = str(doc.get("_id", "")) | |
| if not doc_id: | |
| continue | |
| self.doc_metadata[doc_id] = doc | |
| text_fields = [] | |
| if language == "ar": | |
| fields_to_check = ["resume_ar", "faits_ar", "decision_ar", "text_to_vector_ar.principe"] | |
| else: | |
| fields_to_check = ["resume_fr", "faits_fr", "decision_fr", "text_to_vector_fr.principe"] | |
| for field in fields_to_check: | |
| field_parts = field.split('.') | |
| field_value = doc | |
| for part in field_parts: | |
| if isinstance(field_value, dict): | |
| field_value = field_value.get(part, "") | |
| else: | |
| field_value = "" | |
| break | |
| if field_value: | |
| if isinstance(field_value, list): | |
| field_value = " ".join(field_value) | |
| text_fields.append(str(field_value).lower()) | |
| full_text = " ".join(text_fields) | |
| for keyword in keywords: | |
| if keyword.lower() in full_text: | |
| if keyword not in self.inverted_index: | |
| self.inverted_index[keyword] = [] | |
| occurrences = full_text.count(keyword.lower()) | |
| score = occurrences * 2.0 | |
| if field_parts[0] in ["resume", "text_to_vector"]: | |
| score += 1.5 | |
| self.inverted_index[keyword].append((doc_id, score)) | |
| logger.info(f"✅ Index exact construit: {len(self.inverted_index)} mots-clés indexés") | |
| def search(self, query: str, language: str = "ar", top_k: int = 50) -> List[Tuple[str, float]]: | |
| """Recherche par correspondance exacte""" | |
| if not self.inverted_index: | |
| return [] | |
| query_lower = query.lower() | |
| found_keywords = [] | |
| for keyword in self.inverted_index.keys(): | |
| if keyword.lower() in query_lower: | |
| found_keywords.append(keyword) | |
| if not found_keywords: | |
| return [] | |
| doc_scores = {} | |
| for keyword in found_keywords: | |
| for doc_id, score in self.inverted_index.get(keyword, []): | |
| if doc_id not in doc_scores: | |
| doc_scores[doc_id] = 0 | |
| doc_scores[doc_id] += score | |
| sorted_docs = sorted(doc_scores.items(), key=lambda x: x[1], reverse=True)[:top_k] | |
| results = [(doc_id, score) for doc_id, score in sorted_docs if score > 0] | |
| logger.info(f"🔍 Recherche exacte: {len(results)} résultats (mots-clés trouvés: {found_keywords})") | |
| return results | |
| class HybridJurisprudenceSearch: | |
| """Recherche hybride de jurisprudence avec multiples méthodes""" | |
| def __init__(self, db_manager): | |
| self.db_manager = db_manager | |
| self.bm25_searcher = BM25SearchEngine() | |
| self.tfidf_searcher = TFIDFSearchEngine() | |
| self.exact_searcher = ExactMatchSearchEngine() | |
| self.cached_docs = {} | |
| def load_jurisprudence_docs(self, language: str = "ar", limit: int = 2000) -> List[Dict]: | |
| """Charge les documents de jurisprudence depuis MongoDB""" | |
| cache_key = f"juris_docs_{language}" | |
| if cache_key in self.cached_docs: | |
| cached_time, docs = self.cached_docs[cache_key] | |
| if datetime.now().timestamp() - cached_time < 300: | |
| logger.info(f"📦 Utilisation du cache pour {len(docs)} documents") | |
| return docs | |
| logger.info(f"📥 Chargement des documents de jurisprudence ({language})...") | |
| query = {} | |
| if language == "ar": | |
| query = {"resume_ar": {"$exists": True, "$ne": ""}} | |
| else: | |
| query = {"resume_fr": {"$exists": True, "$ne": ""}} | |
| docs = list(self.db_manager.juris_collection.find(query).limit(limit)) | |
| self.cached_docs[cache_key] = (datetime.now().timestamp(), docs) | |
| logger.info(f"✅ {len(docs)} documents chargés") | |
| return docs | |
| def search_hybrid(self, query: str, query_analysis: QueryAnalysis, | |
| limit: int = 150) -> List[Dict]: | |
| """Recherche hybride avec multiples méthodes""" | |
| language = query_analysis.search_language | |
| docs = self.load_jurisprudence_docs(language, limit=2000) | |
| if not docs: | |
| logger.warning("⚠️ Aucun document de jurisprudence disponible") | |
| return [] | |
| logger.info("🏗️ Construction des index de recherche...") | |
| self.bm25_searcher.build_index(docs, language) | |
| self.tfidf_searcher.build_index(docs, language) | |
| self.exact_searcher.build_index(docs, language) | |
| logger.info("🔍 Exécution des recherches hybrides...") | |
| bm25_results = self.bm25_searcher.search(query, language, top_k=limit) | |
| tfidf_results = self.tfidf_searcher.search(query, language, top_k=limit) | |
| exact_results = self.exact_searcher.search(query, language, top_k=limit) | |
| article_results = self._search_by_articles(query_analysis.cited_articles, docs) | |
| all_results = self._merge_results( | |
| bm25_results, tfidf_results, exact_results, article_results, | |
| docs, limit | |
| ) | |
| logger.info(f"✅ Recherche hybride: {len(all_results)} résultats combinés") | |
| return all_results | |
| def _search_by_articles(self, articles: List[str], docs: List[Dict]) -> List[Tuple[str, float]]: | |
| """Recherche basée sur les articles cités""" | |
| if not articles: | |
| return [] | |
| results = [] | |
| article_set = set(articles) | |
| for doc in docs: | |
| doc_id = str(doc.get("_id", "")) | |
| cited_articles = doc.get("articles_cites", []) | |
| score = 0 | |
| for cited in cited_articles: | |
| article_num = cited.get("article_num", "") | |
| if article_num in article_set: | |
| score += 3.0 | |
| if score > 0: | |
| results.append((doc_id, score)) | |
| return results | |
| def _merge_results(self, bm25_results: List, tfidf_results: List, | |
| exact_results: List, article_results: List, | |
| docs: List[Dict], limit: int) -> List[Dict]: | |
| """Fusionne les résultats de toutes les méthodes""" | |
| doc_map = {str(doc.get("_id", "")): doc for doc in docs} | |
| combined_scores = {} | |
| method_weights = { | |
| "bm25": 0.3, | |
| "tfidf": 0.25, | |
| "exact": 0.3, | |
| "articles": 0.15 | |
| } | |
| for doc_id, score in bm25_results: | |
| if doc_id not in combined_scores: | |
| combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} | |
| combined_scores[doc_id]["bm25"] = score | |
| for doc_id, score in tfidf_results: | |
| if doc_id not in combined_scores: | |
| combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} | |
| combined_scores[doc_id]["tfidf"] = score | |
| for doc_id, score in exact_results: | |
| if doc_id not in combined_scores: | |
| combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} | |
| combined_scores[doc_id]["exact"] = score | |
| for doc_id, score in article_results: | |
| if doc_id not in combined_scores: | |
| combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} | |
| combined_scores[doc_id]["articles"] = score | |
| final_results = [] | |
| for doc_id, scores in combined_scores.items(): | |
| if doc_id in doc_map: | |
| final_score = ( | |
| scores["bm25"] * method_weights["bm25"] + | |
| scores["tfidf"] * method_weights["tfidf"] + | |
| scores["exact"] * method_weights["exact"] + | |
| scores["articles"] * method_weights["articles"] | |
| ) | |
| if final_score > 0: | |
| doc = doc_map[doc_id].copy() | |
| doc["_hybrid_score"] = final_score | |
| doc["_score_details"] = scores | |
| final_results.append(doc) | |
| final_results.sort(key=lambda x: x.get("_hybrid_score", 0), reverse=True) | |
| return final_results[:limit] | |
| # ============================================================ | |
| # LEGAL CODE CLASSIFIER PROFESSIONNEL | |
| # ============================================================ | |
| class LegalCodeClassifier: | |
| def classify_query(query: str, language: str) -> Dict[str, Any]: | |
| """Classification avancée des codes juridiques tunisiens""" | |
| classification_prompt = f"""Analyze this Tunisian legal query to identify relevant codes from ALL 10 Tunisian codes. | |
| TUNISIAN LEGAL CODES AVAILABLE: | |
| 1. CSP (Code du Statut Personnel) - Family law, marriage, divorce, inheritance, personal status | |
| 2. DROITS_REELS (Code des Droits Réels) - Property law, real estate, ownership, mortgages, real rights | |
| 3. OBLIGATIONS_CONTRATS (Code des Obligations et des Contrats) - Contract law, obligations, torts, civil liability | |
| 4. PROCEDURE_CIVILE (Code de Procédure Civile et Commerciale) - Civil procedure, appeals, judicial process | |
| 5. CODE_TRAVAIL (Code du Travail) - Labor law, employment contracts, termination, labor disputes | |
| 6. DROITS_PROCEDURES_FISCAUX (Code des Droits et Procédures Fiscaux) - Tax law, fiscal procedures, tax disputes | |
| 7. DROIT_INTERNATIONAL_PRIVE (Code de Droit International Privé) - Private international law, conflicts of law | |
| 8. CODE_PENAL (Code Pénal) - Criminal law, offenses, penalties, criminal procedure | |
| 9. CODE_COMMERCE (Code de Commerce) - Commercial law, companies, bankruptcy, commercial contracts | |
| 10. PROCEDURES_PENALES (Code des Procédures Pénales) - Criminal procedure, investigation, prosecution | |
| QUERY: {query} | |
| LANGUAGE: {language} | |
| ANALYSIS INSTRUCTIONS: | |
| 1. Identify the PRIMARY code (most relevant) | |
| 2. Identify SECONDARY codes (relevant but less direct) | |
| 3. Provide confidence score (0.0-1.0) | |
| 4. Extract key legal concepts | |
| 5. DO NOT extract article numbers here | |
| OUTPUT ONLY VALID JSON: | |
| {{ | |
| "primary_code": "CODE_TRAVAIL", | |
| "secondary_codes": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"], | |
| "confidence": 0.92, | |
| "key_concepts": ["labor contract", "termination", "CDD renewal"], | |
| "reasoning": "Query focuses on employment contract renewal and termination issues", | |
| "priority_factors": ["employment", "contract", "termination", "labor rights"] | |
| }}""" | |
| try: | |
| response = chat_client.chat.completions.create( | |
| model=Config.CHAT_MODEL, | |
| messages=[ | |
| {"role": "system", "content": "Expert Tunisian legal code classifier. Analyze queries and identify relevant codes accurately. Output ONLY valid JSON."}, | |
| {"role": "user", "content": classification_prompt} | |
| ], | |
| temperature=Config.TEMP_CLASSIFICATION, | |
| max_tokens=800 | |
| ) | |
| result = response.choices[0].message.content.strip() | |
| result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE) | |
| classification = json.loads(result) | |
| primary_code = LegalCodeClassifier._validate_and_normalize_code(classification["primary_code"]) | |
| secondary_codes = [LegalCodeClassifier._validate_and_normalize_code(code) for code in classification.get("secondary_codes", [])] | |
| logger.info(f"✅ Classification réussie: {primary_code.value} (confiance: {classification['confidence']:.2f})") | |
| return { | |
| "primary_code": primary_code, | |
| "secondary_codes": secondary_codes, | |
| "confidence": float(classification.get("confidence", 0.5)), | |
| "key_concepts": classification.get("key_concepts", []), | |
| "reasoning": classification.get("reasoning", ""), | |
| "priority_factors": classification.get("priority_factors", []) | |
| } | |
| except Exception as e: | |
| logger.error(f"Erreur classification GPT: {e}") | |
| return LegalCodeClassifier.fallback_classification(query, language) | |
| def _validate_and_normalize_code(code_str: str) -> LegalCode: | |
| """Valide et normalise un code""" | |
| try: | |
| normalized = code_str.upper().strip() | |
| variations = { | |
| "TRAVAIL": "CODE_TRAVAIL", | |
| "CODE TRAVAIL": "CODE_TRAVAIL", | |
| "CODETRAVAIL": "CODE_TRAVAIL", | |
| "DROIT_REEL": "DROITS_REELS", | |
| "DROIT REEL": "DROITS_REELS", | |
| "DROITS_REEL": "DROITS_REELS", | |
| "PENAL": "CODE_PENAL", | |
| "CODEPENAL": "CODE_PENAL", | |
| "COMMERCE": "CODE_COMMERCE", | |
| "CODECOMMERCE": "CODE_COMMERCE", | |
| "PROCEDURE_PENALE": "PROCEDURES_PENALES", | |
| "PROCEDURE PENALE": "PROCEDURES_PENALES" | |
| } | |
| if normalized in variations: | |
| normalized = variations[normalized] | |
| return LegalCode.from_string(normalized) | |
| except: | |
| logger.warning(f"Code non reconnu: {code_str}, utilisation CSP par défaut") | |
| return LegalCode.CSP | |
| def fallback_classification(query: str, language: str) -> Dict[str, Any]: | |
| """Classification par mots-clés comme fallback""" | |
| query_lower = query.lower() | |
| scores = {code: 0 for code in LegalCode} | |
| for code in LegalCode: | |
| if code.value in Config.CODE_NAMES: | |
| keywords = Config.CODE_NAMES[code.value].get("keywords", []) | |
| for keyword in keywords: | |
| if keyword.lower() in query_lower: | |
| scores[code] += 2 | |
| special_rules = [ | |
| (["contrat", "travail", "licenciement"], LegalCode.CODE_TRAVAIL, 3), | |
| (["contrat", "travail", "salaire"], LegalCode.CODE_TRAVAIL, 2), | |
| (["propriété", "succession", "héritage"], LegalCode.CSP, 2), | |
| (["propriété", "hypothèque", "immobilier"], LegalCode.DROITS_REELS, 2), | |
| (["contrat", "obligation", "responsabilité"], LegalCode.OBLIGATIONS_CONTRATS, 2), | |
| (["procédure", "appel", "jugement"], LegalCode.PROCEDURE_CIVILE, 2), | |
| (["fiscal", "impôt", "taxe"], LegalCode.DROITS_PROCEDURES_FISCAUX, 2), | |
| (["international", "étranger", "nationalité"], LegalCode.DROIT_INTERNATIONAL_PRIVE, 2), | |
| (["crime", "peine", "prison"], LegalCode.CODE_PENAL, 2), | |
| (["commerce", "société", "faillite"], LegalCode.CODE_COMMERCE, 2), | |
| (["enquête", "instruction", "procédure pénale"], LegalCode.PROCEDURES_PENALES, 2) | |
| ] | |
| for keywords, code, bonus in special_rules: | |
| if all(keyword in query_lower for keyword in keywords): | |
| scores[code] += bonus | |
| sorted_codes = sorted(scores.items(), key=lambda x: x[1], reverse=True) | |
| primary_code = sorted_codes[0][0] if sorted_codes else LegalCode.CSP | |
| secondary_codes = [] | |
| for code, score in sorted_codes[1:]: | |
| if score > 0 and len(secondary_codes) < 3: | |
| secondary_codes.append(code) | |
| total_score = sum(scores.values()) | |
| confidence = min(scores[primary_code] / max(total_score, 1) * 1.2, 0.85) | |
| logger.info(f"⚠️ Classification fallback: {primary_code.value} (confiance: {confidence:.2f})") | |
| return { | |
| "primary_code": primary_code, | |
| "secondary_codes": secondary_codes, | |
| "confidence": confidence, | |
| "key_concepts": [], | |
| "reasoning": "Fallback keyword-based classification", | |
| "priority_factors": [] | |
| } | |
| # ============================================================ | |
| # EMBEDDING SERVICE PROFESSIONNEL | |
| # ============================================================ | |
| class EnterpriseEmbeddingService: | |
| _cache = {} | |
| _failed_embeddings = set() | |
| def get_embedding(text: str) -> Optional[List[float]]: | |
| cache_key = hashlib.md5(text.encode()).hexdigest() | |
| if cache_key in EnterpriseEmbeddingService._cache: | |
| return EnterpriseEmbeddingService._cache[cache_key] | |
| if cache_key in EnterpriseEmbeddingService._failed_embeddings: | |
| return None | |
| try: | |
| cleaned_text = EnterpriseEmbeddingService.preprocess_text(text) | |
| if not cleaned_text or len(cleaned_text) < 3: | |
| return None | |
| response = embedding_client.embeddings.create( | |
| input=cleaned_text, | |
| model=Config.EMBEDDING_MODEL | |
| ) | |
| embedding = response.data[0].embedding | |
| EnterpriseEmbeddingService._cache[cache_key] = embedding | |
| return embedding | |
| except Exception as e: | |
| logger.error(f"Erreur génération embedding: {e}") | |
| EnterpriseEmbeddingService._failed_embeddings.add(cache_key) | |
| return None | |
| def preprocess_text(text: str) -> str: | |
| """Prétraitement avancé du texte""" | |
| if not text: | |
| return "" | |
| text = re.sub(r'\s+', ' ', text) | |
| text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s.,;:!?()\[\]-]', ' ', text) | |
| text = text.strip() | |
| if len(text) > 8000: | |
| text = text[:8000] | |
| return text | |
| def calculate_similarity(emb1: Optional[List[float]], emb2: Optional[List[float]]) -> float: | |
| if not emb1 or not emb2: | |
| return 0.0 | |
| try: | |
| arr1 = np.array(emb1).reshape(1, -1) | |
| arr2 = np.array(emb2).reshape(1, -1) | |
| similarity = cosine_similarity(arr1, arr2)[0][0] | |
| return max(0.0, min(1.0, similarity)) | |
| except Exception as e: | |
| logger.error(f"Erreur calcul similarité: {e}") | |
| return 0.0 | |
| # ============================================================ | |
| # QUERY ANALYZER PROFESSIONNEL | |
| # ============================================================ | |
| class EnterpriseQueryAnalyzer: | |
| def analyze_query(query: str) -> QueryAnalysis: | |
| """Analyse complète de la requête""" | |
| translation_result = TranslationService.detect_and_translate(query) | |
| original_query = translation_result["original_query"] | |
| translated_query = translation_result["translated_query"] | |
| original_language = translation_result["original_language"] | |
| search_language = translation_result["search_language"] | |
| logger.info(f"🌐 Langue détectée: {original_language}") | |
| if original_language == "ar": | |
| logger.info(f"📝 Requête traduite pour recherche") | |
| classification = LegalCodeClassifier.classify_query(translated_query, search_language) | |
| cited_articles_original = ArticleNumberNormalizer.extract_all_numbers(original_query) | |
| cited_articles_translated = ArticleNumberNormalizer.extract_all_numbers(translated_query) | |
| all_articles = cited_articles_original.union(cited_articles_translated) | |
| analysis_prompt = f"""Analyze this Tunisian legal query in detail. | |
| PRIMARY CODE IDENTIFIED: {classification['primary_code'].value} | |
| QUERY: {translated_query} | |
| LANGUAGE: {search_language} | |
| CITED ARTICLES DETECTED: {list(all_articles)} | |
| ANALYSIS TASKS: | |
| 1. Validate and filter article numbers (keep only valid Tunisian law article references) | |
| 2. Extract key legal topics | |
| 3. Identify relevant legal entities | |
| 4. Determine question type and complexity | |
| 5. Generate search queries for semantic search | |
| 6. Identify cross-code implications | |
| RULES FOR ARTICLE EXTRACTION: | |
| - Keep only valid article numbers (e.g., "123", "45 bis") | |
| - Remove years, dates, page numbers, etc. | |
| - If query mentions articles by topic without numbers, don't list them | |
| OUTPUT ONLY VALID JSON: | |
| {{ | |
| "validated_articles": [], | |
| "topics": ["topic1", "topic2"], | |
| "legal_entities": ["entity1", "entity2"], | |
| "question_type": "substantive/procedural/hybrid", | |
| "complexity_score": 0.85, | |
| "requires_statutory_law": true, | |
| "requires_procedural_law": false, | |
| "requires_commercial_law": false, | |
| "requires_criminal_law": false, | |
| "requires_labor_law": true, | |
| "requires_tax_law": false, | |
| "requires_international_law": false, | |
| "requires_family_law": false, | |
| "requires_property_law": false, | |
| "search_queries": ["query1", "query2"], | |
| "cross_code_references": [] | |
| }}""" | |
| try: | |
| response = chat_client.chat.completions.create( | |
| model=Config.CHAT_MODEL, | |
| messages=[ | |
| {"role": "system", "content": "Advanced Tunisian legal query analyzer. Focus on accuracy and precision. Output ONLY valid JSON."}, | |
| {"role": "user", "content": analysis_prompt} | |
| ], | |
| temperature=Config.TEMP_ANALYSIS, | |
| max_tokens=800 | |
| ) | |
| result = response.choices[0].message.content.strip() | |
| result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE) | |
| analysis_data = json.loads(result) | |
| validated_articles = [] | |
| for art in analysis_data.get('validated_articles', []): | |
| if ArticleNumberNormalizer.is_valid_article_number(str(art)): | |
| validated_articles.append(str(art)) | |
| else: | |
| logger.warning(f"Article ignoré (invalide): {art}") | |
| final_articles = list(set(validated_articles + list(all_articles))) | |
| primary_code = classification['primary_code'] | |
| secondary_codes = classification['secondary_codes'].copy() | |
| code_mappings = { | |
| 'requires_procedural_law': LegalCode.PROCEDURE_CIVILE, | |
| 'requires_commercial_law': LegalCode.CODE_COMMERCE, | |
| 'requires_criminal_law': LegalCode.CODE_PENAL, | |
| 'requires_labor_law': LegalCode.CODE_TRAVAIL, | |
| 'requires_tax_law': LegalCode.DROITS_PROCEDURES_FISCAUX, | |
| 'requires_international_law': LegalCode.DROIT_INTERNATIONAL_PRIVE, | |
| 'requires_family_law': LegalCode.CSP, | |
| 'requires_property_law': LegalCode.DROITS_REELS | |
| } | |
| for key, code in code_mappings.items(): | |
| if analysis_data.get(key, False) and code not in secondary_codes and code != primary_code: | |
| secondary_codes.append(code) | |
| if not secondary_codes and primary_code.value in Config.SECONDARY_CODE_MAPPING: | |
| default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value] | |
| secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]] | |
| secondary_codes = list(dict.fromkeys(secondary_codes))[:5] | |
| logger.info(f"📊 Classification: {primary_code.value}") | |
| logger.info(f"📋 Codes secondaires: {[c.value for c in secondary_codes]}") | |
| logger.info(f"📄 Articles cités: {final_articles}") | |
| return QueryAnalysis( | |
| original_query=original_query, | |
| translated_query=translated_query, | |
| original_language=original_language, | |
| search_language=search_language, | |
| extracted_topics=analysis_data.get('topics', []), | |
| legal_entities=analysis_data.get('legal_entities', []), | |
| cited_articles=final_articles, | |
| primary_legal_code=primary_code, | |
| secondary_codes=secondary_codes, | |
| question_type=analysis_data.get('question_type', 'substantive'), | |
| complexity_score=float(analysis_data.get('complexity_score', 0.5)), | |
| search_queries=analysis_data.get('search_queries', [translated_query]), | |
| requires_statutory_law=analysis_data.get('requires_statutory_law', True), | |
| code_confidence=classification['confidence'] | |
| ) | |
| except Exception as e: | |
| logger.error(f"Erreur analyse requête: {e}") | |
| return EnterpriseQueryAnalyzer._create_fallback_analysis( | |
| original_query, translated_query, original_language, | |
| search_language, classification, all_articles | |
| ) | |
| def _create_fallback_analysis(original_query, translated_query, original_language, | |
| search_language, classification, all_articles): | |
| """Créer une analyse de fallback""" | |
| primary_code = classification['primary_code'] | |
| if primary_code.value in Config.SECONDARY_CODE_MAPPING: | |
| default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value] | |
| secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]] | |
| else: | |
| secondary_codes = [] | |
| if primary_code != LegalCode.PROCEDURE_CIVILE and LegalCode.PROCEDURE_CIVILE not in secondary_codes: | |
| secondary_codes.append(LegalCode.PROCEDURE_CIVILE) | |
| return QueryAnalysis( | |
| original_query=original_query, | |
| translated_query=translated_query, | |
| original_language=original_language, | |
| search_language=search_language, | |
| extracted_topics=[], | |
| legal_entities=[], | |
| cited_articles=list(all_articles), | |
| primary_legal_code=primary_code, | |
| secondary_codes=secondary_codes, | |
| question_type="substantive", | |
| complexity_score=0.5, | |
| search_queries=[translated_query], | |
| requires_statutory_law=True, | |
| code_confidence=classification['confidence'] | |
| ) | |
| # ============================================================ | |
| # DATABASE MANAGER PROFESSIONNEL | |
| # ============================================================ | |
| class EnterpriseDatabaseManager: | |
| def __init__(self): | |
| self.code_db = mongo_client[Config.DB_CODE] | |
| self.juris_collection = mongo_client[Config.DB_JURIS][Config.COL_JURIS] | |
| self._article_cache = {} | |
| self._collection_cache = {} | |
| self._verify_collections() | |
| def _verify_collections(self): | |
| """Vérifier que toutes les collections existent""" | |
| logger.info("🔍 Vérification des collections...") | |
| available_collections = self.code_db.list_collection_names() | |
| for code_key, collection_name in Config.COLLECTIONS.items(): | |
| if collection_name in available_collections: | |
| logger.info(f" ✓ {code_key}: {collection_name}") | |
| else: | |
| logger.warning(f" ✗ {code_key}: {collection_name} - COLLECTION NON TROUVÉE") | |
| def get_collection(self, code_type: LegalCode): | |
| """Récupère la collection MongoDB pour un type de code""" | |
| if code_type in self._collection_cache: | |
| return self._collection_cache[code_type] | |
| collection_name = Config.COLLECTIONS.get(code_type.value) | |
| if not collection_name: | |
| logger.error(f"❌ Collection non configurée pour: {code_type.value}") | |
| return None | |
| try: | |
| collection = self.code_db[collection_name] | |
| self._collection_cache[code_type] = collection | |
| return collection | |
| except Exception as e: | |
| logger.error(f"❌ Erreur accès collection {collection_name}: {e}") | |
| return None | |
| def get_article_by_number(self, article_number: str, code_type: LegalCode) -> Optional[Dict]: | |
| """Récupère un article par son numéro""" | |
| if not ArticleNumberNormalizer.is_valid_article_number(article_number): | |
| logger.warning(f"Numéro d'article invalide: {article_number}") | |
| return None | |
| cache_key = f"{code_type.value}_{article_number}" | |
| if cache_key in self._article_cache: | |
| return self._article_cache[cache_key] | |
| collection = self.get_collection(code_type) | |
| if collection is None: | |
| return None | |
| normalized = ArticleNumberNormalizer.normalize(article_number) | |
| search_strategies = [ | |
| lambda: collection.find_one({"article_num": normalized}), | |
| lambda: collection.find_one({"article_number": normalized}), | |
| lambda: collection.find_one({"art": normalized}), | |
| lambda: collection.find_one({"$or": [ | |
| {"article_text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}}, | |
| {"text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}}, | |
| {"contenu": {"$regex": f"\\b{normalized}\\b", "$options": "i"}} | |
| ]}) | |
| ] | |
| article = None | |
| for strategy in search_strategies: | |
| article = strategy() | |
| if article: | |
| break | |
| if article: | |
| self._article_cache[cache_key] = article | |
| logger.info(f"✅ Article {normalized} trouvé dans {code_type.value}") | |
| else: | |
| logger.warning(f"❌ Article {normalized} NON TROUVÉ dans {code_type.value}") | |
| return article | |
| def get_multiple_articles(self, article_numbers: List[str], primary_code: LegalCode, | |
| secondary_codes: List[LegalCode] = None, user_language: str = "fr") -> List[RetrievedSource]: | |
| """Récupère plusieurs articles - MÉTHODE MANQUANTE AJOUTÉE""" | |
| retrieved = [] | |
| valid_articles = [art for art in article_numbers if ArticleNumberNormalizer.is_valid_article_number(art)] | |
| if not valid_articles: | |
| logger.info("Aucun numéro d'article valide") | |
| return retrieved | |
| logger.info(f"Recherche de {len(valid_articles)} articles...") | |
| for art_num in valid_articles: | |
| article = self.get_article_by_number(art_num, primary_code) | |
| if article: | |
| retrieved.append(self._create_source_from_article(article, art_num, primary_code, True, user_language)) | |
| if secondary_codes: | |
| for art_num in valid_articles: | |
| already_found = any( | |
| source.article_metadata.normalized_number == ArticleNumberNormalizer.normalize(art_num) | |
| for source in retrieved | |
| ) | |
| if not already_found: | |
| for code in secondary_codes: | |
| article = self.get_article_by_number(art_num, code) | |
| if article: | |
| retrieved.append(self._create_source_from_article(article, art_num, code, False, user_language)) | |
| break | |
| logger.info(f"✅ {len(retrieved)} articles récupérés") | |
| return retrieved | |
| def _create_source_from_article(self, article: Dict, art_num: str, code_type: LegalCode, | |
| primary: bool, user_language: str) -> RetrievedSource: | |
| """Crée un objet RetrievedSource à partir d'un article""" | |
| article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or '' | |
| article_text_ar = "" | |
| if user_language == "ar" and article_text_fr: | |
| article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr) | |
| metadata = ArticleMetadata( | |
| article_number=article.get('article_num', art_num) or article.get('article_number', art_num), | |
| normalized_number=ArticleNumberNormalizer.normalize(art_num), | |
| article_text_fr=article_text_fr, | |
| article_text_ar=article_text_ar, | |
| code_type=code_type, | |
| code_name_fr=Config.CODE_NAMES[code_type.value]["fr"], | |
| code_name_ar=Config.CODE_NAMES[code_type.value]["ar"], | |
| chapter=article.get('chapter'), | |
| section=article.get('section'), | |
| pdf_source=article.get('pdf_source') | |
| ) | |
| if user_language == "ar" and metadata.article_text_ar: | |
| content = metadata.article_text_ar | |
| code_name = metadata.code_name_ar | |
| else: | |
| content = metadata.article_text_fr | |
| code_name = metadata.code_name_fr | |
| full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}" | |
| return RetrievedSource( | |
| source_id=str(article.get('_id', '')), | |
| source_type=SourceType.STATUTE, | |
| article_metadata=metadata, | |
| content=full_content, | |
| relevance=RelevanceScore( | |
| similarity=1.0, topic_overlap=1.0, entity_match=1.0, article_match=1.0, | |
| code_relevance=1.0 if primary else 0.7, | |
| combined_score=1.0 if primary else 0.7, | |
| relevance_level="high", confidence=1.0 | |
| ), | |
| retrieval_timestamp=datetime.now(), | |
| retrieval_method='direct_lookup', | |
| summary=content[:500] + "..." if len(content) > 500 else content, | |
| full_text=content, | |
| primary_code=primary | |
| ) | |
| def search_similar_articles(self, query_embedding: List[float], code_type: LegalCode, | |
| limit: int = 20, user_language: str = "fr") -> List[Dict]: | |
| """Recherche sémantique d'articles similaires""" | |
| try: | |
| collection = self.get_collection(code_type) | |
| if collection is None: | |
| return [] | |
| articles = list(collection.find({"embedding": {"$exists": True}}).limit(200)) | |
| if not articles: | |
| return [] | |
| for article in articles: | |
| embedding = article.get("embedding") | |
| if embedding: | |
| similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding) | |
| article["_similarity"] = similarity | |
| else: | |
| article["_similarity"] = 0.0 | |
| articles.sort(key=lambda x: x.get("_similarity", 0), reverse=True) | |
| return articles[:limit] | |
| except Exception as e: | |
| logger.error(f"Erreur recherche sémantique {code_type.value}: {e}") | |
| return [] | |
| def search_jurisprudence_enhanced(self, query_embedding: List[float], language: str, | |
| topics: List[str] = None, primary_code: str = None, | |
| limit: int = 100) -> List[Dict]: | |
| """Recherche améliorée de jurisprudence avec filtres avancés""" | |
| try: | |
| embedding_field = "embedding_ar" if language == "ar" else "embedding_fr" | |
| sample = self.juris_collection.find_one({embedding_field: {"$exists": True}}) | |
| if not sample: | |
| embedding_field = "embedding" | |
| logger.warning(f"Champ {embedding_field} non trouvé, utilisation du champ générique 'embedding'") | |
| base_query = {embedding_field: {"$exists": True}} | |
| if primary_code: | |
| code_name_ar = Config.CODE_NAMES.get(primary_code, {}).get("ar", "") | |
| code_name_fr = Config.CODE_NAMES.get(primary_code, {}).get("fr", "") | |
| base_query["$or"] = [ | |
| {"code_ar": {"$regex": code_name_ar, "$options": "i"}}, | |
| {"code_fr": {"$regex": code_name_fr, "$options": "i"}}, | |
| {"tags_ar": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", []) if tag.isascii() is False]}}, | |
| {"tags_fr": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", [])]}} | |
| ] | |
| docs = list(self.juris_collection.find(base_query).limit(limit * 2)) | |
| if not docs: | |
| logger.warning("Aucun document de jurisprudence trouvé") | |
| return [] | |
| for doc in docs: | |
| embedding = doc.get(embedding_field) | |
| if embedding: | |
| similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding) | |
| bonus = 0.0 | |
| if topics: | |
| tags_ar = doc.get("tags_ar", []) | |
| tags_fr = doc.get("tags_fr", []) | |
| all_tags = tags_ar + tags_fr | |
| for topic in topics: | |
| topic_lower = topic.lower() | |
| for tag in all_tags: | |
| if topic_lower in tag.lower() or tag.lower() in topic_lower: | |
| bonus += 0.05 | |
| doc["_similarity"] = min(1.0, similarity + bonus) | |
| doc["_search_method"] = "enhanced_semantic_search" | |
| else: | |
| doc["_similarity"] = 0.0 | |
| docs.sort(key=lambda x: x.get("_similarity", 0), reverse=True) | |
| filtered_docs = [doc for doc in docs if doc.get("_similarity", 0) >= Config.JURIS_MINIMUM_RELEVANCE] | |
| logger.info(f"Jurisprudence: {len(filtered_docs)} documents après filtrage (sur {len(docs)})") | |
| return filtered_docs[:limit] | |
| except Exception as e: | |
| logger.error(f"Erreur recherche jurisprudence améliorée: {e}") | |
| return [] | |
| def search_jurisprudence_by_keywords(self, keywords: List[str], language: str, | |
| primary_code: str = None, limit: int = 40) -> List[Dict]: | |
| """Recherche de jurisprudence par mots-clés""" | |
| try: | |
| query = {} | |
| if language == "ar": | |
| text_fields = ["resume_ar", "faits_ar", "text_to_vector_ar.principe", "decision_ar"] | |
| tag_field = "tags_ar" | |
| else: | |
| text_fields = ["resume_fr", "faits_fr", "text_to_vector_fr.principe", "decision_fr"] | |
| tag_field = "tags_fr" | |
| keyword_queries = [] | |
| for keyword in keywords: | |
| if keyword: | |
| for field in text_fields: | |
| keyword_queries.append({field: {"$regex": keyword, "$options": "i"}}) | |
| if keyword_queries: | |
| query["$or"] = keyword_queries | |
| if primary_code: | |
| code_keywords = Config.CODE_NAMES.get(primary_code, {}).get("keywords", []) | |
| if code_keywords: | |
| code_query = {"tags_fr": {"$in": code_keywords}} | |
| if language == "ar": | |
| arabic_keywords = [kw for kw in code_keywords if not kw.isascii()] | |
| if arabic_keywords: | |
| code_query["$or"] = [{"tags_ar": {"$in": arabic_keywords}}] | |
| if "$or" in query: | |
| query["$and"] = [{"$or": query.pop("$or")}, code_query] | |
| else: | |
| query.update(code_query) | |
| docs = list(self.juris_collection.find(query).limit(limit)) | |
| for doc in docs: | |
| score = 0.0 | |
| for keyword in keywords: | |
| for field in text_fields: | |
| field_value = doc | |
| for part in field.split('.'): | |
| field_value = field_value.get(part, {}) if isinstance(field_value, dict) else "" | |
| if isinstance(field_value, str) and keyword.lower() in field_value.lower(): | |
| score += 0.1 | |
| tags = doc.get(tag_field, []) | |
| for tag in tags: | |
| for keyword in keywords: | |
| if keyword.lower() in tag.lower(): | |
| score += 0.15 | |
| doc["_keyword_score"] = min(1.0, score) | |
| doc["_search_method"] = "keyword_search" | |
| docs.sort(key=lambda x: x.get("_keyword_score", 0), reverse=True) | |
| logger.info(f"Jurisprudence par mots-clés: {len(docs)} documents trouvés") | |
| return docs[:limit] | |
| except Exception as e: | |
| logger.error(f"Erreur recherche par mots-clés: {e}") | |
| return [] | |
| def search_jurisprudence_lexical(self, query_analysis: QueryAnalysis, limit: int = 100) -> List[Dict]: | |
| """Recherche lexicale avancée dans la jurisprudence""" | |
| try: | |
| lexical_queries = LexicalSearchEngine.create_lexical_queries(query_analysis) | |
| if not lexical_queries: | |
| logger.warning("Aucune requête lexicale générée") | |
| return [] | |
| lexical_results = LexicalSearchEngine.search_lexical( | |
| self.juris_collection, | |
| lexical_queries, | |
| limit=limit | |
| ) | |
| logger.info(f"🔍 Recherche lexicale: {len(lexical_results)} documents trouvés") | |
| filtered_results = [] | |
| for doc in lexical_results: | |
| lexical_score = doc.get("_lexical_score", 0) | |
| if lexical_score >= 0.25: | |
| code_relevant = False | |
| all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] | |
| code_fr = doc.get("code_fr", "") | |
| code_ar = doc.get("code_ar", "") | |
| for code in all_codes: | |
| code_info = Config.CODE_NAMES.get(code, {}) | |
| if code_info: | |
| if (code_info.get("fr") and code_info["fr"] in code_fr) or \ | |
| (code_info.get("ar") and code_info["ar"] in code_ar): | |
| code_relevant = True | |
| break | |
| if code_relevant: | |
| lexical_score = min(1.0, lexical_score + 0.2) | |
| doc["_lexical_score"] = lexical_score | |
| filtered_results.append(doc) | |
| filtered_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True) | |
| logger.info(f"✅ Recherche lexicale filtrée: {len(filtered_results)} documents pertinents") | |
| return filtered_results[:limit] | |
| except Exception as e: | |
| logger.error(f"Erreur recherche lexicale: {e}") | |
| return [] | |
| def search_jurisprudence_by_all_codes(self, query_analysis: QueryAnalysis, query_embedding: List[float], | |
| limit: int = 150) -> List[Dict]: | |
| """Recherche de jurisprudence dans tous les codes pertinents""" | |
| all_results = [] | |
| all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] | |
| for code in all_codes: | |
| try: | |
| code_results = self.search_jurisprudence_enhanced( | |
| query_embedding, | |
| query_analysis.search_language, | |
| topics=query_analysis.extracted_topics, | |
| primary_code=code, | |
| limit=limit // len(all_codes) | |
| ) | |
| for doc in code_results: | |
| doc_code_fr = doc.get("code_fr", "") | |
| doc_code_ar = doc.get("code_ar", "") | |
| code_info = Config.CODE_NAMES.get(code, {}) | |
| if code_info: | |
| code_fr = code_info.get("fr", "") | |
| code_ar = code_info.get("ar", "") | |
| if (code_fr and code_fr in doc_code_fr) or (code_ar and code_ar in doc_code_ar): | |
| doc["_similarity"] = min(1.0, doc.get("_similarity", 0) + 0.15) | |
| doc["_code_filter"] = code | |
| all_results.extend(code_results) | |
| except Exception as e: | |
| logger.error(f"Erreur recherche jurisprudence pour code {code}: {e}") | |
| continue | |
| seen_ids = set() | |
| unique_results = [] | |
| for doc in all_results: | |
| doc_id = doc.get("_id") | |
| if doc_id and doc_id not in seen_ids: | |
| seen_ids.add(doc_id) | |
| unique_results.append(doc) | |
| unique_results.sort(key=lambda x: x.get("_similarity", 0), reverse=True) | |
| logger.info(f"🌐 Recherche multi-code: {len(unique_results)} documents uniques") | |
| return unique_results[:limit] | |
| # ============================================================ | |
| # ENHANCED SEARCH ENGINE WITH HYBRID SEARCH | |
| # ============================================================ | |
| class EnhancedEnterpriseSearchEngine: | |
| def __init__(self): | |
| self.db_manager = EnterpriseDatabaseManager() | |
| self.hybrid_searcher = HybridJurisprudenceSearch(self.db_manager) | |
| def search(self, query_analysis: QueryAnalysis) -> Dict[str, Any]: | |
| """Exécute une recherche complète avec méthode hybride""" | |
| logger.info(f"🔍 Lancement recherche hybride multi-méthodes") | |
| # 1. Récupération directe des articles cités | |
| direct_sources = self.db_manager.get_multiple_articles( | |
| query_analysis.cited_articles, | |
| query_analysis.primary_legal_code, | |
| query_analysis.secondary_codes, | |
| query_analysis.original_language | |
| ) | |
| # 2. Recherche sémantique dans les codes | |
| query_embedding = EnterpriseEmbeddingService.get_embedding(query_analysis.translated_query) | |
| primary_semantic = [] | |
| secondary_semantic = [] | |
| if query_embedding: | |
| # Recherche sémantique dans le code principal | |
| primary_articles = self.db_manager.search_similar_articles( | |
| query_embedding, query_analysis.primary_legal_code, | |
| limit=30, user_language=query_analysis.original_language | |
| ) | |
| primary_semantic = self._process_semantic_results( | |
| primary_articles, query_analysis.primary_legal_code, | |
| True, query_analysis.original_language | |
| ) | |
| # Recherche sémantique dans les codes secondaires | |
| for code in query_analysis.secondary_codes: | |
| articles = self.db_manager.search_similar_articles( | |
| query_embedding, code, | |
| limit=Config.MAX_CROSS_CODE_ARTICLES, | |
| user_language=query_analysis.original_language | |
| ) | |
| secondary_results = self._process_semantic_results( | |
| articles, code, False, query_analysis.original_language | |
| ) | |
| secondary_semantic.extend(secondary_results) | |
| # 3. Recherche hybride de jurisprudence | |
| logger.info(" 🔍 Recherche jurisprudence hybride...") | |
| hybrid_juris_docs = self.hybrid_searcher.search_hybrid( | |
| query_analysis.translated_query, | |
| query_analysis, | |
| limit=Config.MAX_JURIS_RETRIEVAL | |
| ) | |
| # 4. Recherches traditionnelles (pour complément) | |
| juris_docs_semantic = [] | |
| juris_docs_lexical = [] | |
| juris_docs_keywords = [] | |
| if query_embedding: | |
| juris_docs_semantic = self.db_manager.search_jurisprudence_by_all_codes( | |
| query_analysis, | |
| query_embedding, | |
| limit=Config.MAX_JURIS_RETRIEVAL // 3 | |
| ) | |
| juris_docs_lexical = self.db_manager.search_jurisprudence_lexical( | |
| query_analysis, | |
| limit=Config.MAX_JURIS_RETRIEVAL // 3 | |
| ) | |
| keywords = self._extract_juris_keywords(query_analysis) | |
| juris_docs_keywords = self.db_manager.search_jurisprudence_by_keywords( | |
| keywords, | |
| query_analysis.search_language, | |
| primary_code=query_analysis.primary_legal_code.value, | |
| limit=Config.MAX_JURIS_RETRIEVAL // 3 | |
| ) | |
| # 5. Fusionner TOUS les résultats de jurisprudence | |
| all_juris_docs = self._merge_all_jurisprudence_results( | |
| hybrid_juris_docs, | |
| juris_docs_semantic, | |
| juris_docs_lexical, | |
| juris_docs_keywords | |
| ) | |
| juris_sources = self._process_jurisprudence_results(all_juris_docs, query_analysis.original_language) | |
| # 6. Fusionner et organiser | |
| all_statute_sources = self._merge_sources(direct_sources, primary_semantic, secondary_semantic) | |
| statute_results = self._organize_by_relevance(all_statute_sources, query_analysis.primary_legal_code) | |
| missing_articles = self._identify_missing_articles(query_analysis.cited_articles, all_statute_sources) | |
| retrieval_metadata = { | |
| "primary_code": query_analysis.primary_legal_code.value, | |
| "secondary_codes": [c.value for c in query_analysis.secondary_codes], | |
| "direct_retrievals": len(direct_sources), | |
| "primary_semantic": len(primary_semantic), | |
| "secondary_semantic": len(secondary_semantic), | |
| "total_statutes": len(all_statute_sources), | |
| "total_jurisprudence": len(juris_sources), | |
| "missing_articles": missing_articles, | |
| "search_methods": ["direct", "semantic", "hybrid", "lexical", "keywords", "bm25", "tfidf", "exact"] | |
| } | |
| return { | |
| "statute_sources": statute_results, | |
| "jurisprudence_sources": juris_sources[:Config.MAX_JURIS_DOCS], | |
| "query_analysis": query_analysis, | |
| "retrieval_metadata": retrieval_metadata | |
| } | |
| def _extract_juris_keywords(self, query_analysis: QueryAnalysis) -> List[str]: | |
| """Extrait les mots-clés pour la recherche de jurisprudence""" | |
| keywords = set() | |
| keywords.update(query_analysis.extracted_topics) | |
| all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] | |
| for code in all_codes: | |
| if code in Config.CODE_NAMES: | |
| code_keywords = Config.CODE_NAMES[code].get("keywords", []) | |
| keywords.update(code_keywords[:15]) | |
| keywords.update([ | |
| "طلبات جديدة", "الطلبات الجديدة", "demandes nouvelles", "new claims", | |
| "استئناف", "appel", "appeal", "تعقيب", "cassation", | |
| "طلاق إنشاء", "divorce création", "طلاق الضرر", "divorce préjudice", | |
| "جراية عمرية", "pension viagère", "lifetime pension", | |
| "تعويض", "indemnité", "compensation", "ضرر", "préjudice", | |
| "محكمة البداية", "first instance court", "محكمة الاستئناف", "appeal court", | |
| "الفصل 147", "147", "article 147", "مرفوض شكلاً", "non recevable", | |
| "إجراءات", "procédure", "شكلية", "formalité", "تقاضي", "litigation", | |
| "مبدأ التقاضي", "principe de procès", "درجة التقاضي", "degré de juridiction" | |
| ]) | |
| cleaned_keywords = [] | |
| for kw in keywords: | |
| if kw and len(kw) > 2: | |
| cleaned_keywords.append(kw.strip().lower()) | |
| return list(set(cleaned_keywords))[:40] | |
| def _merge_all_jurisprudence_results(self, hybrid_results: List[Dict], | |
| semantic_results: List[Dict], | |
| lexical_results: List[Dict], | |
| keyword_results: List[Dict]) -> List[Dict]: | |
| """Fusionne tous les types de résultats de jurisprudence""" | |
| seen_ids = set() | |
| merged = [] | |
| def add_document(doc, score_field, method): | |
| doc_id = doc.get("_id") | |
| if doc_id and doc_id not in seen_ids: | |
| seen_ids.add(doc_id) | |
| if score_field in doc: | |
| doc["_similarity"] = doc[score_field] | |
| doc["_search_method"] = method | |
| merged.append(doc) | |
| for doc in hybrid_results: | |
| add_document(doc, "_hybrid_score", "hybrid_search") | |
| for doc in semantic_results: | |
| if doc.get("_id") not in seen_ids: | |
| add_document(doc, "_similarity", "semantic_search") | |
| for doc in lexical_results: | |
| if doc.get("_id") not in seen_ids: | |
| add_document(doc, "_lexical_score", "lexical_search") | |
| for doc in keyword_results: | |
| if doc.get("_id") not in seen_ids: | |
| add_document(doc, "_keyword_score", "keyword_search") | |
| merged.sort(key=lambda x: self._calculate_combined_juris_score(x), reverse=True) | |
| logger.info(f"Fusion complète jurisprudence: {len(merged)} documents uniques") | |
| return merged | |
| def _calculate_combined_juris_score(self, doc: Dict) -> float: | |
| """Calcule un score combiné pour la jurisprudence""" | |
| hybrid_score = doc.get("_hybrid_score", 0.0) | |
| if hybrid_score > 0: | |
| return hybrid_score * 1.2 | |
| semantic_score = doc.get("_similarity", 0.0) | |
| lexical_score = doc.get("_lexical_score", 0.0) | |
| keyword_score = doc.get("_keyword_score", 0.0) | |
| method = doc.get("_search_method", "") | |
| if method == "semantic_search": | |
| return semantic_score | |
| elif method == "lexical_search": | |
| return lexical_score * 0.7 + semantic_score * 0.3 | |
| elif method == "keyword_search": | |
| return keyword_score * 0.6 + semantic_score * 0.4 | |
| else: | |
| return max(semantic_score, lexical_score, keyword_score) | |
| def _process_semantic_results(self, articles: List[Dict], code_type: LegalCode, | |
| primary: bool, user_language: str) -> List[RetrievedSource]: | |
| """Traite les résultats de recherche sémantique""" | |
| processed = [] | |
| thresholds = Config.RELEVANCE_THRESHOLDS.get(code_type.value, {"HIGH": 0.7, "MEDIUM": 0.6, "MINIMUM": 0.5}) | |
| for article in articles: | |
| similarity = article.get('_similarity', 0.0) | |
| if similarity >= thresholds["MINIMUM"]: | |
| art_num = article.get('article_num') or article.get('article_number') or article.get('art') or 'N/A' | |
| article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or '' | |
| article_text_ar = "" | |
| if user_language == "ar" and article_text_fr: | |
| article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr) | |
| metadata = ArticleMetadata( | |
| article_number=art_num, | |
| normalized_number=ArticleNumberNormalizer.normalize(art_num), | |
| article_text_fr=article_text_fr, | |
| article_text_ar=article_text_ar, | |
| code_type=code_type, | |
| code_name_fr=Config.CODE_NAMES[code_type.value]["fr"], | |
| code_name_ar=Config.CODE_NAMES[code_type.value]["ar"], | |
| chapter=article.get('chapter'), | |
| section=article.get('section'), | |
| pdf_source=article.get('pdf_source') | |
| ) | |
| if similarity >= thresholds["HIGH"]: | |
| level = "high" | |
| elif similarity >= thresholds["MEDIUM"]: | |
| level = "medium" | |
| else: | |
| level = "low" | |
| if user_language == "ar" and metadata.article_text_ar: | |
| content = metadata.article_text_ar | |
| code_name = metadata.code_name_ar | |
| else: | |
| content = metadata.article_text_fr | |
| code_name = metadata.code_name_fr | |
| full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}" | |
| code_relevance = 1.0 if primary else 0.7 | |
| combined_score = similarity * code_relevance | |
| processed.append(RetrievedSource( | |
| source_id=str(article.get('_id', '')), | |
| source_type=SourceType.STATUTE, | |
| article_metadata=metadata, | |
| content=full_content, | |
| relevance=RelevanceScore( | |
| similarity=similarity, | |
| topic_overlap=0.0, | |
| entity_match=0.0, | |
| article_match=0.0, | |
| code_relevance=code_relevance, | |
| combined_score=combined_score, | |
| relevance_level=level, | |
| confidence=similarity | |
| ), | |
| retrieval_timestamp=datetime.now(), | |
| retrieval_method='semantic_search', | |
| summary=content[:500] + "..." if len(content) > 500 else content, | |
| full_text=content, | |
| primary_code=primary | |
| )) | |
| return processed | |
| def _merge_sources(self, direct: List[RetrievedSource], primary_semantic: List[RetrievedSource], | |
| secondary_semantic: List[RetrievedSource]) -> List[RetrievedSource]: | |
| """Fusionne les sources en évitant les doublons""" | |
| seen_articles = set() | |
| merged = [] | |
| for source in direct + primary_semantic + secondary_semantic: | |
| if source.article_metadata: | |
| article_id = f"{source.article_metadata.code_type.value}_{source.article_metadata.normalized_number}" | |
| if article_id not in seen_articles: | |
| seen_articles.add(article_id) | |
| merged.append(source) | |
| return merged | |
| def _organize_by_relevance(self, sources: List[RetrievedSource], primary_code: LegalCode) -> Dict[str, List[RetrievedSource]]: | |
| """Organise les sources par niveau de pertinence""" | |
| organized = {"high": [], "medium": [], "low": []} | |
| for source in sources: | |
| level = source.relevance.relevance_level | |
| if level in organized: | |
| organized[level].append(source) | |
| for level in organized: | |
| organized[level].sort(key=lambda x: (x.primary_code, x.relevance.combined_score), reverse=True) | |
| organized["high"] = organized["high"][:Config.MAX_HIGH_ARTICLES] | |
| organized["medium"] = organized["medium"][:Config.MAX_MEDIUM_ARTICLES] | |
| organized["low"] = organized["low"][:Config.MAX_LOW_ARTICLES] | |
| return organized | |
| def _identify_missing_articles(self, cited: List[str], retrieved: List[RetrievedSource]) -> List[str]: | |
| """Identifie les articles cités mais non récupérés""" | |
| retrieved_numbers = { | |
| source.article_metadata.normalized_number | |
| for source in retrieved if source.article_metadata | |
| } | |
| missing = [] | |
| for article in cited: | |
| normalized = ArticleNumberNormalizer.normalize(article) | |
| if normalized not in retrieved_numbers and ArticleNumberNormalizer.is_valid_article_number(article): | |
| missing.append(article) | |
| return missing | |
| def _process_jurisprudence_results(self, juris_docs: List[Dict], user_language: str) -> List[RetrievedSource]: | |
| """Traite les résultats de jurisprudence de manière améliorée""" | |
| processed = [] | |
| for doc in juris_docs: | |
| similarity = self._calculate_combined_juris_score(doc) | |
| if similarity >= Config.JURIS_HIGH_RELEVANCE: | |
| relevance_level = "high" | |
| elif similarity >= Config.JURIS_MEDIUM_RELEVANCE: | |
| relevance_level = "medium" | |
| elif similarity >= Config.JURIS_MINIMUM_RELEVANCE: | |
| relevance_level = "low" | |
| else: | |
| continue | |
| if user_language == "ar": | |
| resume = doc.get('resume_ar') or doc.get('summary_ar') or doc.get('description_ar') or doc.get('contenu_ar') or '' | |
| faits = doc.get('faits_ar') or '' | |
| decision = doc.get('decision_ar') or '' | |
| code_name = doc.get('code_ar', '') | |
| juridiction = doc.get('juridiction', '') | |
| tags = doc.get('tags_ar', []) | |
| if not resume: | |
| resume_fr = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or '' | |
| if resume_fr: | |
| resume = TranslationService.translate_fr_to_ar(resume_fr) | |
| if not faits: | |
| faits_fr = doc.get('faits_fr') or '' | |
| if faits_fr: | |
| faits = TranslationService.translate_fr_to_ar(faits_fr) | |
| if not decision: | |
| decision_fr = doc.get('decision_fr') or '' | |
| if decision_fr: | |
| decision = TranslationService.translate_fr_to_ar(decision_fr) | |
| else: | |
| resume = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or '' | |
| faits = doc.get('faits_fr') or '' | |
| decision = doc.get('decision_fr') or '' | |
| code_name = doc.get('code_fr', '') | |
| juridiction = doc.get('juridiction', '') | |
| tags = doc.get('tags_fr', []) | |
| case_number = doc.get('numero_dossier') or doc.get('case_number') or '' | |
| date_decision = doc.get('date') or doc.get('date_jugement') or doc.get('date_decision') or '' | |
| full_content = "" | |
| if user_language == "ar": | |
| full_content += f"رقم القضية: {case_number}\n" if case_number else "" | |
| full_content += f"المحكمة: {juridiction}\n" if juridiction else "" | |
| full_content += f"التاريخ: {date_decision}\n" if date_decision else "" | |
| full_content += f"المجلة: {code_name}\n" if code_name else "" | |
| full_content += f"طريقة البحث: {doc.get('_search_method', '')}\n" | |
| full_content += f"درجة الصلة: {similarity:.3f}\n" | |
| full_content += f"\nالملخص:\n{resume}\n" if resume else "" | |
| full_content += f"\nالوقائع:\n{faits}\n" if faits else "" | |
| full_content += f"\nالقرار:\n{decision}\n" if decision else "" | |
| else: | |
| full_content += f"N° Affaire: {case_number}\n" if case_number else "" | |
| full_content += f"Juridiction: {juridiction}\n" if juridiction else "" | |
| full_content += f"Date: {date_decision}\n" if date_decision else "" | |
| full_content += f"Code: {code_name}\n" if code_name else "" | |
| full_content += f"Méthode de recherche: {doc.get('_search_method', '')}\n" | |
| full_content += f"Score de pertinence: {similarity:.3f}\n" | |
| full_content += f"\nRésumé:\n{resume}\n" if resume else "" | |
| full_content += f"\nFaits:\n{faits}\n" if faits else "" | |
| full_content += f"\nDécision:\n{decision}\n" if decision else "" | |
| summary = resume[:200] + "..." if len(resume) > 200 else resume | |
| processed.append(RetrievedSource( | |
| source_id=str(doc.get('_id', '')), | |
| source_type=SourceType.JURISPRUDENCE, | |
| article_metadata=None, | |
| content=full_content, | |
| relevance=RelevanceScore( | |
| similarity=similarity, | |
| topic_overlap=0.0, | |
| entity_match=0.0, | |
| article_match=0.0, | |
| code_relevance=0.9, | |
| combined_score=similarity, | |
| relevance_level=relevance_level, | |
| confidence=similarity | |
| ), | |
| retrieval_timestamp=datetime.now(), | |
| retrieval_method=doc.get('_search_method', 'jurisprudence_search'), | |
| summary=summary, | |
| full_text=full_content, | |
| primary_code=False, | |
| tags=tags[:10], | |
| code_fr=doc.get('code_fr', ''), | |
| code_ar=doc.get('code_ar', ''), | |
| juridiction=juridiction, | |
| date_decision=date_decision | |
| )) | |
| processed.sort(key=lambda x: x.relevance.combined_score, reverse=True) | |
| return processed | |
| # ============================================================ | |
| # CONTEXT BUILDER PROFESSIONNEL | |
| # ============================================================ | |
| class EnterpriseContextBuilder: | |
| def build_context(search_results: Dict[str, Any]) -> Tuple[str, Dict[str, Any]]: | |
| """Construit le contexte pour la génération avec améliorations""" | |
| statute_results = search_results["statute_sources"] | |
| juris_sources = search_results["jurisprudence_sources"] | |
| query_analysis = search_results["query_analysis"] | |
| retrieval_meta = search_results["retrieval_metadata"] | |
| context_parts = [] | |
| source_registry = { | |
| "statutes": {}, | |
| "jurisprudence": {}, | |
| "primary_code": query_analysis.primary_legal_code.value, | |
| "cited_articles": query_analysis.cited_articles, | |
| "missing_articles": retrieval_meta["missing_articles"], | |
| "search_methods": retrieval_meta.get("search_methods", []) | |
| } | |
| statutes_by_code = {} | |
| for level in ["high", "medium", "low"]: | |
| for source in statute_results.get(level, []): | |
| if source.article_metadata: | |
| code = source.article_metadata.code_type.value | |
| if code not in statutes_by_code: | |
| statutes_by_code[code] = [] | |
| statutes_by_code[code].append(source) | |
| statute_count = 0 | |
| primary_code = query_analysis.primary_legal_code.value | |
| if primary_code in statutes_by_code: | |
| context_parts.append(f"\n=== {Config.CODE_NAMES[primary_code]['fr'].upper()} ===") | |
| context_parts.append(f"=== {Config.CODE_NAMES[primary_code]['ar']} ===") | |
| for source in statutes_by_code[primary_code]: | |
| statute_count += 1 | |
| art_num = source.article_metadata.article_number | |
| context_parts.append(f"\n[STATUTE_{statute_count}]") | |
| context_parts.append(f"Article: {art_num}") | |
| context_parts.append(f"Code: {primary_code}") | |
| context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}") | |
| context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}") | |
| context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n") | |
| source_registry["statutes"][art_num] = { | |
| "text": source.content, | |
| "full_text": source.full_text if source.full_text else source.content, | |
| "code": primary_code, | |
| "code_name_fr": Config.CODE_NAMES[primary_code]["fr"], | |
| "code_name_ar": Config.CODE_NAMES[primary_code]["ar"], | |
| "relevance": source.relevance.combined_score, | |
| "primary_code": True | |
| } | |
| for code, sources in statutes_by_code.items(): | |
| if code != primary_code: | |
| context_parts.append(f"\n=== {Config.CODE_NAMES[code]['fr'].upper()} (Contextual) ===") | |
| context_parts.append(f"=== {Config.CODE_NAMES[code]['ar']} (سياقي) ===") | |
| for source in sources[:Config.MAX_CROSS_CODE_ARTICLES]: | |
| statute_count += 1 | |
| art_num = source.article_metadata.article_number | |
| context_parts.append(f"\n[STATUTE_{statute_count}]") | |
| context_parts.append(f"Article: {art_num}") | |
| context_parts.append(f"Code: {code}") | |
| context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}") | |
| context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}") | |
| context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n") | |
| source_registry["statutes"][art_num] = { | |
| "text": source.content, | |
| "full_text": source.full_text if source.full_text else source.content, | |
| "code": code, | |
| "code_name_fr": Config.CODE_NAMES[code]["fr"], | |
| "code_name_ar": Config.CODE_NAMES[code]["ar"], | |
| "relevance": source.relevance.combined_score, | |
| "primary_code": False | |
| } | |
| if juris_sources: | |
| if query_analysis.original_language == "ar": | |
| context_parts.append("\n=== JURISPRUDENCE (الأحكام القضائية) ===") | |
| else: | |
| context_parts.append("\n=== JURISPRUDENCE (CASE LAW SUPPORT) ===") | |
| juris_by_level = {"high": [], "medium": [], "low": []} | |
| for source in juris_sources: | |
| level = source.relevance.relevance_level | |
| if level in juris_by_level: | |
| juris_by_level[level].append(source) | |
| for level in ["high", "medium", "low"]: | |
| if juris_by_level[level]: | |
| level_display = level.upper() | |
| if query_analysis.original_language == "ar": | |
| level_names = {"high": "عالية", "medium": "متوسطة", "low": "منخفضة"} | |
| context_parts.append(f"\n=== أحكام ذات أهمية {level_names[level]} ===") | |
| else: | |
| context_parts.append(f"\n=== {level_display} RELEVANCE JURISPRUDENCE ===") | |
| for i, source in enumerate(juris_by_level[level], 1): | |
| context_parts.append(f"\n[JURIS_{level.upper()}_{i}]") | |
| if source.juridiction: | |
| context_parts.append(f"Juridiction: {source.juridiction}") | |
| if source.date_decision: | |
| context_parts.append(f"Date: {source.date_decision}") | |
| if source.code_fr or source.code_ar: | |
| code_display = source.code_ar if query_analysis.original_language == "ar" else source.code_fr | |
| context_parts.append(f"Code: {code_display}") | |
| if source.tags: | |
| tags_display = ", ".join(source.tags[:8]) | |
| context_parts.append(f"Tags: {tags_display}") | |
| if source.retrieval_method: | |
| context_parts.append(f"Search Method: {source.retrieval_method}") | |
| context_parts.append(f"Full Decision / القرار الكامل:") | |
| context_parts.append(f"{source.full_text if source.full_text else source.content}") | |
| context_parts.append(f"Relevance Score: {source.relevance.combined_score:.3f}\n") | |
| source_registry["jurisprudence"][f"JURIS_{level.upper()}_{i}"] = { | |
| "content": source.content, | |
| "full_text": source.full_text if source.full_text else source.content, | |
| "relevance": source.relevance.combined_score, | |
| "source_id": source.source_id, | |
| "level": level, | |
| "method": source.retrieval_method, | |
| "juridiction": source.juridiction, | |
| "date": source.date_decision, | |
| "tags": source.tags, | |
| "code_fr": source.code_fr, | |
| "code_ar": source.code_ar | |
| } | |
| context_parts.append("\n=== CRITICAL INSTRUCTIONS ===") | |
| context_parts.append(f"Primary Legal Code: {Config.CODE_NAMES[primary_code]['fr']}") | |
| context_parts.append(f"مجلة القانون الأساسي: {Config.CODE_NAMES[primary_code]['ar']}") | |
| context_parts.append(f"Total statutes retrieved: {statute_count}") | |
| context_parts.append(f"Total jurisprudence decisions: {len(juris_sources)}") | |
| context_parts.append(f"Cited articles requested: {query_analysis.cited_articles}") | |
| context_parts.append(f"Missing from retrieval: {retrieval_meta['missing_articles']}") | |
| context_parts.append(f"Search methods used: {', '.join(retrieval_meta.get('search_methods', []))}") | |
| context_parts.append("\n=== JURISPRUDENCE SPECIFIC RULES ===") | |
| context_parts.append("1. You CAN and SHOULD reference relevant jurisprudence principles") | |
| context_parts.append("2. When citing jurisprudence, mention the court and date if available") | |
| context_parts.append("3. Focus on the legal principles established in the jurisprudence") | |
| context_parts.append("4. Use jurisprudence to support statutory interpretation") | |
| context_parts.append("5. Highlight how jurisprudence applies to the specific case") | |
| context_parts.append("6. Consider jurisprudence from ALL relevant codes, not just the primary one") | |
| context_parts.append("\n=== STRICT RULES TO PREVENT HALLUCINATIONS ===") | |
| context_parts.append("1. ONLY cite articles that appear in the retrieved sources above") | |
| context_parts.append(" استشهد فقط بالمواد التي تظهر في المصادر المسترجعة أعلاه") | |
| context_parts.append("2. If NO statutes are retrieved, DO NOT cite any articles") | |
| context_parts.append(" إذا لم يتم استرجاع أي مواد قانونية، لا تستشهد بأي مواد") | |
| context_parts.append("3. You CAN reference principles from jurisprudence, but DO NOT cite article numbers from jurisprudence") | |
| context_parts.append(" يمكنك الإشارة إلى المبادئ من الأحكام القضائية، لكن لا تستشهد بأرقام المواد من الأحكام") | |
| context_parts.append("\nYOU MAY ONLY CITE ARTICLES THAT APPEAR ABOVE.") | |
| context_parts.append("يُسمح لك بالاستشهاد فقط بالمواد التي تظهر أعلاه.") | |
| context_parts.append("When citing, ALWAYS specify which code the article comes from.") | |
| context_parts.append("عند الاستشهاد، حدد دائمًا المجلة التي ينتمي إليها النص.") | |
| context_parts.append("DO NOT cite, quote, or reference any article not explicitly retrieved.") | |
| context_parts.append("لا تستشهد أو تنقل أو تشير إلى أي مادة لم يتم استرجاعها صراحة.") | |
| context_parts.append("USE JURISPRUDENCE FROM ALL RELEVANT CODES TO SUPPORT AND ILLUSTRATE LEGAL PRINCIPLES.") | |
| context_parts.append("استخدم الأحكام القضائية من جميع المجلات ذات الصلة لدعم وتوضيح المبادئ القانونية.") | |
| context = "\n".join(context_parts) | |
| return context, source_registry | |
| # ============================================================ | |
| # ANSWER GENERATOR PROFESSIONNEL | |
| # ============================================================ | |
| class EnterpriseAnswerGenerator: | |
| def generate_answer(query: str, context: str, source_registry: Dict[str, Any], | |
| query_analysis: QueryAnalysis) -> Tuple[str, List[CitedSource]]: | |
| """Génère une réponse juridique améliorée""" | |
| language = query_analysis.original_language | |
| system_prompt = EnterpriseAnswerGenerator._build_system_prompt(language, source_registry, query_analysis) | |
| user_prompt = f"""CONTEXTE ET SOURCES RÉCUPÉRÉES / السياق والمصادر المسترجعة: | |
| {context} | |
| QUESTION ORIGINALE / السؤال الأصلي: {query} | |
| QUESTION TRADUITE (pour référence) / السؤال المترجم (للإشارة): {query_analysis.translated_query} | |
| RÈGLES STRICTES POUR LA GÉNÉRATION / قواعد صارمة للتوليد: | |
| 1. Basez-vous uniquement sur les sources récupérées ci-dessus | |
| اعتمد فقط على المصادر المسترجعة أعلاه | |
| 2. Citez exactement les articles comme ils apparaissent dans les sources | |
| استشهد بالضبط بالمواد كما تظهر في المصادر | |
| 3. Si une source n'est pas disponible, expliquez clairement cette limite | |
| إذا لم يكن المصدر متوفراً، اشرح هذا القيد بوضوح | |
| 4. Fournissez une analyse juridique complète et pratique | |
| قدم تحليلاً قانونياً شاملاً وعملياً | |
| 5. Adaptez la réponse à la langue de l'utilisateur ({language}) | |
| قم بتكييف الإجابة مع لغة المستخدم ({language}) | |
| 6. Utilisez la jurisprudence pour illustrer et soutenir les principes légaux | |
| استخدم الأحكام القضائية لتوضيح ودعم المبادئ القانونية | |
| 7. Considérez la jurisprudence de TOUS les codes pertinents | |
| ضع في الاعتبار الأحكام القضائية من جميع المجلات ذات الصلة | |
| Générez une réponse juridique complète, précise et pratique. | |
| قم بتوليد إجابة قانونية شاملة ودقيقة وعملية.""" | |
| try: | |
| response = chat_client.chat.completions.create( | |
| model=Config.CHAT_MODEL, | |
| messages=[ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": user_prompt} | |
| ], | |
| temperature=Config.TEMP_GENERATION, | |
| max_tokens=4000 | |
| ) | |
| answer = response.choices[0].message.content.strip() | |
| cited_sources = EnterpriseAnswerGenerator._extract_citations(answer, source_registry) | |
| logger.info(f"✅ Réponse générée ({len(answer)} caractères)") | |
| logger.info(f"📌 Sources citées: {len(cited_sources)}") | |
| return answer, cited_sources | |
| except Exception as e: | |
| logger.error(f"Erreur génération réponse: {e}") | |
| return "Une erreur est survenue lors de la génération de la réponse. Veuillez réessayer.", [] | |
| def _build_system_prompt(language: str, source_registry: Dict[str, Any], query_analysis: QueryAnalysis) -> str: | |
| """Construit le prompt système amélioré""" | |
| primary_code = query_analysis.primary_legal_code.value | |
| primary_code_name_fr = Config.CODE_NAMES[primary_code]["fr"] | |
| primary_code_name_ar = Config.CODE_NAMES[primary_code]["ar"] | |
| articles_by_code = {} | |
| for article, info in source_registry["statutes"].items(): | |
| code = info["code"] | |
| if code not in articles_by_code: | |
| articles_by_code[code] = [] | |
| code_name = info["code_name_ar"] if language == "ar" else info["code_name_fr"] | |
| articles_by_code[code].append(f"{article} ({code_name})") | |
| available_text = "" | |
| for code, articles in articles_by_code.items(): | |
| code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"] | |
| available_text += f"\n{code_name}: {', '.join(articles)}" | |
| juris_count = len(source_registry.get("jurisprudence", {})) | |
| missing_articles = source_registry.get("missing_articles", []) | |
| juris_by_code = {} | |
| for juris_id, juris_info in source_registry.get("jurisprudence", {}).items(): | |
| code_fr = juris_info.get("code_fr", "") | |
| code_ar = juris_info.get("code_ar", "") | |
| for code, code_info in Config.CODE_NAMES.items(): | |
| if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar): | |
| if code not in juris_by_code: | |
| juris_by_code[code] = 0 | |
| juris_by_code[code] += 1 | |
| break | |
| juris_by_code_text = "" | |
| for code, count in juris_by_code.items(): | |
| code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"] | |
| juris_by_code_text += f"\n{code_name}: {count} decisions" | |
| if language == "ar": | |
| return f"""أنت خبير قانوني تونسي محترف متخصص في جميع المجلات العشرة التونسية. | |
| المجلة الأساسية المعنية: {primary_code_name_ar} | |
| المواد القانونية المتاحة:{available_text if articles_by_code else " لا توجد مواد قانونية مسترجعة"} | |
| عدد الأحكام القضائية المتاحة: {juris_count} حكم | |
| توزيع الأحكام القضائية حسب المجلة:{juris_by_code_text if juris_by_code_text else " لا توجد أحكام قضائية مصنفة"} | |
| المواد المطلوبة وغير المتوفرة: {', '.join(missing_articles)} | |
| أنت خبير في: | |
| 1. تفسير النصوص القانونية التونسية | |
| 2. تحليل الأحكام القضائية وتطبيقها على الحالات الواقعية | |
| 3. تقديم نصائح قانونية عملية ومفصلة | |
| 4. التمييز بين المصادر القانونية المختلفة | |
| 5. استخدام الاجتهاد القضائي من جميع المجلات ذات الصلة لتوضيح المبادئ القانونية | |
| قواعد صارمة: | |
| 1. لا تخترع أي مواد أو أحكام غير موجودة في المصادر | |
| 2. إذا لم تجد مصدراً، اعترف بذلك واشرح البدائل | |
| 3. كن دقيقاً في الاستشهادات والإحالات | |
| 4. قدم إجابة متوازنة وعملية | |
| 5. استخدم الأحكام القضائية من جميع المجلات ذات الصلة لتوضيح كيفية تطبيق النصوص القانونية | |
| استخدم الأحكام القضائية المتاحة لتوضيح: | |
| - كيفية تفسير المحاكم للنصوص القانونية | |
| - المبادئ القانونية المستقرة في الاجتهاد | |
| - كيفية تطبيق القانون على حالات مشابهة | |
| - الاتجاهات الحديثة في التفسير القضائي | |
| - الاختلافات أو التشابهات بين تفسيرات المحاكم للمواد المختلفة | |
| قم بتحليل السؤال بدقة وقدم إجابة شاملة تعتمد على المصادر المتاحة من جميع المجلات ذات الصلة.""" | |
| else: | |
| return f"""You are a professional Tunisian legal expert specializing in all 10 Tunisian codes. | |
| Primary Legal Code: {primary_code_name_fr} | |
| Available Legal Articles:{available_text if articles_by_code else " NO statutes retrieved"} | |
| Available Jurisprudence Decisions: {juris_count} decisions | |
| Jurisprudence Distribution by Code:{juris_by_code_text if juris_by_code_text else " No jurisprudence classified by code"} | |
| Requested but Unavailable Articles: {', '.join(missing_articles)} | |
| You are expert in: | |
| 1. Interpreting Tunisian legal texts | |
| 2. Analyzing case law and applying it to real cases | |
| 3. Providing practical, detailed legal advice | |
| 4. Distinguishing between different legal sources | |
| 5. Using jurisprudence from ALL relevant codes to illustrate legal principles | |
| Strict Rules: | |
| 1. DO NOT invent any articles or jurisprudence not in the sources | |
| 2. If a source is not found, acknowledge this and explain alternatives | |
| 3. Be precise in citations and references | |
| 4. Provide balanced, practical advice | |
| 5. Use available jurisprudence from ALL relevant codes to clarify how legal texts are applied | |
| Use available jurisprudence to illustrate: | |
| - How courts interpret legal texts | |
| - Established legal principles in case law | |
| - How the law is applied to similar cases | |
| - Recent trends in judicial interpretation | |
| - Differences or similarities between court interpretations of different articles | |
| Analyze the question accurately and provide a comprehensive answer based on available sources from ALL relevant codes.""" | |
| def _extract_citations(answer: str, source_registry: Dict[str, Any]) -> List[CitedSource]: | |
| """Extrait les citations de la réponse""" | |
| citations = [] | |
| for article, info in source_registry["statutes"].items(): | |
| normalized_article = ArticleNumberNormalizer.normalize(article) | |
| patterns = [ | |
| f"(?:الفصل|فصل|Article|article|المادة|مادة)\\s*{normalized_article}\\b", | |
| f"\\b{normalized_article}\\b(?!\\s*bis|\\s*ter|\\s*quater)" | |
| ] | |
| for pattern in patterns: | |
| matches = re.finditer(pattern, answer, re.IGNORECASE) | |
| for match in matches: | |
| context = answer[max(0, match.start()-50):min(len(answer), match.end()+50)] | |
| citations.append(CitedSource( | |
| article_number=article, | |
| code_type=LegalCode.from_string(info["code"]), | |
| citation_context=context, | |
| citation_position=match.start(), | |
| retrieval_status=RetrievalStatus.SUCCESSFULLY_RETRIEVED | |
| )) | |
| return citations | |
| # ============================================================ | |
| # VALIDATOR PROFESSIONNEL | |
| # ============================================================ | |
| class AnswerValidator: | |
| def validate(answer: str, cited_sources: List[CitedSource], source_registry: Dict[str, Any]) -> ValidationResult: | |
| """Valide la réponse générée""" | |
| hallucinated = [] | |
| missing_retrievals = [] | |
| errors = [] | |
| warnings = [] | |
| retrieved_lookup = {} | |
| for article, info in source_registry["statutes"].items(): | |
| normalized = ArticleNumberNormalizer.normalize(article) | |
| retrieved_lookup[normalized] = { | |
| "code": info["code"], | |
| "original": article, | |
| "info": info | |
| } | |
| for citation in cited_sources: | |
| normalized_cite = ArticleNumberNormalizer.normalize(citation.article_number) | |
| if normalized_cite in retrieved_lookup: | |
| retrieved_info = retrieved_lookup[normalized_cite] | |
| if retrieved_info["code"] == citation.code_type.value: | |
| citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED | |
| else: | |
| warnings.append(f"Article {citation.article_number} cited with code {citation.code_type.value} but retrieved from {retrieved_info['code']}") | |
| citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED | |
| else: | |
| hallucinated.append(f"{citation.article_number} ({citation.code_type.value})") | |
| citation.retrieval_status = RetrievalStatus.CITED_NOT_RETRIEVED | |
| errors.append(f"Article {citation.article_number} from {citation.code_type.value} cited but NOT retrieved") | |
| all_article_numbers = ArticleNumberNormalizer.extract_all_numbers(answer) | |
| for art_num in all_article_numbers: | |
| normalized_art_num = ArticleNumberNormalizer.normalize(art_num) | |
| found_in_sources = normalized_art_num in retrieved_lookup | |
| already_cited = any( | |
| ArticleNumberNormalizer.normalize(citation.article_number) == normalized_art_num | |
| for citation in cited_sources | |
| ) | |
| if not found_in_sources and not already_cited: | |
| warnings.append(f"Potential uncited reference to article {art_num}") | |
| for article in source_registry.get("cited_articles", []): | |
| normalized = ArticleNumberNormalizer.normalize(article) | |
| if normalized not in retrieved_lookup: | |
| missing_retrievals.append(article) | |
| warnings.append(f"Requested article {article} was NOT retrieved") | |
| if hallucinated: | |
| confidence = 0.0 | |
| elif missing_retrievals: | |
| confidence = 0.7 | |
| elif warnings: | |
| confidence = 0.85 | |
| else: | |
| confidence = 0.95 | |
| is_valid = len(hallucinated) == 0 | |
| if not is_valid: | |
| errors.append(f"CRITICAL: {len(hallucinated)} hallucinated citations detected") | |
| return ValidationResult( | |
| is_valid=is_valid, | |
| confidence_score=confidence, | |
| hallucinated_citations=hallucinated, | |
| missing_retrievals=missing_retrievals, | |
| validation_errors=errors, | |
| validation_warnings=warnings | |
| ) | |
| # ============================================================ | |
| # CHATBOT FINAL AMÉLIORÉ | |
| # ============================================================ | |
| class UltimateLegalChatbot: | |
| """Chatbot juridique ultime avec toutes les améliorations""" | |
| def __init__(self): | |
| self.query_analyzer = EnterpriseQueryAnalyzer() | |
| self.search_engine = EnhancedEnterpriseSearchEngine() | |
| self.context_builder = EnterpriseContextBuilder() | |
| self.answer_generator = EnterpriseAnswerGenerator() | |
| self.validator = AnswerValidator() | |
| def process_query(self, query: str) -> LegalAnswer: | |
| """Traite une requête avec toutes les améliorations""" | |
| logger.info(f"\n{'='*120}") | |
| logger.info("🚀 ULTIMATE LEGAL CHATBOT - 10 TUNISIAN CODES (HYBRID SEARCH)") | |
| logger.info(f"{'='*120}\n") | |
| start_time = datetime.now() | |
| logger.info("📊 ÉTAPE 1: Analyse de la requête") | |
| query_analysis = self.query_analyzer.analyze_query(query) | |
| logger.info("\n🔍 ÉTAPE 2: Récupération des sources améliorée avec recherche hybride") | |
| search_results = self.search_engine.search(query_analysis) | |
| logger.info("\n🧱 ÉTAPE 3: Construction du contexte améliorée") | |
| context, source_registry = self.context_builder.build_context(search_results) | |
| logger.info("\n✍️ ÉTAPE 4: Génération de la réponse améliorée") | |
| answer_text, cited_sources = self.answer_generator.generate_answer( | |
| query, context, source_registry, query_analysis | |
| ) | |
| logger.info("\n✅ ÉTAPE 5: Validation de la réponse") | |
| validation_result = self.validator.validate(answer_text, cited_sources, source_registry) | |
| all_sources = [] | |
| for level in ["high", "medium", "low"]: | |
| all_sources.extend(search_results["statute_sources"].get(level, [])) | |
| all_sources.extend(search_results["jurisprudence_sources"]) | |
| end_time = datetime.now() | |
| processing_time = (end_time - start_time).total_seconds() | |
| juris_code_distribution = {} | |
| for source in search_results["jurisprudence_sources"]: | |
| code_fr = source.code_fr | |
| code_ar = source.code_ar | |
| for code_key, code_info in Config.CODE_NAMES.items(): | |
| if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar): | |
| if code_key not in juris_code_distribution: | |
| juris_code_distribution[code_key] = 0 | |
| juris_code_distribution[code_key] += 1 | |
| legal_answer = LegalAnswer( | |
| answer_text=answer_text, | |
| retrieved_sources=all_sources, | |
| cited_sources=cited_sources, | |
| validation_result=validation_result, | |
| query_analysis=query_analysis, | |
| processing_metadata={ | |
| "processing_time": processing_time, | |
| "retrieval_stats": search_results["retrieval_metadata"], | |
| "translation_applied": query_analysis.original_language == "ar", | |
| "jurisprudence_found": len(search_results["jurisprudence_sources"]), | |
| "primary_code": query_analysis.primary_legal_code.value, | |
| "secondary_codes": [c.value for c in query_analysis.secondary_codes], | |
| "total_codes_available": 10, | |
| "search_methods": search_results["retrieval_metadata"]["search_methods"], | |
| "hybrid_search_used": True, | |
| "jurisprudence_code_distribution": juris_code_distribution, | |
| "timestamp": datetime.now().isoformat() | |
| }, | |
| confidence_score=validation_result.confidence_score * query_analysis.code_confidence | |
| ) | |
| logger.info(f"\n✅ Traitement terminé en {processing_time:.2f} secondes") | |
| logger.info(f"📊 Résumé: {len(all_sources)} sources totales") | |
| logger.info(f"🎯 Système 10 codes tunisiens avec recherche hybride opérationnel") | |
| return legal_answer | |
| # ============================================================ | |
| # CHAINLIT APPLICATION (CHAT INTERFACE) | |
| # ============================================================ | |
| # ... (all previous imports and classes remain exactly the same until the Chainlit part) ... | |
| # ============================================================ | |
| # CHAINLIT APPLICATION (CHAT INTERFACE) - MODIFIED | |
| # ============================================================ | |
| chatbot = UltimateLegalChatbot() | |
| async def start(): | |
| """Initialise la session de chat avec un message d'accueil minimal""" | |
| await cl.Message( | |
| content="""Legal Assistant | مساعد قانوني | |
| Ask your legal question | اطرح سؤالك القانوني""" ).send() | |
| async def main(message: cl.Message): | |
| """Traite le message de l'utilisateur et renvoie la réponse + sources complètes sans termes techniques""" | |
| user_query = message.content.strip() | |
| if not user_query: | |
| await cl.Message(content="Veuillez poser une question valide.").send() | |
| return | |
| async with cl.Step(name="Analyse et recherche en cours", type="loading"): | |
| legal_answer = await asyncio.to_thread(chatbot.process_query, user_query) | |
| await cl.Message(content=legal_answer.answer_text).send() | |
| sources_message = "" | |
| statutes = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.STATUTE] | |
| if statutes: | |
| sources_message += "### 📜 Articles de loi (texte intégral)\n\n" | |
| for idx, stat in enumerate(statutes, 1): | |
| if stat.article_metadata: | |
| art_num = stat.article_metadata.article_number | |
| if legal_answer.query_analysis.original_language == "ar": | |
| code_name = stat.article_metadata.code_name_ar | |
| else: | |
| code_name = stat.article_metadata.code_name_fr | |
| sources_message += f"**{idx}. {code_name} – Article {art_num}**\n" | |
| sources_message += f"```\n{stat.full_text}\n```\n\n" | |
| juris = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.JURISPRUDENCE] | |
| if juris: | |
| sources_message += "### ⚖️ Décisions de jurisprudence (texte intégral)\n\n" | |
| for idx, dec in enumerate(juris, 1): | |
| title = f"**{idx}. {dec.juridiction or 'Décision'}**" | |
| if dec.date_decision: | |
| title += f" – {dec.date_decision}" | |
| sources_message += title + "\n" | |
| if legal_answer.query_analysis.original_language == "ar": | |
| code_display = dec.code_ar | |
| else: | |
| code_display = dec.code_fr | |
| if code_display: | |
| sources_message += f"*Code : {code_display}*\n" | |
| sources_message += f"```\n{dec.full_text}\n```\n\n" | |
| if sources_message: | |
| await cl.Message(content=sources_message).send() | |
| else: | |
| await cl.Message(content="*Aucune source textuelle n'a été trouvée pour cette question.*").send() | |
| if __name__ == "__main__": | |
| from chainlit.cli import run | |
| run() |