import re import json import os from dotenv import load_dotenv from pymongo import MongoClient from openai import OpenAI, AzureOpenAI import numpy as np from sklearn.metrics.pairwise import cosine_similarity from typing import List, Dict, Tuple, Optional, Any, Set import logging from functools import lru_cache from dataclasses import dataclass, field from datetime import datetime import hashlib from enum import Enum import time import math from collections import Counter import asyncio import chainlit as cl # Charger les variables d'environnement load_dotenv() logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') logger = logging.getLogger(__name__) # ============================================================ # CONFIGURATION PROFESSIONNELLE AVEC NOMS EXACTS # ============================================================ class Config: """Professional configuration for all Tunisian codes""" MONGO_URI = os.getenv("MONGO_URI") DB_CODE = os.getenv("DB_CODE") COLLECTIONS = { "CSP": os.getenv("CSP", "csp"), "DROITS_REELS": os.getenv("DROITS_REELS", "code_droits_reels"), "OBLIGATIONS_CONTRATS": os.getenv("OBLIGATIONS_CONTRATS", "code_des_obligations_et_des_contrats"), "PROCEDURE_CIVILE": os.getenv("PROCEDURE_CIVILE", "code_de_procédure_civile_et_commerciale"), "CODE_TRAVAIL": os.getenv("CODE_TRAVAIL", "code-de-travail"), "DROITS_PROCEDURES_FISCAUX": os.getenv("DROITS_PROCEDURES_FISCAUX", "code-des-droits-et-procedures-fiscaux"), "DROIT_INTERNATIONAL_PRIVE": os.getenv("DROIT_INTERNATIONAL_PRIVE", "code-du-droit-international-privé"), "CODE_PENAL": os.getenv("CODE_PENAL", "code-pénal"), "CODE_COMMERCE": os.getenv("CODE_COMMERCE", "code_de_commerce"), "PROCEDURES_PENALES": os.getenv("PROCEDURES_PENALES", "code_des_procedures_penales") } DB_JURIS = os.getenv("DB_JURIS") COL_JURIS = os.getenv("COL_JURIS") AZURE_ENDPOINT = os.getenv("AZURE_ENDPOINT") AZURE_API_KEY = os.getenv("AZURE_API_KEY") AZURE_API_VERSION = os.getenv("AZURE_API_VERSION") EMBEDDING_MODEL = os.getenv("EMBEDDING_MODEL") CHAT_MODEL = os.getenv("CHAT_MODEL") RELEVANCE_THRESHOLDS = { "CSP": {"HIGH": 0.78, "MEDIUM": 0.65, "MINIMUM": 0.55}, "DROITS_REELS": {"HIGH": 0.75, "MEDIUM": 0.62, "MINIMUM": 0.52}, "OBLIGATIONS_CONTRATS": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50}, "PROCEDURE_CIVILE": {"HIGH": 0.65, "MEDIUM": 0.55, "MINIMUM": 0.45}, "CODE_TRAVAIL": {"HIGH": 0.73, "MEDIUM": 0.61, "MINIMUM": 0.51}, "DROITS_PROCEDURES_FISCAUX": {"HIGH": 0.68, "MEDIUM": 0.56, "MINIMUM": 0.46}, "DROIT_INTERNATIONAL_PRIVE": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50}, "CODE_PENAL": {"HIGH": 0.76, "MEDIUM": 0.64, "MINIMUM": 0.54}, "CODE_COMMERCE": {"HIGH": 0.71, "MEDIUM": 0.59, "MINIMUM": 0.49}, "PROCEDURES_PENALES": {"HIGH": 0.74, "MEDIUM": 0.62, "MINIMUM": 0.52} } # Optimized jurisprudence thresholds JURIS_HIGH_RELEVANCE = 0.68 JURIS_MEDIUM_RELEVANCE = 0.52 JURIS_MINIMUM_RELEVANCE = 0.35 TEMP_ANALYSIS = 0.05 TEMP_CLASSIFICATION = 0.1 TEMP_GENERATION = 0.05 MAX_HIGH_ARTICLES = 12 MAX_MEDIUM_ARTICLES = 8 MAX_LOW_ARTICLES = 4 MAX_JURIS_DOCS = 20 # REDUCED FROM 60 MAX_JURIS_RETRIEVAL = 80 # REDUCED FROM 250 MAX_CROSS_CODE_ARTICLES = 6 MAX_LEXICAL_RESULTS = 30 # Optimized secondary code mapping SECONDARY_CODE_MAPPING = { "CSP": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "DROITS_REELS"], "DROITS_REELS": ["PROCEDURE_CIVILE", "CSP", "OBLIGATIONS_CONTRATS", "CODE_COMMERCE"], "OBLIGATIONS_CONTRATS": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "CODE_TRAVAIL", "CSP"], "PROCEDURE_CIVILE": ["CSP", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL"], "CODE_TRAVAIL": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CSP"], "DROITS_PROCEDURES_FISCAUX": ["CODE_COMMERCE", "PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"], "DROIT_INTERNATIONAL_PRIVE": ["CSP", "PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS"], "CODE_PENAL": ["PROCEDURES_PENALES", "PROCEDURE_CIVILE", "CSP"], "CODE_COMMERCE": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL", "DROITS_REELS"], "PROCEDURES_PENALES": ["CODE_PENAL", "PROCEDURE_CIVILE", "CSP"] } CODE_NAMES = { "CSP": { "fr": "Code du Statut Personnel", "ar": "مجلة الأحوال الشخصية", "keywords": ["mariage", "divorce", "héritage", "succession", "paternité", "نكاح", "طلاق", "إرث", "ميراث", "نسب", "نفقة", "حضانة", "زواج", "فرقة"] }, "DROITS_REELS": { "fr": "Code des Droits Réels", "ar": "مجلة الحقوق العينية", "keywords": ["propriété", "immobilier", "hypothèque", "servitude", "usufruit", "ملكية", "عقار", "رهن", "أراضي", "عقارية", "حيازة", "تملك"] }, "OBLIGATIONS_CONTRATS": { "fr": "Code des Obligations et des Contrats", "ar": "مجلة الالتزامات والعقود", "keywords": ["contrat", "obligation", "responsabilité", "délit", "عقد", "التزام", "مسؤولية", "خطأ", "تعويض", "إبرام", "فسخ", "إبطال"] }, "PROCEDURE_CIVILE": { "fr": "Code de Procédure Civile et Commerciale", "ar": "مجلة المرافعات المدنية والتجارية", "keywords": ["procédure", "appel", "recours", "jugement", "تعقيب", "نقض", "محكمة", "حكم", "قرار", "طعن", "إجراءات", "استئناف", "طلبات جديدة", "الطلبات الجديدة"] }, "CODE_TRAVAIL": { "fr": "Code du Travail", "ar": "مجلة الشغل", "keywords": ["travail", "emploi", "licenciement", "salaire", "contrat", "شغل", "عقد شغل", "فصل", "أجير", "أجرة", "إجازة", "تعويض"] }, "DROITS_PROCEDURES_FISCAUX": { "fr": "Code des Droits et Procédures Fiscaux", "ar": "مجلة الحقوق والإجراءات الضريبية", "keywords": ["fiscal", "impôt", "taxe", "droit fiscal", "procédure fiscale", "ضريبة", "جباية", "ديوان", "غرامة", "تحصيل", "تهرب"] }, "DROIT_INTERNATIONAL_PRIVE": { "fr": "Code de Droit International Privé", "ar": "مجلة القانون الدولي الخاص", "keywords": ["international", "conflit de lois", "nationalité", "étranger", "تنازع القوانين", "اختصاص دولي", "جنسية", "أجانب", "إقليمية"] }, "CODE_PENAL": { "fr": "Code Pénal", "ar": "المجلة الجزائية", "keywords": ["pénal", "crime", "délit", "contravention", "peine", "جناية", "جنحة", "مخالفة", "عقوبة", "سجن", "حبس", "جرم"] }, "CODE_COMMERCE": { "fr": "Code de Commerce", "ar": "مجلة التجارية", "keywords": ["commerce", "commerçant", "entreprise", "société", "faillite", "تاجر", "تجار", "شركة", "سجل تجاري", "إفلاس", "تسوية"] }, "PROCEDURES_PENALES": { "fr": "Code des Procédures Pénales", "ar": "مجلة الإجراءات الجزائية", "keywords": ["procédure pénale", "enquête", "instruction", "تحقيق", "تحقيق جزائي", "قاضي التحقيق", "إحالة", "نيابة", "محاكمة"] } } # Initialisation des clients mongo_client = MongoClient(Config.MONGO_URI) embedding_client = OpenAI(base_url=f"{Config.AZURE_ENDPOINT}openai/v1/", api_key=Config.AZURE_API_KEY) chat_client = AzureOpenAI(api_key=Config.AZURE_API_KEY, azure_endpoint=Config.AZURE_ENDPOINT, api_version=Config.AZURE_API_VERSION) # ============================================================ # TRANSLATION SERVICE # ============================================================ class TranslationService: """Service de traduction professionnel""" _translation_cache = {} @staticmethod def translate_text(text: str, source_lang: str, target_lang: str) -> str: if source_lang == target_lang or not text: return text cache_key = f"{source_lang}_{target_lang}_{hashlib.md5(text.encode()).hexdigest()}" if cache_key in TranslationService._translation_cache: return TranslationService._translation_cache[cache_key] try: if target_lang == "ar": system_content = "أنت مترجم قانوني محترف متخصص في الترجمة من الفرنسية إلى العربية. حافظ على الدقة القانونية والمصطلحات الفنية." prompt = f"""ترجم النص القانوني التالي بدقة مع الحفاظ على: 1. المعنى القانوني الدقيق 2. المصطلحات القانونية المتخصصة 3. الأرقام والمراجع القانونية 4. السياق القانوني التونسي النص الفرنسي: {text} الترجمة العربية (تجنب الإضافة أو الحذف، كن دقيقًا):""" else: system_content = "Tu es un traducteur juridique professionnel spécialisé en droit tunisien. Préserve la précision juridique et la terminologie technique." prompt = f"""Traduis ce texte juridique avec précision en préservant: 1. Le sens juridique exact 2. La terminologie juridique spécialisée 3. Les chiffres et références légales 4. Le contexte juridique tunisien Texte arabe: {text} Traduction française (sans ajout ni omission, sois précis):""" response = chat_client.chat.completions.create( model=Config.CHAT_MODEL, messages=[ {"role": "system", "content": system_content}, {"role": "user", "content": prompt} ], temperature=0.1, max_tokens=2000 ) translation = response.choices[0].message.content.strip() TranslationService._translation_cache[cache_key] = translation logger.info(f"✅ Traduction {source_lang} → {target_lang} effectuée") return translation except Exception as e: logger.error(f"Erreur de traduction: {e}") return text @staticmethod def translate_ar_to_fr(text: str) -> str: return TranslationService.translate_text(text, "ar", "fr") @staticmethod def translate_fr_to_ar(text: str) -> str: return TranslationService.translate_text(text, "fr", "ar") @staticmethod def detect_and_translate(query: str) -> Dict[str, str]: arabic_chars = len(re.findall(r'[\u0600-\u06FF]', query)) total_chars = len(query.replace(' ', '')) if total_chars > 0 and (arabic_chars / total_chars) > 0.2: translated = TranslationService.translate_ar_to_fr(query) return { "original_query": query, "translated_query": translated, "original_language": "ar", "search_language": "fr" } else: return { "original_query": query, "translated_query": query, "original_language": "fr", "search_language": "fr" } # ============================================================ # ENUMS & DATA STRUCTURES # ============================================================ class SourceType(Enum): STATUTE = "statute" JURISPRUDENCE = "jurisprudence" class LegalCode(Enum): CSP = "CSP" DROITS_REELS = "DROITS_REELS" OBLIGATIONS_CONTRATS = "OBLIGATIONS_CONTRATS" PROCEDURE_CIVILE = "PROCEDURE_CIVILE" CODE_TRAVAIL = "CODE_TRAVAIL" DROITS_PROCEDURES_FISCAUX = "DROITS_PROCEDURES_FISCAUX" DROIT_INTERNATIONAL_PRIVE = "DROIT_INTERNATIONAL_PRIVE" CODE_PENAL = "CODE_PENAL" CODE_COMMERCE = "CODE_COMMERCE" PROCEDURES_PENALES = "PROCEDURES_PENALES" @classmethod def from_string(cls, value: str): try: normalized = value.upper().replace("-", "_").replace(" ", "_") return cls(normalized) except: return cls.CSP class RetrievalStatus(Enum): SUCCESSFULLY_RETRIEVED = "retrieved" CITED_NOT_RETRIEVED = "cited_not_retrieved" @dataclass class ArticleMetadata: article_number: str normalized_number: str article_text_fr: str article_text_ar: str code_type: LegalCode code_name_fr: str code_name_ar: str chapter: Optional[str] = None section: Optional[str] = None pdf_source: Optional[str] = None @dataclass class RelevanceScore: similarity: float topic_overlap: float entity_match: float article_match: float code_relevance: float combined_score: float relevance_level: str confidence: float @dataclass class RetrievedSource: source_id: str source_type: SourceType article_metadata: Optional[ArticleMetadata] content: str relevance: RelevanceScore retrieval_timestamp: datetime retrieval_method: str summary: str = "" full_text: str = "" primary_code: bool = True tags: List[str] = field(default_factory=list) code_fr: str = "" code_ar: str = "" juridiction: str = "" date_decision: str = "" @dataclass class CitedSource: article_number: str code_type: LegalCode citation_context: str citation_position: int retrieval_status: RetrievalStatus @dataclass class ValidationResult: is_valid: bool confidence_score: float hallucinated_citations: List[str] missing_retrievals: List[str] validation_errors: List[str] validation_warnings: List[str] @dataclass class QueryAnalysis: original_query: str translated_query: str original_language: str search_language: str extracted_topics: List[str] legal_entities: List[str] cited_articles: List[str] primary_legal_code: LegalCode secondary_codes: List[LegalCode] question_type: str complexity_score: float search_queries: List[str] requires_statutory_law: bool code_confidence: float @dataclass class LegalAnswer: answer_text: str retrieved_sources: List[RetrievedSource] cited_sources: List[CitedSource] validation_result: ValidationResult query_analysis: QueryAnalysis processing_metadata: Dict[str, Any] confidence_score: float # ============================================================ # ARTICLE NUMBER NORMALIZER PROFESSIONNEL # ============================================================ class ArticleNumberNormalizer: @staticmethod def normalize(article_ref: str) -> str: """Normalise une référence d'article pour obtenir uniquement le numéro""" if not article_ref or not isinstance(article_ref, str): return "" cleaned = re.sub(r'[^\d\s]', '', article_ref.strip()) cleaned = re.sub(r'\s+', ' ', cleaned) match = re.search(r'(\d+)(?:\s*(?:bis|ter|quater))?', cleaned, re.IGNORECASE) if match: return match.group(0).strip().lower() return cleaned @staticmethod def extract_all_numbers(text: str) -> Set[str]: """Extrait uniquement les numéros d'articles valides avec mots-clés spécifiques""" if not text: return set() patterns = [ r'\b(?:Article|article|art\.|Art\.)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', r'\b(?:الفصل|فصل|المادة|مادة|المادّة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', r'\b(?:n°|N°|numéro|رقم)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b', r'[\(\[]\s*(?:article|Article|art\.|الفصل|المادة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\s*[\)\]]' ] numbers = set() for pattern in patterns: matches = re.findall(pattern, text, re.IGNORECASE | re.UNICODE) for match in matches: if isinstance(match, tuple): num = match[0] else: num = match if num.isdigit(): num_int = int(num) if (1900 <= num_int <= 2100) or num_int > 9999: continue normalized = ArticleNumberNormalizer.normalize(num) if normalized and re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE): numbers.add(normalized) return numbers @staticmethod def is_valid_article_number(art_num: str) -> bool: """Vérifie si un numéro d'article est valide""" if not art_num or not isinstance(art_num, str): return False normalized = ArticleNumberNormalizer.normalize(art_num) if not normalized: return False if not re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE): return False match = re.search(r'(\d+)', normalized) if not match: return False num = int(match.group(1)) if 1900 <= num <= 2100: return False if num > 9999: return False return True # ============================================================ # LEXICAL SEARCH ENGINE # ============================================================ class LexicalSearchEngine: """Moteur de recherche lexicale avancée pour la jurisprudence""" @staticmethod def create_lexical_queries(query_analysis: QueryAnalysis) -> List[Dict]: """Crée des requêtes lexicales pour la recherche dans la jurisprudence""" queries = [] keywords_fr = [] keywords_ar = [] for topic in query_analysis.extracted_topics: if topic and len(topic) > 2: keywords_fr.append(topic.lower()) if not any(char in topic for char in 'اأإآبتثجحخدذرزسشصضطظعغفقكلمنهوي'): try: translated = TranslationService.translate_fr_to_ar(topic) keywords_ar.append(translated) except: pass query_terms = re.findall(r'\b\w+\b', query_analysis.translated_query.lower()) keywords_fr.extend([term for term in query_terms if len(term) > 3]) if query_analysis.original_language == "ar": arabic_terms = re.findall(r'[\u0600-\u06FF]+', query_analysis.original_query) keywords_ar.extend([term for term in arabic_terms if len(term) > 2]) keywords_fr = list(set([k for k in keywords_fr if len(k) > 2]))[:30] keywords_ar = list(set([k for k in keywords_ar if len(k) > 2]))[:30] if keywords_fr: queries.append({ "language": "fr", "keywords": keywords_fr, "search_fields": ["resume_fr", "faits_fr", "decision_fr", "tags_fr", "text_to_vector_fr.principe"], "boost_fields": { "resume_fr": 2.0, "faits_fr": 1.5, "text_to_vector_fr.principe": 2.5, "tags_fr": 3.0 } }) if keywords_ar: queries.append({ "language": "ar", "keywords": keywords_ar, "search_fields": ["resume_ar", "faits_ar", "decision_ar", "tags_ar", "text_to_vector_ar.principe"], "boost_fields": { "resume_ar": 2.0, "faits_ar": 1.5, "text_to_vector_ar.principe": 2.5, "tags_ar": 3.0 } }) all_codes = [query_analysis.primary_legal_code] + query_analysis.secondary_codes for code in all_codes: code_keywords = Config.CODE_NAMES.get(code.value, {}).get("keywords", []) if code_keywords: fr_keywords = [kw for kw in code_keywords if kw.isascii()] ar_keywords = [kw for kw in code_keywords if not kw.isascii()] if fr_keywords: queries.append({ "language": "fr", "keywords": fr_keywords[:15], "search_fields": ["tags_fr", "resume_fr", "code_fr"], "boost_fields": {"tags_fr": 3.0, "code_fr": 2.0}, "code_filter": code.value }) if ar_keywords: queries.append({ "language": "ar", "keywords": ar_keywords[:15], "search_fields": ["tags_ar", "resume_ar", "code_ar"], "boost_fields": {"tags_ar": 3.0, "code_ar": 2.0}, "code_filter": code.value }) if query_analysis.cited_articles: for article in query_analysis.cited_articles: if ArticleNumberNormalizer.is_valid_article_number(article): queries.append({ "language": "both", "keywords": [f"article {article}", f"الفصل {article}"], "search_fields": ["articles_cites.article_num", "text_to_vector_ar.principe", "text_to_vector_fr.principe"], "boost_fields": {"articles_cites.article_num": 5.0} }) return queries @staticmethod def search_lexical(collection, queries: List[Dict], limit: int = 100) -> List[Dict]: """Exécute une recherche lexicale dans la collection""" all_results = [] for query_config in queries: try: mongo_query = LexicalSearchEngine._build_mongo_query(query_config) results = list(collection.find(mongo_query).limit(limit)) for doc in results: lexical_score = LexicalSearchEngine._calculate_lexical_score(doc, query_config) doc["_lexical_score"] = lexical_score doc["_lexical_query"] = query_config doc["_search_method"] = "lexical_search" found = False for existing in all_results: if existing.get("_id") == doc.get("_id"): found = True if lexical_score > existing.get("_lexical_score", 0): existing.update(doc) break if not found: all_results.append(doc) except Exception as e: logger.error(f"Erreur recherche lexicale: {e}") continue all_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True) return all_results[:limit] @staticmethod def _build_mongo_query(query_config: Dict) -> Dict: """Construit une requête MongoDB pour la recherche lexicale""" language = query_config.get("language", "fr") keywords = query_config.get("keywords", []) search_fields = query_config.get("search_fields", []) code_filter = query_config.get("code_filter") if not keywords or not search_fields: return {} or_conditions = [] for keyword in keywords: if not keyword or len(keyword) < 2: continue escaped_keyword = re.escape(keyword) for field in search_fields: field_parts = field.split('.') if len(field_parts) > 1: nested_field = field_parts[0] nested_subfield = field_parts[1] condition = { f"{nested_field}.{nested_subfield}": { "$regex": escaped_keyword, "$options": "i" } } else: condition = { field: { "$regex": escaped_keyword, "$options": "i" } } or_conditions.append(condition) if not or_conditions: return {} mongo_query = {"$or": or_conditions} if code_filter: code_names = Config.CODE_NAMES.get(code_filter, {}) if code_names: code_fr = code_names.get("fr", "") code_ar = code_names.get("ar", "") code_conditions = [] if code_fr: code_conditions.append({"code_fr": {"$regex": code_fr, "$options": "i"}}) if code_ar: code_conditions.append({"code_ar": {"$regex": code_ar, "$options": "i"}}) if code_conditions: mongo_query["$and"] = [{"$or": code_conditions}] return mongo_query @staticmethod def _calculate_lexical_score(doc: Dict, query_config: Dict) -> float: """Calcule un score lexical basé sur la pertinence""" keywords = query_config.get("keywords", []) search_fields = query_config.get("search_fields", []) boost_fields = query_config.get("boost_fields", {}) if not keywords: return 0.0 total_score = 0.0 keyword_count = 0 for keyword in keywords: keyword_lower = keyword.lower() keyword_score = 0.0 for field in search_fields: field_value = LexicalSearchEngine._get_field_value(doc, field) if not field_value: continue if keyword_lower in field_value.lower(): base_score = 1.0 boost = boost_fields.get(field, 1.0) if re.search(rf'\b{re.escape(keyword_lower)}\b', field_value.lower()): base_score *= 1.5 keyword_score = max(keyword_score, base_score * boost) if keyword_score > 0: total_score += keyword_score keyword_count += 1 if keyword_count == 0: return 0.0 average_score = total_score / keyword_count coverage_bonus = keyword_count / len(keywords) * 0.5 tags_fr = doc.get("tags_fr", []) tags_ar = doc.get("tags_ar", []) all_tags = tags_fr + tags_ar tag_bonus = 0.0 for tag in all_tags: for keyword in keywords: if keyword.lower() in tag.lower(): tag_bonus += 0.2 final_score = min(1.0, average_score + coverage_bonus + tag_bonus) return final_score @staticmethod def _get_field_value(doc: Dict, field_path: str) -> str: """Obtient la valeur d'un champ, gère les champs imbriqués""" if not field_path: return "" parts = field_path.split('.') current = doc for part in parts: if isinstance(current, dict): current = current.get(part, {}) else: return "" if isinstance(current, str): return current elif isinstance(current, list): return " ".join([str(item) for item in current]) elif current: return str(current) return "" # ============================================================ # ADVANCED SEARCH ENGINES (BM25, TF-IDF, EXACT MATCH) # ============================================================ class BM25SearchEngine: """Moteur de recherche BM25 avancé""" def __init__(self): self.k1 = 1.5 self.b = 0.75 self.avgdl = 0 self.doc_freqs = {} self.idf = {} self.doc_lengths = [] self.corpus_size = 0 self.corpus = [] self.doc_ids = [] self.fields_weights = { "resume_ar": 2.0, "resume_fr": 2.0, "faits_ar": 1.5, "faits_fr": 1.5, "decision_ar": 1.8, "decision_fr": 1.8, "tags_ar": 3.0, "tags_fr": 3.0, "text_to_vector_ar.principe": 2.5, "text_to_vector_fr.principe": 2.5, "code_ar": 1.2, "code_fr": 1.2 } def preprocess_text(self, text: str, language: str = "ar") -> List[str]: """Prétraitement avancé du texte""" if not text: return [] text = text.lower() text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s]', ' ', text) text = re.sub(r'\s+', ' ', text).strip() if language == "ar": tokens = re.findall(r'[\u0600-\u06FF]+', text) arabic_stopwords = { 'في', 'من', 'إلى', 'على', 'أن', 'إن', 'ما', 'هو', 'هي', 'كان', 'يكون', 'كانت', 'ليس', 'لا', 'ولكن', 'أو', 'و', 'لكن', 'إذا', 'ذلك', 'هذا', 'هذه', 'تلك', 'التي', 'الذي', 'الذين', 'قد', 'حيث', 'عن', 'مع', 'بين', 'فيما', 'كل', 'بعض', 'أي', 'كل', 'مادة', 'فصل', 'المادة', 'الفصل', 'قانون', 'القانون', 'مجلة', 'المجلة' } tokens = [token for token in tokens if token not in arabic_stopwords and len(token) > 2] else: tokens = re.findall(r'\b\w+\b', text) french_stopwords = { 'le', 'la', 'les', 'de', 'des', 'du', 'et', 'est', 'une', 'un', 'dans', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'article', 'articles', 'code', 'loi', 'droit', 'juridique', 'tribunal', 'cour', 'jugement', 'décision', 'affaire', 'procédure' } tokens = [token for token in tokens if token not in french_stopwords and len(token) > 2] return tokens def build_index(self, documents: List[Dict], language: str = "ar"): """Construit l'index BM25""" self.corpus = [] self.doc_ids = [] self.doc_lengths = [] logger.info(f"🔨 Construction index BM25 pour {len(documents)} documents...") for doc in documents: doc_id = str(doc.get("_id", "")) if not doc_id: continue full_text_tokens = [] for field, weight in self.fields_weights.items(): field_parts = field.split('.') field_value = doc for part in field_parts: if isinstance(field_value, dict): field_value = field_value.get(part, "") else: field_value = "" break if field_value: if isinstance(field_value, list): field_value = " ".join(field_value) tokens = self.preprocess_text(str(field_value), language) for _ in range(int(weight)): full_text_tokens.extend(tokens) if full_text_tokens: self.corpus.append(full_text_tokens) self.doc_ids.append(doc_id) self.doc_lengths.append(len(full_text_tokens)) if not self.corpus: logger.warning("⚠️ Aucun document indexé pour BM25") return self.corpus_size = len(self.corpus) self.avgdl = sum(self.doc_lengths) / self.corpus_size self.doc_freqs = {} for doc_tokens in self.corpus: unique_tokens = set(doc_tokens) for token in unique_tokens: self.doc_freqs[token] = self.doc_freqs.get(token, 0) + 1 self.idf = {} for token, freq in self.doc_freqs.items(): self.idf[token] = math.log((self.corpus_size - freq + 0.5) / (freq + 0.5) + 1) logger.info(f"✅ Index BM25 construit: {self.corpus_size} documents, {len(self.idf)} tokens uniques") def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]: """Recherche BM25 avec la requête""" if not self.corpus: return [] query_tokens = self.preprocess_text(query, language) if not query_tokens: return [] scores = np.zeros(self.corpus_size) for i, doc_tokens in enumerate(self.corpus): doc_len = self.doc_lengths[i] for token in query_tokens: if token in self.idf: f = doc_tokens.count(token) idf_score = self.idf[token] numerator = f * (self.k1 + 1) denominator = f + self.k1 * (1 - self.b + self.b * doc_len / self.avgdl) scores[i] += idf_score * numerator / denominator if denominator != 0 else 0 sorted_indices = np.argsort(scores)[::-1][:top_k] results = [] for idx in sorted_indices: if scores[idx] > 0: results.append((self.doc_ids[idx], float(scores[idx]))) logger.info(f"🔍 Recherche BM25: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})") return results class TFIDFSearchEngine: """Moteur de recherche TF-IDF""" def __init__(self): self.vocabulary = {} self.idf = {} self.tfidf_matrix = None self.doc_ids = [] def build_index(self, documents: List[Dict], language: str = "ar"): """Construit l'index TF-IDF""" self.doc_ids = [] all_docs_tokens = [] logger.info(f"🔨 Construction index TF-IDF pour {len(documents)} documents...") for doc in documents: doc_id = str(doc.get("_id", "")) if not doc_id: continue text_parts = [] priority_fields = [ "resume_ar", "resume_fr", "text_to_vector_ar.principe", "text_to_vector_fr.principe", "tags_ar", "tags_fr", "faits_ar", "faits_fr" ] for field in priority_fields: field_parts = field.split('.') field_value = doc for part in field_parts: if isinstance(field_value, dict): field_value = field_value.get(part, "") else: field_value = "" break if field_value: if isinstance(field_value, list): field_value = " ".join(field_value) text_parts.append(str(field_value)) full_text = " ".join(text_parts) if language == "ar": tokens = re.findall(r'[\u0600-\u06FF]{3,}', full_text.lower()) else: tokens = re.findall(r'\b\w{3,}\b', full_text.lower()) if tokens: all_docs_tokens.append(tokens) self.doc_ids.append(doc_id) if not all_docs_tokens: return all_tokens = set() for doc_tokens in all_docs_tokens: all_tokens.update(doc_tokens) self.vocabulary = {token: idx for idx, token in enumerate(sorted(all_tokens))} tf_matrix = np.zeros((len(all_docs_tokens), len(self.vocabulary))) for i, doc_tokens in enumerate(all_docs_tokens): token_counts = Counter(doc_tokens) total_tokens = len(doc_tokens) for token, count in token_counts.items(): if token in self.vocabulary: idx = self.vocabulary[token] tf_matrix[i, idx] = count / total_tokens if total_tokens > 0 else 0 doc_count = len(all_docs_tokens) df = np.sum(tf_matrix > 0, axis=0) self.idf = np.log((doc_count + 1) / (df + 1)) + 1 self.tfidf_matrix = tf_matrix * self.idf logger.info(f"✅ Index TF-IDF construit: {len(self.doc_ids)} documents, {len(self.vocabulary)} tokens") def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]: """Recherche TF-IDF""" if self.tfidf_matrix is None or not self.doc_ids: return [] if language == "ar": query_tokens = re.findall(r'[\u0600-\u06FF]{3,}', query.lower()) else: query_tokens = re.findall(r'\b\w{3,}\b', query.lower()) if not query_tokens: return [] query_vector = np.zeros(len(self.vocabulary)) query_counts = Counter(query_tokens) total_tokens = len(query_tokens) for token, count in query_counts.items(): if token in self.vocabulary: idx = self.vocabulary[token] query_vector[idx] = count / total_tokens if total_tokens > 0 else 0 query_vector = query_vector * self.idf norm_docs = np.linalg.norm(self.tfidf_matrix, axis=1, keepdims=True) norm_query = np.linalg.norm(query_vector) similarities = np.dot(self.tfidf_matrix, query_vector) / (norm_docs.flatten() * norm_query + 1e-8) sorted_indices = np.argsort(similarities)[::-1][:top_k] results = [] for idx in sorted_indices: if similarities[idx] > 0: results.append((self.doc_ids[idx], float(similarities[idx]))) logger.info(f"🔍 Recherche TF-IDF: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})") return results class ExactMatchSearchEngine: """Moteur de recherche par correspondance exacte""" def __init__(self): self.inverted_index = {} self.doc_metadata = {} def build_index(self, documents: List[Dict], language: str = "ar"): """Construit un index inversé pour la recherche exacte""" self.inverted_index = {} self.doc_metadata = {} logger.info(f"🔨 Construction index exact pour {len(documents)} documents...") legal_keywords = { "ar": [ "كراء تجاري", "تجديد الكراء", "مؤسسات التعليم الخاص", "تعليم خاص", "الأكرية التجارية", "تأجير", "عقد كراء", "مدة الكراء", "حق التجديد", "المحلات التجارية", "المادة 532", "الفصل 532", "مجلة الالتزامات والعقود", "محكمة التعقيب", "المحكمة التجارية", "عقود التسويغ" ], "fr": [ "bail commercial", "renouvellement bail", "institutions enseignement privé", "enseignement privé", "baux commerciaux", "location", "contrat bail", "durée bail", "droit renouvellement", "locaux commerciaux", "article 532", "code obligations contrats", "cour cassation", "tribunal commerce", "contrats location" ] } keywords = legal_keywords.get(language, []) for doc in documents: doc_id = str(doc.get("_id", "")) if not doc_id: continue self.doc_metadata[doc_id] = doc text_fields = [] if language == "ar": fields_to_check = ["resume_ar", "faits_ar", "decision_ar", "text_to_vector_ar.principe"] else: fields_to_check = ["resume_fr", "faits_fr", "decision_fr", "text_to_vector_fr.principe"] for field in fields_to_check: field_parts = field.split('.') field_value = doc for part in field_parts: if isinstance(field_value, dict): field_value = field_value.get(part, "") else: field_value = "" break if field_value: if isinstance(field_value, list): field_value = " ".join(field_value) text_fields.append(str(field_value).lower()) full_text = " ".join(text_fields) for keyword in keywords: if keyword.lower() in full_text: if keyword not in self.inverted_index: self.inverted_index[keyword] = [] occurrences = full_text.count(keyword.lower()) score = occurrences * 2.0 if field_parts[0] in ["resume", "text_to_vector"]: score += 1.5 self.inverted_index[keyword].append((doc_id, score)) logger.info(f"✅ Index exact construit: {len(self.inverted_index)} mots-clés indexés") def search(self, query: str, language: str = "ar", top_k: int = 50) -> List[Tuple[str, float]]: """Recherche par correspondance exacte""" if not self.inverted_index: return [] query_lower = query.lower() found_keywords = [] for keyword in self.inverted_index.keys(): if keyword.lower() in query_lower: found_keywords.append(keyword) if not found_keywords: return [] doc_scores = {} for keyword in found_keywords: for doc_id, score in self.inverted_index.get(keyword, []): if doc_id not in doc_scores: doc_scores[doc_id] = 0 doc_scores[doc_id] += score sorted_docs = sorted(doc_scores.items(), key=lambda x: x[1], reverse=True)[:top_k] results = [(doc_id, score) for doc_id, score in sorted_docs if score > 0] logger.info(f"🔍 Recherche exacte: {len(results)} résultats (mots-clés trouvés: {found_keywords})") return results class HybridJurisprudenceSearch: """Recherche hybride de jurisprudence avec multiples méthodes""" def __init__(self, db_manager): self.db_manager = db_manager self.bm25_searcher = BM25SearchEngine() self.tfidf_searcher = TFIDFSearchEngine() self.exact_searcher = ExactMatchSearchEngine() self.cached_docs = {} def load_jurisprudence_docs(self, language: str = "ar", limit: int = 2000) -> List[Dict]: """Charge les documents de jurisprudence depuis MongoDB""" cache_key = f"juris_docs_{language}" if cache_key in self.cached_docs: cached_time, docs = self.cached_docs[cache_key] if datetime.now().timestamp() - cached_time < 300: logger.info(f"📦 Utilisation du cache pour {len(docs)} documents") return docs logger.info(f"📥 Chargement des documents de jurisprudence ({language})...") query = {} if language == "ar": query = {"resume_ar": {"$exists": True, "$ne": ""}} else: query = {"resume_fr": {"$exists": True, "$ne": ""}} docs = list(self.db_manager.juris_collection.find(query).limit(limit)) self.cached_docs[cache_key] = (datetime.now().timestamp(), docs) logger.info(f"✅ {len(docs)} documents chargés") return docs def search_hybrid(self, query: str, query_analysis: QueryAnalysis, limit: int = 150) -> List[Dict]: """Recherche hybride avec multiples méthodes""" language = query_analysis.search_language docs = self.load_jurisprudence_docs(language, limit=2000) if not docs: logger.warning("⚠️ Aucun document de jurisprudence disponible") return [] logger.info("🏗️ Construction des index de recherche...") self.bm25_searcher.build_index(docs, language) self.tfidf_searcher.build_index(docs, language) self.exact_searcher.build_index(docs, language) logger.info("🔍 Exécution des recherches hybrides...") bm25_results = self.bm25_searcher.search(query, language, top_k=limit) tfidf_results = self.tfidf_searcher.search(query, language, top_k=limit) exact_results = self.exact_searcher.search(query, language, top_k=limit) article_results = self._search_by_articles(query_analysis.cited_articles, docs) all_results = self._merge_results( bm25_results, tfidf_results, exact_results, article_results, docs, limit ) logger.info(f"✅ Recherche hybride: {len(all_results)} résultats combinés") return all_results def _search_by_articles(self, articles: List[str], docs: List[Dict]) -> List[Tuple[str, float]]: """Recherche basée sur les articles cités""" if not articles: return [] results = [] article_set = set(articles) for doc in docs: doc_id = str(doc.get("_id", "")) cited_articles = doc.get("articles_cites", []) score = 0 for cited in cited_articles: article_num = cited.get("article_num", "") if article_num in article_set: score += 3.0 if score > 0: results.append((doc_id, score)) return results def _merge_results(self, bm25_results: List, tfidf_results: List, exact_results: List, article_results: List, docs: List[Dict], limit: int) -> List[Dict]: """Fusionne les résultats de toutes les méthodes""" doc_map = {str(doc.get("_id", "")): doc for doc in docs} combined_scores = {} method_weights = { "bm25": 0.3, "tfidf": 0.25, "exact": 0.3, "articles": 0.15 } for doc_id, score in bm25_results: if doc_id not in combined_scores: combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} combined_scores[doc_id]["bm25"] = score for doc_id, score in tfidf_results: if doc_id not in combined_scores: combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} combined_scores[doc_id]["tfidf"] = score for doc_id, score in exact_results: if doc_id not in combined_scores: combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} combined_scores[doc_id]["exact"] = score for doc_id, score in article_results: if doc_id not in combined_scores: combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0} combined_scores[doc_id]["articles"] = score final_results = [] for doc_id, scores in combined_scores.items(): if doc_id in doc_map: final_score = ( scores["bm25"] * method_weights["bm25"] + scores["tfidf"] * method_weights["tfidf"] + scores["exact"] * method_weights["exact"] + scores["articles"] * method_weights["articles"] ) if final_score > 0: doc = doc_map[doc_id].copy() doc["_hybrid_score"] = final_score doc["_score_details"] = scores final_results.append(doc) final_results.sort(key=lambda x: x.get("_hybrid_score", 0), reverse=True) return final_results[:limit] # ============================================================ # LEGAL CODE CLASSIFIER PROFESSIONNEL # ============================================================ class LegalCodeClassifier: @staticmethod def classify_query(query: str, language: str) -> Dict[str, Any]: """Classification avancée des codes juridiques tunisiens""" classification_prompt = f"""Analyze this Tunisian legal query to identify relevant codes from ALL 10 Tunisian codes. TUNISIAN LEGAL CODES AVAILABLE: 1. CSP (Code du Statut Personnel) - Family law, marriage, divorce, inheritance, personal status 2. DROITS_REELS (Code des Droits Réels) - Property law, real estate, ownership, mortgages, real rights 3. OBLIGATIONS_CONTRATS (Code des Obligations et des Contrats) - Contract law, obligations, torts, civil liability 4. PROCEDURE_CIVILE (Code de Procédure Civile et Commerciale) - Civil procedure, appeals, judicial process 5. CODE_TRAVAIL (Code du Travail) - Labor law, employment contracts, termination, labor disputes 6. DROITS_PROCEDURES_FISCAUX (Code des Droits et Procédures Fiscaux) - Tax law, fiscal procedures, tax disputes 7. DROIT_INTERNATIONAL_PRIVE (Code de Droit International Privé) - Private international law, conflicts of law 8. CODE_PENAL (Code Pénal) - Criminal law, offenses, penalties, criminal procedure 9. CODE_COMMERCE (Code de Commerce) - Commercial law, companies, bankruptcy, commercial contracts 10. PROCEDURES_PENALES (Code des Procédures Pénales) - Criminal procedure, investigation, prosecution QUERY: {query} LANGUAGE: {language} ANALYSIS INSTRUCTIONS: 1. Identify the PRIMARY code (most relevant) 2. Identify SECONDARY codes (relevant but less direct) 3. Provide confidence score (0.0-1.0) 4. Extract key legal concepts 5. DO NOT extract article numbers here OUTPUT ONLY VALID JSON: {{ "primary_code": "CODE_TRAVAIL", "secondary_codes": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"], "confidence": 0.92, "key_concepts": ["labor contract", "termination", "CDD renewal"], "reasoning": "Query focuses on employment contract renewal and termination issues", "priority_factors": ["employment", "contract", "termination", "labor rights"] }}""" try: response = chat_client.chat.completions.create( model=Config.CHAT_MODEL, messages=[ {"role": "system", "content": "Expert Tunisian legal code classifier. Analyze queries and identify relevant codes accurately. Output ONLY valid JSON."}, {"role": "user", "content": classification_prompt} ], temperature=Config.TEMP_CLASSIFICATION, max_tokens=800 ) result = response.choices[0].message.content.strip() result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE) classification = json.loads(result) primary_code = LegalCodeClassifier._validate_and_normalize_code(classification["primary_code"]) secondary_codes = [LegalCodeClassifier._validate_and_normalize_code(code) for code in classification.get("secondary_codes", [])] logger.info(f"✅ Classification réussie: {primary_code.value} (confiance: {classification['confidence']:.2f})") return { "primary_code": primary_code, "secondary_codes": secondary_codes, "confidence": float(classification.get("confidence", 0.5)), "key_concepts": classification.get("key_concepts", []), "reasoning": classification.get("reasoning", ""), "priority_factors": classification.get("priority_factors", []) } except Exception as e: logger.error(f"Erreur classification GPT: {e}") return LegalCodeClassifier.fallback_classification(query, language) @staticmethod def _validate_and_normalize_code(code_str: str) -> LegalCode: """Valide et normalise un code""" try: normalized = code_str.upper().strip() variations = { "TRAVAIL": "CODE_TRAVAIL", "CODE TRAVAIL": "CODE_TRAVAIL", "CODETRAVAIL": "CODE_TRAVAIL", "DROIT_REEL": "DROITS_REELS", "DROIT REEL": "DROITS_REELS", "DROITS_REEL": "DROITS_REELS", "PENAL": "CODE_PENAL", "CODEPENAL": "CODE_PENAL", "COMMERCE": "CODE_COMMERCE", "CODECOMMERCE": "CODE_COMMERCE", "PROCEDURE_PENALE": "PROCEDURES_PENALES", "PROCEDURE PENALE": "PROCEDURES_PENALES" } if normalized in variations: normalized = variations[normalized] return LegalCode.from_string(normalized) except: logger.warning(f"Code non reconnu: {code_str}, utilisation CSP par défaut") return LegalCode.CSP @staticmethod def fallback_classification(query: str, language: str) -> Dict[str, Any]: """Classification par mots-clés comme fallback""" query_lower = query.lower() scores = {code: 0 for code in LegalCode} for code in LegalCode: if code.value in Config.CODE_NAMES: keywords = Config.CODE_NAMES[code.value].get("keywords", []) for keyword in keywords: if keyword.lower() in query_lower: scores[code] += 2 special_rules = [ (["contrat", "travail", "licenciement"], LegalCode.CODE_TRAVAIL, 3), (["contrat", "travail", "salaire"], LegalCode.CODE_TRAVAIL, 2), (["propriété", "succession", "héritage"], LegalCode.CSP, 2), (["propriété", "hypothèque", "immobilier"], LegalCode.DROITS_REELS, 2), (["contrat", "obligation", "responsabilité"], LegalCode.OBLIGATIONS_CONTRATS, 2), (["procédure", "appel", "jugement"], LegalCode.PROCEDURE_CIVILE, 2), (["fiscal", "impôt", "taxe"], LegalCode.DROITS_PROCEDURES_FISCAUX, 2), (["international", "étranger", "nationalité"], LegalCode.DROIT_INTERNATIONAL_PRIVE, 2), (["crime", "peine", "prison"], LegalCode.CODE_PENAL, 2), (["commerce", "société", "faillite"], LegalCode.CODE_COMMERCE, 2), (["enquête", "instruction", "procédure pénale"], LegalCode.PROCEDURES_PENALES, 2) ] for keywords, code, bonus in special_rules: if all(keyword in query_lower for keyword in keywords): scores[code] += bonus sorted_codes = sorted(scores.items(), key=lambda x: x[1], reverse=True) primary_code = sorted_codes[0][0] if sorted_codes else LegalCode.CSP secondary_codes = [] for code, score in sorted_codes[1:]: if score > 0 and len(secondary_codes) < 3: secondary_codes.append(code) total_score = sum(scores.values()) confidence = min(scores[primary_code] / max(total_score, 1) * 1.2, 0.85) logger.info(f"⚠️ Classification fallback: {primary_code.value} (confiance: {confidence:.2f})") return { "primary_code": primary_code, "secondary_codes": secondary_codes, "confidence": confidence, "key_concepts": [], "reasoning": "Fallback keyword-based classification", "priority_factors": [] } # ============================================================ # EMBEDDING SERVICE PROFESSIONNEL # ============================================================ class EnterpriseEmbeddingService: _cache = {} _failed_embeddings = set() @staticmethod def get_embedding(text: str) -> Optional[List[float]]: cache_key = hashlib.md5(text.encode()).hexdigest() if cache_key in EnterpriseEmbeddingService._cache: return EnterpriseEmbeddingService._cache[cache_key] if cache_key in EnterpriseEmbeddingService._failed_embeddings: return None try: cleaned_text = EnterpriseEmbeddingService.preprocess_text(text) if not cleaned_text or len(cleaned_text) < 3: return None response = embedding_client.embeddings.create( input=cleaned_text, model=Config.EMBEDDING_MODEL ) embedding = response.data[0].embedding EnterpriseEmbeddingService._cache[cache_key] = embedding return embedding except Exception as e: logger.error(f"Erreur génération embedding: {e}") EnterpriseEmbeddingService._failed_embeddings.add(cache_key) return None @staticmethod def preprocess_text(text: str) -> str: """Prétraitement avancé du texte""" if not text: return "" text = re.sub(r'\s+', ' ', text) text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s.,;:!?()\[\]-]', ' ', text) text = text.strip() if len(text) > 8000: text = text[:8000] return text @staticmethod def calculate_similarity(emb1: Optional[List[float]], emb2: Optional[List[float]]) -> float: if not emb1 or not emb2: return 0.0 try: arr1 = np.array(emb1).reshape(1, -1) arr2 = np.array(emb2).reshape(1, -1) similarity = cosine_similarity(arr1, arr2)[0][0] return max(0.0, min(1.0, similarity)) except Exception as e: logger.error(f"Erreur calcul similarité: {e}") return 0.0 # ============================================================ # QUERY ANALYZER PROFESSIONNEL # ============================================================ class EnterpriseQueryAnalyzer: @staticmethod def analyze_query(query: str) -> QueryAnalysis: """Analyse complète de la requête""" translation_result = TranslationService.detect_and_translate(query) original_query = translation_result["original_query"] translated_query = translation_result["translated_query"] original_language = translation_result["original_language"] search_language = translation_result["search_language"] logger.info(f"🌐 Langue détectée: {original_language}") if original_language == "ar": logger.info(f"📝 Requête traduite pour recherche") classification = LegalCodeClassifier.classify_query(translated_query, search_language) cited_articles_original = ArticleNumberNormalizer.extract_all_numbers(original_query) cited_articles_translated = ArticleNumberNormalizer.extract_all_numbers(translated_query) all_articles = cited_articles_original.union(cited_articles_translated) analysis_prompt = f"""Analyze this Tunisian legal query in detail. PRIMARY CODE IDENTIFIED: {classification['primary_code'].value} QUERY: {translated_query} LANGUAGE: {search_language} CITED ARTICLES DETECTED: {list(all_articles)} ANALYSIS TASKS: 1. Validate and filter article numbers (keep only valid Tunisian law article references) 2. Extract key legal topics 3. Identify relevant legal entities 4. Determine question type and complexity 5. Generate search queries for semantic search 6. Identify cross-code implications RULES FOR ARTICLE EXTRACTION: - Keep only valid article numbers (e.g., "123", "45 bis") - Remove years, dates, page numbers, etc. - If query mentions articles by topic without numbers, don't list them OUTPUT ONLY VALID JSON: {{ "validated_articles": [], "topics": ["topic1", "topic2"], "legal_entities": ["entity1", "entity2"], "question_type": "substantive/procedural/hybrid", "complexity_score": 0.85, "requires_statutory_law": true, "requires_procedural_law": false, "requires_commercial_law": false, "requires_criminal_law": false, "requires_labor_law": true, "requires_tax_law": false, "requires_international_law": false, "requires_family_law": false, "requires_property_law": false, "search_queries": ["query1", "query2"], "cross_code_references": [] }}""" try: response = chat_client.chat.completions.create( model=Config.CHAT_MODEL, messages=[ {"role": "system", "content": "Advanced Tunisian legal query analyzer. Focus on accuracy and precision. Output ONLY valid JSON."}, {"role": "user", "content": analysis_prompt} ], temperature=Config.TEMP_ANALYSIS, max_tokens=800 ) result = response.choices[0].message.content.strip() result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE) analysis_data = json.loads(result) validated_articles = [] for art in analysis_data.get('validated_articles', []): if ArticleNumberNormalizer.is_valid_article_number(str(art)): validated_articles.append(str(art)) else: logger.warning(f"Article ignoré (invalide): {art}") final_articles = list(set(validated_articles + list(all_articles))) primary_code = classification['primary_code'] secondary_codes = classification['secondary_codes'].copy() code_mappings = { 'requires_procedural_law': LegalCode.PROCEDURE_CIVILE, 'requires_commercial_law': LegalCode.CODE_COMMERCE, 'requires_criminal_law': LegalCode.CODE_PENAL, 'requires_labor_law': LegalCode.CODE_TRAVAIL, 'requires_tax_law': LegalCode.DROITS_PROCEDURES_FISCAUX, 'requires_international_law': LegalCode.DROIT_INTERNATIONAL_PRIVE, 'requires_family_law': LegalCode.CSP, 'requires_property_law': LegalCode.DROITS_REELS } for key, code in code_mappings.items(): if analysis_data.get(key, False) and code not in secondary_codes and code != primary_code: secondary_codes.append(code) if not secondary_codes and primary_code.value in Config.SECONDARY_CODE_MAPPING: default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value] secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]] secondary_codes = list(dict.fromkeys(secondary_codes))[:5] logger.info(f"📊 Classification: {primary_code.value}") logger.info(f"📋 Codes secondaires: {[c.value for c in secondary_codes]}") logger.info(f"📄 Articles cités: {final_articles}") return QueryAnalysis( original_query=original_query, translated_query=translated_query, original_language=original_language, search_language=search_language, extracted_topics=analysis_data.get('topics', []), legal_entities=analysis_data.get('legal_entities', []), cited_articles=final_articles, primary_legal_code=primary_code, secondary_codes=secondary_codes, question_type=analysis_data.get('question_type', 'substantive'), complexity_score=float(analysis_data.get('complexity_score', 0.5)), search_queries=analysis_data.get('search_queries', [translated_query]), requires_statutory_law=analysis_data.get('requires_statutory_law', True), code_confidence=classification['confidence'] ) except Exception as e: logger.error(f"Erreur analyse requête: {e}") return EnterpriseQueryAnalyzer._create_fallback_analysis( original_query, translated_query, original_language, search_language, classification, all_articles ) @staticmethod def _create_fallback_analysis(original_query, translated_query, original_language, search_language, classification, all_articles): """Créer une analyse de fallback""" primary_code = classification['primary_code'] if primary_code.value in Config.SECONDARY_CODE_MAPPING: default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value] secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]] else: secondary_codes = [] if primary_code != LegalCode.PROCEDURE_CIVILE and LegalCode.PROCEDURE_CIVILE not in secondary_codes: secondary_codes.append(LegalCode.PROCEDURE_CIVILE) return QueryAnalysis( original_query=original_query, translated_query=translated_query, original_language=original_language, search_language=search_language, extracted_topics=[], legal_entities=[], cited_articles=list(all_articles), primary_legal_code=primary_code, secondary_codes=secondary_codes, question_type="substantive", complexity_score=0.5, search_queries=[translated_query], requires_statutory_law=True, code_confidence=classification['confidence'] ) # ============================================================ # DATABASE MANAGER PROFESSIONNEL # ============================================================ class EnterpriseDatabaseManager: def __init__(self): self.code_db = mongo_client[Config.DB_CODE] self.juris_collection = mongo_client[Config.DB_JURIS][Config.COL_JURIS] self._article_cache = {} self._collection_cache = {} self._verify_collections() def _verify_collections(self): """Vérifier que toutes les collections existent""" logger.info("🔍 Vérification des collections...") available_collections = self.code_db.list_collection_names() for code_key, collection_name in Config.COLLECTIONS.items(): if collection_name in available_collections: logger.info(f" ✓ {code_key}: {collection_name}") else: logger.warning(f" ✗ {code_key}: {collection_name} - COLLECTION NON TROUVÉE") def get_collection(self, code_type: LegalCode): """Récupère la collection MongoDB pour un type de code""" if code_type in self._collection_cache: return self._collection_cache[code_type] collection_name = Config.COLLECTIONS.get(code_type.value) if not collection_name: logger.error(f"❌ Collection non configurée pour: {code_type.value}") return None try: collection = self.code_db[collection_name] self._collection_cache[code_type] = collection return collection except Exception as e: logger.error(f"❌ Erreur accès collection {collection_name}: {e}") return None def get_article_by_number(self, article_number: str, code_type: LegalCode) -> Optional[Dict]: """Récupère un article par son numéro""" if not ArticleNumberNormalizer.is_valid_article_number(article_number): logger.warning(f"Numéro d'article invalide: {article_number}") return None cache_key = f"{code_type.value}_{article_number}" if cache_key in self._article_cache: return self._article_cache[cache_key] collection = self.get_collection(code_type) if collection is None: return None normalized = ArticleNumberNormalizer.normalize(article_number) search_strategies = [ lambda: collection.find_one({"article_num": normalized}), lambda: collection.find_one({"article_number": normalized}), lambda: collection.find_one({"art": normalized}), lambda: collection.find_one({"$or": [ {"article_text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}}, {"text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}}, {"contenu": {"$regex": f"\\b{normalized}\\b", "$options": "i"}} ]}) ] article = None for strategy in search_strategies: article = strategy() if article: break if article: self._article_cache[cache_key] = article logger.info(f"✅ Article {normalized} trouvé dans {code_type.value}") else: logger.warning(f"❌ Article {normalized} NON TROUVÉ dans {code_type.value}") return article def get_multiple_articles(self, article_numbers: List[str], primary_code: LegalCode, secondary_codes: List[LegalCode] = None, user_language: str = "fr") -> List[RetrievedSource]: """Récupère plusieurs articles - MÉTHODE MANQUANTE AJOUTÉE""" retrieved = [] valid_articles = [art for art in article_numbers if ArticleNumberNormalizer.is_valid_article_number(art)] if not valid_articles: logger.info("Aucun numéro d'article valide") return retrieved logger.info(f"Recherche de {len(valid_articles)} articles...") for art_num in valid_articles: article = self.get_article_by_number(art_num, primary_code) if article: retrieved.append(self._create_source_from_article(article, art_num, primary_code, True, user_language)) if secondary_codes: for art_num in valid_articles: already_found = any( source.article_metadata.normalized_number == ArticleNumberNormalizer.normalize(art_num) for source in retrieved ) if not already_found: for code in secondary_codes: article = self.get_article_by_number(art_num, code) if article: retrieved.append(self._create_source_from_article(article, art_num, code, False, user_language)) break logger.info(f"✅ {len(retrieved)} articles récupérés") return retrieved def _create_source_from_article(self, article: Dict, art_num: str, code_type: LegalCode, primary: bool, user_language: str) -> RetrievedSource: """Crée un objet RetrievedSource à partir d'un article""" article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or '' article_text_ar = "" if user_language == "ar" and article_text_fr: article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr) metadata = ArticleMetadata( article_number=article.get('article_num', art_num) or article.get('article_number', art_num), normalized_number=ArticleNumberNormalizer.normalize(art_num), article_text_fr=article_text_fr, article_text_ar=article_text_ar, code_type=code_type, code_name_fr=Config.CODE_NAMES[code_type.value]["fr"], code_name_ar=Config.CODE_NAMES[code_type.value]["ar"], chapter=article.get('chapter'), section=article.get('section'), pdf_source=article.get('pdf_source') ) if user_language == "ar" and metadata.article_text_ar: content = metadata.article_text_ar code_name = metadata.code_name_ar else: content = metadata.article_text_fr code_name = metadata.code_name_fr full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}" return RetrievedSource( source_id=str(article.get('_id', '')), source_type=SourceType.STATUTE, article_metadata=metadata, content=full_content, relevance=RelevanceScore( similarity=1.0, topic_overlap=1.0, entity_match=1.0, article_match=1.0, code_relevance=1.0 if primary else 0.7, combined_score=1.0 if primary else 0.7, relevance_level="high", confidence=1.0 ), retrieval_timestamp=datetime.now(), retrieval_method='direct_lookup', summary=content[:500] + "..." if len(content) > 500 else content, full_text=content, primary_code=primary ) def search_similar_articles(self, query_embedding: List[float], code_type: LegalCode, limit: int = 20, user_language: str = "fr") -> List[Dict]: """Recherche sémantique d'articles similaires""" try: collection = self.get_collection(code_type) if collection is None: return [] articles = list(collection.find({"embedding": {"$exists": True}}).limit(200)) if not articles: return [] for article in articles: embedding = article.get("embedding") if embedding: similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding) article["_similarity"] = similarity else: article["_similarity"] = 0.0 articles.sort(key=lambda x: x.get("_similarity", 0), reverse=True) return articles[:limit] except Exception as e: logger.error(f"Erreur recherche sémantique {code_type.value}: {e}") return [] def search_jurisprudence_enhanced(self, query_embedding: List[float], language: str, topics: List[str] = None, primary_code: str = None, limit: int = 100) -> List[Dict]: """Recherche améliorée de jurisprudence avec filtres avancés""" try: embedding_field = "embedding_ar" if language == "ar" else "embedding_fr" sample = self.juris_collection.find_one({embedding_field: {"$exists": True}}) if not sample: embedding_field = "embedding" logger.warning(f"Champ {embedding_field} non trouvé, utilisation du champ générique 'embedding'") base_query = {embedding_field: {"$exists": True}} if primary_code: code_name_ar = Config.CODE_NAMES.get(primary_code, {}).get("ar", "") code_name_fr = Config.CODE_NAMES.get(primary_code, {}).get("fr", "") base_query["$or"] = [ {"code_ar": {"$regex": code_name_ar, "$options": "i"}}, {"code_fr": {"$regex": code_name_fr, "$options": "i"}}, {"tags_ar": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", []) if tag.isascii() is False]}}, {"tags_fr": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", [])]}} ] docs = list(self.juris_collection.find(base_query).limit(limit * 2)) if not docs: logger.warning("Aucun document de jurisprudence trouvé") return [] for doc in docs: embedding = doc.get(embedding_field) if embedding: similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding) bonus = 0.0 if topics: tags_ar = doc.get("tags_ar", []) tags_fr = doc.get("tags_fr", []) all_tags = tags_ar + tags_fr for topic in topics: topic_lower = topic.lower() for tag in all_tags: if topic_lower in tag.lower() or tag.lower() in topic_lower: bonus += 0.05 doc["_similarity"] = min(1.0, similarity + bonus) doc["_search_method"] = "enhanced_semantic_search" else: doc["_similarity"] = 0.0 docs.sort(key=lambda x: x.get("_similarity", 0), reverse=True) filtered_docs = [doc for doc in docs if doc.get("_similarity", 0) >= Config.JURIS_MINIMUM_RELEVANCE] logger.info(f"Jurisprudence: {len(filtered_docs)} documents après filtrage (sur {len(docs)})") return filtered_docs[:limit] except Exception as e: logger.error(f"Erreur recherche jurisprudence améliorée: {e}") return [] def search_jurisprudence_by_keywords(self, keywords: List[str], language: str, primary_code: str = None, limit: int = 40) -> List[Dict]: """Recherche de jurisprudence par mots-clés""" try: query = {} if language == "ar": text_fields = ["resume_ar", "faits_ar", "text_to_vector_ar.principe", "decision_ar"] tag_field = "tags_ar" else: text_fields = ["resume_fr", "faits_fr", "text_to_vector_fr.principe", "decision_fr"] tag_field = "tags_fr" keyword_queries = [] for keyword in keywords: if keyword: for field in text_fields: keyword_queries.append({field: {"$regex": keyword, "$options": "i"}}) if keyword_queries: query["$or"] = keyword_queries if primary_code: code_keywords = Config.CODE_NAMES.get(primary_code, {}).get("keywords", []) if code_keywords: code_query = {"tags_fr": {"$in": code_keywords}} if language == "ar": arabic_keywords = [kw for kw in code_keywords if not kw.isascii()] if arabic_keywords: code_query["$or"] = [{"tags_ar": {"$in": arabic_keywords}}] if "$or" in query: query["$and"] = [{"$or": query.pop("$or")}, code_query] else: query.update(code_query) docs = list(self.juris_collection.find(query).limit(limit)) for doc in docs: score = 0.0 for keyword in keywords: for field in text_fields: field_value = doc for part in field.split('.'): field_value = field_value.get(part, {}) if isinstance(field_value, dict) else "" if isinstance(field_value, str) and keyword.lower() in field_value.lower(): score += 0.1 tags = doc.get(tag_field, []) for tag in tags: for keyword in keywords: if keyword.lower() in tag.lower(): score += 0.15 doc["_keyword_score"] = min(1.0, score) doc["_search_method"] = "keyword_search" docs.sort(key=lambda x: x.get("_keyword_score", 0), reverse=True) logger.info(f"Jurisprudence par mots-clés: {len(docs)} documents trouvés") return docs[:limit] except Exception as e: logger.error(f"Erreur recherche par mots-clés: {e}") return [] def search_jurisprudence_lexical(self, query_analysis: QueryAnalysis, limit: int = 100) -> List[Dict]: """Recherche lexicale avancée dans la jurisprudence""" try: lexical_queries = LexicalSearchEngine.create_lexical_queries(query_analysis) if not lexical_queries: logger.warning("Aucune requête lexicale générée") return [] lexical_results = LexicalSearchEngine.search_lexical( self.juris_collection, lexical_queries, limit=limit ) logger.info(f"🔍 Recherche lexicale: {len(lexical_results)} documents trouvés") filtered_results = [] for doc in lexical_results: lexical_score = doc.get("_lexical_score", 0) if lexical_score >= 0.25: code_relevant = False all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] code_fr = doc.get("code_fr", "") code_ar = doc.get("code_ar", "") for code in all_codes: code_info = Config.CODE_NAMES.get(code, {}) if code_info: if (code_info.get("fr") and code_info["fr"] in code_fr) or \ (code_info.get("ar") and code_info["ar"] in code_ar): code_relevant = True break if code_relevant: lexical_score = min(1.0, lexical_score + 0.2) doc["_lexical_score"] = lexical_score filtered_results.append(doc) filtered_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True) logger.info(f"✅ Recherche lexicale filtrée: {len(filtered_results)} documents pertinents") return filtered_results[:limit] except Exception as e: logger.error(f"Erreur recherche lexicale: {e}") return [] def search_jurisprudence_by_all_codes(self, query_analysis: QueryAnalysis, query_embedding: List[float], limit: int = 150) -> List[Dict]: """Recherche de jurisprudence dans tous les codes pertinents""" all_results = [] all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] for code in all_codes: try: code_results = self.search_jurisprudence_enhanced( query_embedding, query_analysis.search_language, topics=query_analysis.extracted_topics, primary_code=code, limit=limit // len(all_codes) ) for doc in code_results: doc_code_fr = doc.get("code_fr", "") doc_code_ar = doc.get("code_ar", "") code_info = Config.CODE_NAMES.get(code, {}) if code_info: code_fr = code_info.get("fr", "") code_ar = code_info.get("ar", "") if (code_fr and code_fr in doc_code_fr) or (code_ar and code_ar in doc_code_ar): doc["_similarity"] = min(1.0, doc.get("_similarity", 0) + 0.15) doc["_code_filter"] = code all_results.extend(code_results) except Exception as e: logger.error(f"Erreur recherche jurisprudence pour code {code}: {e}") continue seen_ids = set() unique_results = [] for doc in all_results: doc_id = doc.get("_id") if doc_id and doc_id not in seen_ids: seen_ids.add(doc_id) unique_results.append(doc) unique_results.sort(key=lambda x: x.get("_similarity", 0), reverse=True) logger.info(f"🌐 Recherche multi-code: {len(unique_results)} documents uniques") return unique_results[:limit] # ============================================================ # ENHANCED SEARCH ENGINE WITH HYBRID SEARCH # ============================================================ class EnhancedEnterpriseSearchEngine: def __init__(self): self.db_manager = EnterpriseDatabaseManager() self.hybrid_searcher = HybridJurisprudenceSearch(self.db_manager) def search(self, query_analysis: QueryAnalysis) -> Dict[str, Any]: """Exécute une recherche complète avec méthode hybride""" logger.info(f"🔍 Lancement recherche hybride multi-méthodes") # 1. Récupération directe des articles cités direct_sources = self.db_manager.get_multiple_articles( query_analysis.cited_articles, query_analysis.primary_legal_code, query_analysis.secondary_codes, query_analysis.original_language ) # 2. Recherche sémantique dans les codes query_embedding = EnterpriseEmbeddingService.get_embedding(query_analysis.translated_query) primary_semantic = [] secondary_semantic = [] if query_embedding: # Recherche sémantique dans le code principal primary_articles = self.db_manager.search_similar_articles( query_embedding, query_analysis.primary_legal_code, limit=30, user_language=query_analysis.original_language ) primary_semantic = self._process_semantic_results( primary_articles, query_analysis.primary_legal_code, True, query_analysis.original_language ) # Recherche sémantique dans les codes secondaires for code in query_analysis.secondary_codes: articles = self.db_manager.search_similar_articles( query_embedding, code, limit=Config.MAX_CROSS_CODE_ARTICLES, user_language=query_analysis.original_language ) secondary_results = self._process_semantic_results( articles, code, False, query_analysis.original_language ) secondary_semantic.extend(secondary_results) # 3. Recherche hybride de jurisprudence logger.info(" 🔍 Recherche jurisprudence hybride...") hybrid_juris_docs = self.hybrid_searcher.search_hybrid( query_analysis.translated_query, query_analysis, limit=Config.MAX_JURIS_RETRIEVAL ) # 4. Recherches traditionnelles (pour complément) juris_docs_semantic = [] juris_docs_lexical = [] juris_docs_keywords = [] if query_embedding: juris_docs_semantic = self.db_manager.search_jurisprudence_by_all_codes( query_analysis, query_embedding, limit=Config.MAX_JURIS_RETRIEVAL // 3 ) juris_docs_lexical = self.db_manager.search_jurisprudence_lexical( query_analysis, limit=Config.MAX_JURIS_RETRIEVAL // 3 ) keywords = self._extract_juris_keywords(query_analysis) juris_docs_keywords = self.db_manager.search_jurisprudence_by_keywords( keywords, query_analysis.search_language, primary_code=query_analysis.primary_legal_code.value, limit=Config.MAX_JURIS_RETRIEVAL // 3 ) # 5. Fusionner TOUS les résultats de jurisprudence all_juris_docs = self._merge_all_jurisprudence_results( hybrid_juris_docs, juris_docs_semantic, juris_docs_lexical, juris_docs_keywords ) juris_sources = self._process_jurisprudence_results(all_juris_docs, query_analysis.original_language) # 6. Fusionner et organiser all_statute_sources = self._merge_sources(direct_sources, primary_semantic, secondary_semantic) statute_results = self._organize_by_relevance(all_statute_sources, query_analysis.primary_legal_code) missing_articles = self._identify_missing_articles(query_analysis.cited_articles, all_statute_sources) retrieval_metadata = { "primary_code": query_analysis.primary_legal_code.value, "secondary_codes": [c.value for c in query_analysis.secondary_codes], "direct_retrievals": len(direct_sources), "primary_semantic": len(primary_semantic), "secondary_semantic": len(secondary_semantic), "total_statutes": len(all_statute_sources), "total_jurisprudence": len(juris_sources), "missing_articles": missing_articles, "search_methods": ["direct", "semantic", "hybrid", "lexical", "keywords", "bm25", "tfidf", "exact"] } return { "statute_sources": statute_results, "jurisprudence_sources": juris_sources[:Config.MAX_JURIS_DOCS], "query_analysis": query_analysis, "retrieval_metadata": retrieval_metadata } def _extract_juris_keywords(self, query_analysis: QueryAnalysis) -> List[str]: """Extrait les mots-clés pour la recherche de jurisprudence""" keywords = set() keywords.update(query_analysis.extracted_topics) all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes] for code in all_codes: if code in Config.CODE_NAMES: code_keywords = Config.CODE_NAMES[code].get("keywords", []) keywords.update(code_keywords[:15]) keywords.update([ "طلبات جديدة", "الطلبات الجديدة", "demandes nouvelles", "new claims", "استئناف", "appel", "appeal", "تعقيب", "cassation", "طلاق إنشاء", "divorce création", "طلاق الضرر", "divorce préjudice", "جراية عمرية", "pension viagère", "lifetime pension", "تعويض", "indemnité", "compensation", "ضرر", "préjudice", "محكمة البداية", "first instance court", "محكمة الاستئناف", "appeal court", "الفصل 147", "147", "article 147", "مرفوض شكلاً", "non recevable", "إجراءات", "procédure", "شكلية", "formalité", "تقاضي", "litigation", "مبدأ التقاضي", "principe de procès", "درجة التقاضي", "degré de juridiction" ]) cleaned_keywords = [] for kw in keywords: if kw and len(kw) > 2: cleaned_keywords.append(kw.strip().lower()) return list(set(cleaned_keywords))[:40] def _merge_all_jurisprudence_results(self, hybrid_results: List[Dict], semantic_results: List[Dict], lexical_results: List[Dict], keyword_results: List[Dict]) -> List[Dict]: """Fusionne tous les types de résultats de jurisprudence""" seen_ids = set() merged = [] def add_document(doc, score_field, method): doc_id = doc.get("_id") if doc_id and doc_id not in seen_ids: seen_ids.add(doc_id) if score_field in doc: doc["_similarity"] = doc[score_field] doc["_search_method"] = method merged.append(doc) for doc in hybrid_results: add_document(doc, "_hybrid_score", "hybrid_search") for doc in semantic_results: if doc.get("_id") not in seen_ids: add_document(doc, "_similarity", "semantic_search") for doc in lexical_results: if doc.get("_id") not in seen_ids: add_document(doc, "_lexical_score", "lexical_search") for doc in keyword_results: if doc.get("_id") not in seen_ids: add_document(doc, "_keyword_score", "keyword_search") merged.sort(key=lambda x: self._calculate_combined_juris_score(x), reverse=True) logger.info(f"Fusion complète jurisprudence: {len(merged)} documents uniques") return merged def _calculate_combined_juris_score(self, doc: Dict) -> float: """Calcule un score combiné pour la jurisprudence""" hybrid_score = doc.get("_hybrid_score", 0.0) if hybrid_score > 0: return hybrid_score * 1.2 semantic_score = doc.get("_similarity", 0.0) lexical_score = doc.get("_lexical_score", 0.0) keyword_score = doc.get("_keyword_score", 0.0) method = doc.get("_search_method", "") if method == "semantic_search": return semantic_score elif method == "lexical_search": return lexical_score * 0.7 + semantic_score * 0.3 elif method == "keyword_search": return keyword_score * 0.6 + semantic_score * 0.4 else: return max(semantic_score, lexical_score, keyword_score) def _process_semantic_results(self, articles: List[Dict], code_type: LegalCode, primary: bool, user_language: str) -> List[RetrievedSource]: """Traite les résultats de recherche sémantique""" processed = [] thresholds = Config.RELEVANCE_THRESHOLDS.get(code_type.value, {"HIGH": 0.7, "MEDIUM": 0.6, "MINIMUM": 0.5}) for article in articles: similarity = article.get('_similarity', 0.0) if similarity >= thresholds["MINIMUM"]: art_num = article.get('article_num') or article.get('article_number') or article.get('art') or 'N/A' article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or '' article_text_ar = "" if user_language == "ar" and article_text_fr: article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr) metadata = ArticleMetadata( article_number=art_num, normalized_number=ArticleNumberNormalizer.normalize(art_num), article_text_fr=article_text_fr, article_text_ar=article_text_ar, code_type=code_type, code_name_fr=Config.CODE_NAMES[code_type.value]["fr"], code_name_ar=Config.CODE_NAMES[code_type.value]["ar"], chapter=article.get('chapter'), section=article.get('section'), pdf_source=article.get('pdf_source') ) if similarity >= thresholds["HIGH"]: level = "high" elif similarity >= thresholds["MEDIUM"]: level = "medium" else: level = "low" if user_language == "ar" and metadata.article_text_ar: content = metadata.article_text_ar code_name = metadata.code_name_ar else: content = metadata.article_text_fr code_name = metadata.code_name_fr full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}" code_relevance = 1.0 if primary else 0.7 combined_score = similarity * code_relevance processed.append(RetrievedSource( source_id=str(article.get('_id', '')), source_type=SourceType.STATUTE, article_metadata=metadata, content=full_content, relevance=RelevanceScore( similarity=similarity, topic_overlap=0.0, entity_match=0.0, article_match=0.0, code_relevance=code_relevance, combined_score=combined_score, relevance_level=level, confidence=similarity ), retrieval_timestamp=datetime.now(), retrieval_method='semantic_search', summary=content[:500] + "..." if len(content) > 500 else content, full_text=content, primary_code=primary )) return processed def _merge_sources(self, direct: List[RetrievedSource], primary_semantic: List[RetrievedSource], secondary_semantic: List[RetrievedSource]) -> List[RetrievedSource]: """Fusionne les sources en évitant les doublons""" seen_articles = set() merged = [] for source in direct + primary_semantic + secondary_semantic: if source.article_metadata: article_id = f"{source.article_metadata.code_type.value}_{source.article_metadata.normalized_number}" if article_id not in seen_articles: seen_articles.add(article_id) merged.append(source) return merged def _organize_by_relevance(self, sources: List[RetrievedSource], primary_code: LegalCode) -> Dict[str, List[RetrievedSource]]: """Organise les sources par niveau de pertinence""" organized = {"high": [], "medium": [], "low": []} for source in sources: level = source.relevance.relevance_level if level in organized: organized[level].append(source) for level in organized: organized[level].sort(key=lambda x: (x.primary_code, x.relevance.combined_score), reverse=True) organized["high"] = organized["high"][:Config.MAX_HIGH_ARTICLES] organized["medium"] = organized["medium"][:Config.MAX_MEDIUM_ARTICLES] organized["low"] = organized["low"][:Config.MAX_LOW_ARTICLES] return organized def _identify_missing_articles(self, cited: List[str], retrieved: List[RetrievedSource]) -> List[str]: """Identifie les articles cités mais non récupérés""" retrieved_numbers = { source.article_metadata.normalized_number for source in retrieved if source.article_metadata } missing = [] for article in cited: normalized = ArticleNumberNormalizer.normalize(article) if normalized not in retrieved_numbers and ArticleNumberNormalizer.is_valid_article_number(article): missing.append(article) return missing def _process_jurisprudence_results(self, juris_docs: List[Dict], user_language: str) -> List[RetrievedSource]: """Traite les résultats de jurisprudence de manière améliorée""" processed = [] for doc in juris_docs: similarity = self._calculate_combined_juris_score(doc) if similarity >= Config.JURIS_HIGH_RELEVANCE: relevance_level = "high" elif similarity >= Config.JURIS_MEDIUM_RELEVANCE: relevance_level = "medium" elif similarity >= Config.JURIS_MINIMUM_RELEVANCE: relevance_level = "low" else: continue if user_language == "ar": resume = doc.get('resume_ar') or doc.get('summary_ar') or doc.get('description_ar') or doc.get('contenu_ar') or '' faits = doc.get('faits_ar') or '' decision = doc.get('decision_ar') or '' code_name = doc.get('code_ar', '') juridiction = doc.get('juridiction', '') tags = doc.get('tags_ar', []) if not resume: resume_fr = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or '' if resume_fr: resume = TranslationService.translate_fr_to_ar(resume_fr) if not faits: faits_fr = doc.get('faits_fr') or '' if faits_fr: faits = TranslationService.translate_fr_to_ar(faits_fr) if not decision: decision_fr = doc.get('decision_fr') or '' if decision_fr: decision = TranslationService.translate_fr_to_ar(decision_fr) else: resume = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or '' faits = doc.get('faits_fr') or '' decision = doc.get('decision_fr') or '' code_name = doc.get('code_fr', '') juridiction = doc.get('juridiction', '') tags = doc.get('tags_fr', []) case_number = doc.get('numero_dossier') or doc.get('case_number') or '' date_decision = doc.get('date') or doc.get('date_jugement') or doc.get('date_decision') or '' full_content = "" if user_language == "ar": full_content += f"رقم القضية: {case_number}\n" if case_number else "" full_content += f"المحكمة: {juridiction}\n" if juridiction else "" full_content += f"التاريخ: {date_decision}\n" if date_decision else "" full_content += f"المجلة: {code_name}\n" if code_name else "" full_content += f"طريقة البحث: {doc.get('_search_method', '')}\n" full_content += f"درجة الصلة: {similarity:.3f}\n" full_content += f"\nالملخص:\n{resume}\n" if resume else "" full_content += f"\nالوقائع:\n{faits}\n" if faits else "" full_content += f"\nالقرار:\n{decision}\n" if decision else "" else: full_content += f"N° Affaire: {case_number}\n" if case_number else "" full_content += f"Juridiction: {juridiction}\n" if juridiction else "" full_content += f"Date: {date_decision}\n" if date_decision else "" full_content += f"Code: {code_name}\n" if code_name else "" full_content += f"Méthode de recherche: {doc.get('_search_method', '')}\n" full_content += f"Score de pertinence: {similarity:.3f}\n" full_content += f"\nRésumé:\n{resume}\n" if resume else "" full_content += f"\nFaits:\n{faits}\n" if faits else "" full_content += f"\nDécision:\n{decision}\n" if decision else "" summary = resume[:200] + "..." if len(resume) > 200 else resume processed.append(RetrievedSource( source_id=str(doc.get('_id', '')), source_type=SourceType.JURISPRUDENCE, article_metadata=None, content=full_content, relevance=RelevanceScore( similarity=similarity, topic_overlap=0.0, entity_match=0.0, article_match=0.0, code_relevance=0.9, combined_score=similarity, relevance_level=relevance_level, confidence=similarity ), retrieval_timestamp=datetime.now(), retrieval_method=doc.get('_search_method', 'jurisprudence_search'), summary=summary, full_text=full_content, primary_code=False, tags=tags[:10], code_fr=doc.get('code_fr', ''), code_ar=doc.get('code_ar', ''), juridiction=juridiction, date_decision=date_decision )) processed.sort(key=lambda x: x.relevance.combined_score, reverse=True) return processed # ============================================================ # CONTEXT BUILDER PROFESSIONNEL # ============================================================ class EnterpriseContextBuilder: @staticmethod def build_context(search_results: Dict[str, Any]) -> Tuple[str, Dict[str, Any]]: """Construit le contexte pour la génération avec améliorations""" statute_results = search_results["statute_sources"] juris_sources = search_results["jurisprudence_sources"] query_analysis = search_results["query_analysis"] retrieval_meta = search_results["retrieval_metadata"] context_parts = [] source_registry = { "statutes": {}, "jurisprudence": {}, "primary_code": query_analysis.primary_legal_code.value, "cited_articles": query_analysis.cited_articles, "missing_articles": retrieval_meta["missing_articles"], "search_methods": retrieval_meta.get("search_methods", []) } statutes_by_code = {} for level in ["high", "medium", "low"]: for source in statute_results.get(level, []): if source.article_metadata: code = source.article_metadata.code_type.value if code not in statutes_by_code: statutes_by_code[code] = [] statutes_by_code[code].append(source) statute_count = 0 primary_code = query_analysis.primary_legal_code.value if primary_code in statutes_by_code: context_parts.append(f"\n=== {Config.CODE_NAMES[primary_code]['fr'].upper()} ===") context_parts.append(f"=== {Config.CODE_NAMES[primary_code]['ar']} ===") for source in statutes_by_code[primary_code]: statute_count += 1 art_num = source.article_metadata.article_number context_parts.append(f"\n[STATUTE_{statute_count}]") context_parts.append(f"Article: {art_num}") context_parts.append(f"Code: {primary_code}") context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}") context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}") context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n") source_registry["statutes"][art_num] = { "text": source.content, "full_text": source.full_text if source.full_text else source.content, "code": primary_code, "code_name_fr": Config.CODE_NAMES[primary_code]["fr"], "code_name_ar": Config.CODE_NAMES[primary_code]["ar"], "relevance": source.relevance.combined_score, "primary_code": True } for code, sources in statutes_by_code.items(): if code != primary_code: context_parts.append(f"\n=== {Config.CODE_NAMES[code]['fr'].upper()} (Contextual) ===") context_parts.append(f"=== {Config.CODE_NAMES[code]['ar']} (سياقي) ===") for source in sources[:Config.MAX_CROSS_CODE_ARTICLES]: statute_count += 1 art_num = source.article_metadata.article_number context_parts.append(f"\n[STATUTE_{statute_count}]") context_parts.append(f"Article: {art_num}") context_parts.append(f"Code: {code}") context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}") context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}") context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n") source_registry["statutes"][art_num] = { "text": source.content, "full_text": source.full_text if source.full_text else source.content, "code": code, "code_name_fr": Config.CODE_NAMES[code]["fr"], "code_name_ar": Config.CODE_NAMES[code]["ar"], "relevance": source.relevance.combined_score, "primary_code": False } if juris_sources: if query_analysis.original_language == "ar": context_parts.append("\n=== JURISPRUDENCE (الأحكام القضائية) ===") else: context_parts.append("\n=== JURISPRUDENCE (CASE LAW SUPPORT) ===") juris_by_level = {"high": [], "medium": [], "low": []} for source in juris_sources: level = source.relevance.relevance_level if level in juris_by_level: juris_by_level[level].append(source) for level in ["high", "medium", "low"]: if juris_by_level[level]: level_display = level.upper() if query_analysis.original_language == "ar": level_names = {"high": "عالية", "medium": "متوسطة", "low": "منخفضة"} context_parts.append(f"\n=== أحكام ذات أهمية {level_names[level]} ===") else: context_parts.append(f"\n=== {level_display} RELEVANCE JURISPRUDENCE ===") for i, source in enumerate(juris_by_level[level], 1): context_parts.append(f"\n[JURIS_{level.upper()}_{i}]") if source.juridiction: context_parts.append(f"Juridiction: {source.juridiction}") if source.date_decision: context_parts.append(f"Date: {source.date_decision}") if source.code_fr or source.code_ar: code_display = source.code_ar if query_analysis.original_language == "ar" else source.code_fr context_parts.append(f"Code: {code_display}") if source.tags: tags_display = ", ".join(source.tags[:8]) context_parts.append(f"Tags: {tags_display}") if source.retrieval_method: context_parts.append(f"Search Method: {source.retrieval_method}") context_parts.append(f"Full Decision / القرار الكامل:") context_parts.append(f"{source.full_text if source.full_text else source.content}") context_parts.append(f"Relevance Score: {source.relevance.combined_score:.3f}\n") source_registry["jurisprudence"][f"JURIS_{level.upper()}_{i}"] = { "content": source.content, "full_text": source.full_text if source.full_text else source.content, "relevance": source.relevance.combined_score, "source_id": source.source_id, "level": level, "method": source.retrieval_method, "juridiction": source.juridiction, "date": source.date_decision, "tags": source.tags, "code_fr": source.code_fr, "code_ar": source.code_ar } context_parts.append("\n=== CRITICAL INSTRUCTIONS ===") context_parts.append(f"Primary Legal Code: {Config.CODE_NAMES[primary_code]['fr']}") context_parts.append(f"مجلة القانون الأساسي: {Config.CODE_NAMES[primary_code]['ar']}") context_parts.append(f"Total statutes retrieved: {statute_count}") context_parts.append(f"Total jurisprudence decisions: {len(juris_sources)}") context_parts.append(f"Cited articles requested: {query_analysis.cited_articles}") context_parts.append(f"Missing from retrieval: {retrieval_meta['missing_articles']}") context_parts.append(f"Search methods used: {', '.join(retrieval_meta.get('search_methods', []))}") context_parts.append("\n=== JURISPRUDENCE SPECIFIC RULES ===") context_parts.append("1. You CAN and SHOULD reference relevant jurisprudence principles") context_parts.append("2. When citing jurisprudence, mention the court and date if available") context_parts.append("3. Focus on the legal principles established in the jurisprudence") context_parts.append("4. Use jurisprudence to support statutory interpretation") context_parts.append("5. Highlight how jurisprudence applies to the specific case") context_parts.append("6. Consider jurisprudence from ALL relevant codes, not just the primary one") context_parts.append("\n=== STRICT RULES TO PREVENT HALLUCINATIONS ===") context_parts.append("1. ONLY cite articles that appear in the retrieved sources above") context_parts.append(" استشهد فقط بالمواد التي تظهر في المصادر المسترجعة أعلاه") context_parts.append("2. If NO statutes are retrieved, DO NOT cite any articles") context_parts.append(" إذا لم يتم استرجاع أي مواد قانونية، لا تستشهد بأي مواد") context_parts.append("3. You CAN reference principles from jurisprudence, but DO NOT cite article numbers from jurisprudence") context_parts.append(" يمكنك الإشارة إلى المبادئ من الأحكام القضائية، لكن لا تستشهد بأرقام المواد من الأحكام") context_parts.append("\nYOU MAY ONLY CITE ARTICLES THAT APPEAR ABOVE.") context_parts.append("يُسمح لك بالاستشهاد فقط بالمواد التي تظهر أعلاه.") context_parts.append("When citing, ALWAYS specify which code the article comes from.") context_parts.append("عند الاستشهاد، حدد دائمًا المجلة التي ينتمي إليها النص.") context_parts.append("DO NOT cite, quote, or reference any article not explicitly retrieved.") context_parts.append("لا تستشهد أو تنقل أو تشير إلى أي مادة لم يتم استرجاعها صراحة.") context_parts.append("USE JURISPRUDENCE FROM ALL RELEVANT CODES TO SUPPORT AND ILLUSTRATE LEGAL PRINCIPLES.") context_parts.append("استخدم الأحكام القضائية من جميع المجلات ذات الصلة لدعم وتوضيح المبادئ القانونية.") context = "\n".join(context_parts) return context, source_registry # ============================================================ # ANSWER GENERATOR PROFESSIONNEL # ============================================================ class EnterpriseAnswerGenerator: @staticmethod def generate_answer(query: str, context: str, source_registry: Dict[str, Any], query_analysis: QueryAnalysis) -> Tuple[str, List[CitedSource]]: """Génère une réponse juridique améliorée""" language = query_analysis.original_language system_prompt = EnterpriseAnswerGenerator._build_system_prompt(language, source_registry, query_analysis) user_prompt = f"""CONTEXTE ET SOURCES RÉCUPÉRÉES / السياق والمصادر المسترجعة: {context} QUESTION ORIGINALE / السؤال الأصلي: {query} QUESTION TRADUITE (pour référence) / السؤال المترجم (للإشارة): {query_analysis.translated_query} RÈGLES STRICTES POUR LA GÉNÉRATION / قواعد صارمة للتوليد: 1. Basez-vous uniquement sur les sources récupérées ci-dessus اعتمد فقط على المصادر المسترجعة أعلاه 2. Citez exactement les articles comme ils apparaissent dans les sources استشهد بالضبط بالمواد كما تظهر في المصادر 3. Si une source n'est pas disponible, expliquez clairement cette limite إذا لم يكن المصدر متوفراً، اشرح هذا القيد بوضوح 4. Fournissez une analyse juridique complète et pratique قدم تحليلاً قانونياً شاملاً وعملياً 5. Adaptez la réponse à la langue de l'utilisateur ({language}) قم بتكييف الإجابة مع لغة المستخدم ({language}) 6. Utilisez la jurisprudence pour illustrer et soutenir les principes légaux استخدم الأحكام القضائية لتوضيح ودعم المبادئ القانونية 7. Considérez la jurisprudence de TOUS les codes pertinents ضع في الاعتبار الأحكام القضائية من جميع المجلات ذات الصلة Générez une réponse juridique complète, précise et pratique. قم بتوليد إجابة قانونية شاملة ودقيقة وعملية.""" try: response = chat_client.chat.completions.create( model=Config.CHAT_MODEL, messages=[ {"role": "system", "content": system_prompt}, {"role": "user", "content": user_prompt} ], temperature=Config.TEMP_GENERATION, max_tokens=4000 ) answer = response.choices[0].message.content.strip() cited_sources = EnterpriseAnswerGenerator._extract_citations(answer, source_registry) logger.info(f"✅ Réponse générée ({len(answer)} caractères)") logger.info(f"📌 Sources citées: {len(cited_sources)}") return answer, cited_sources except Exception as e: logger.error(f"Erreur génération réponse: {e}") return "Une erreur est survenue lors de la génération de la réponse. Veuillez réessayer.", [] @staticmethod def _build_system_prompt(language: str, source_registry: Dict[str, Any], query_analysis: QueryAnalysis) -> str: """Construit le prompt système amélioré""" primary_code = query_analysis.primary_legal_code.value primary_code_name_fr = Config.CODE_NAMES[primary_code]["fr"] primary_code_name_ar = Config.CODE_NAMES[primary_code]["ar"] articles_by_code = {} for article, info in source_registry["statutes"].items(): code = info["code"] if code not in articles_by_code: articles_by_code[code] = [] code_name = info["code_name_ar"] if language == "ar" else info["code_name_fr"] articles_by_code[code].append(f"{article} ({code_name})") available_text = "" for code, articles in articles_by_code.items(): code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"] available_text += f"\n{code_name}: {', '.join(articles)}" juris_count = len(source_registry.get("jurisprudence", {})) missing_articles = source_registry.get("missing_articles", []) juris_by_code = {} for juris_id, juris_info in source_registry.get("jurisprudence", {}).items(): code_fr = juris_info.get("code_fr", "") code_ar = juris_info.get("code_ar", "") for code, code_info in Config.CODE_NAMES.items(): if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar): if code not in juris_by_code: juris_by_code[code] = 0 juris_by_code[code] += 1 break juris_by_code_text = "" for code, count in juris_by_code.items(): code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"] juris_by_code_text += f"\n{code_name}: {count} decisions" if language == "ar": return f"""أنت خبير قانوني تونسي محترف متخصص في جميع المجلات العشرة التونسية. المجلة الأساسية المعنية: {primary_code_name_ar} المواد القانونية المتاحة:{available_text if articles_by_code else " لا توجد مواد قانونية مسترجعة"} عدد الأحكام القضائية المتاحة: {juris_count} حكم توزيع الأحكام القضائية حسب المجلة:{juris_by_code_text if juris_by_code_text else " لا توجد أحكام قضائية مصنفة"} المواد المطلوبة وغير المتوفرة: {', '.join(missing_articles)} أنت خبير في: 1. تفسير النصوص القانونية التونسية 2. تحليل الأحكام القضائية وتطبيقها على الحالات الواقعية 3. تقديم نصائح قانونية عملية ومفصلة 4. التمييز بين المصادر القانونية المختلفة 5. استخدام الاجتهاد القضائي من جميع المجلات ذات الصلة لتوضيح المبادئ القانونية قواعد صارمة: 1. لا تخترع أي مواد أو أحكام غير موجودة في المصادر 2. إذا لم تجد مصدراً، اعترف بذلك واشرح البدائل 3. كن دقيقاً في الاستشهادات والإحالات 4. قدم إجابة متوازنة وعملية 5. استخدم الأحكام القضائية من جميع المجلات ذات الصلة لتوضيح كيفية تطبيق النصوص القانونية استخدم الأحكام القضائية المتاحة لتوضيح: - كيفية تفسير المحاكم للنصوص القانونية - المبادئ القانونية المستقرة في الاجتهاد - كيفية تطبيق القانون على حالات مشابهة - الاتجاهات الحديثة في التفسير القضائي - الاختلافات أو التشابهات بين تفسيرات المحاكم للمواد المختلفة قم بتحليل السؤال بدقة وقدم إجابة شاملة تعتمد على المصادر المتاحة من جميع المجلات ذات الصلة.""" else: return f"""You are a professional Tunisian legal expert specializing in all 10 Tunisian codes. Primary Legal Code: {primary_code_name_fr} Available Legal Articles:{available_text if articles_by_code else " NO statutes retrieved"} Available Jurisprudence Decisions: {juris_count} decisions Jurisprudence Distribution by Code:{juris_by_code_text if juris_by_code_text else " No jurisprudence classified by code"} Requested but Unavailable Articles: {', '.join(missing_articles)} You are expert in: 1. Interpreting Tunisian legal texts 2. Analyzing case law and applying it to real cases 3. Providing practical, detailed legal advice 4. Distinguishing between different legal sources 5. Using jurisprudence from ALL relevant codes to illustrate legal principles Strict Rules: 1. DO NOT invent any articles or jurisprudence not in the sources 2. If a source is not found, acknowledge this and explain alternatives 3. Be precise in citations and references 4. Provide balanced, practical advice 5. Use available jurisprudence from ALL relevant codes to clarify how legal texts are applied Use available jurisprudence to illustrate: - How courts interpret legal texts - Established legal principles in case law - How the law is applied to similar cases - Recent trends in judicial interpretation - Differences or similarities between court interpretations of different articles Analyze the question accurately and provide a comprehensive answer based on available sources from ALL relevant codes.""" @staticmethod def _extract_citations(answer: str, source_registry: Dict[str, Any]) -> List[CitedSource]: """Extrait les citations de la réponse""" citations = [] for article, info in source_registry["statutes"].items(): normalized_article = ArticleNumberNormalizer.normalize(article) patterns = [ f"(?:الفصل|فصل|Article|article|المادة|مادة)\\s*{normalized_article}\\b", f"\\b{normalized_article}\\b(?!\\s*bis|\\s*ter|\\s*quater)" ] for pattern in patterns: matches = re.finditer(pattern, answer, re.IGNORECASE) for match in matches: context = answer[max(0, match.start()-50):min(len(answer), match.end()+50)] citations.append(CitedSource( article_number=article, code_type=LegalCode.from_string(info["code"]), citation_context=context, citation_position=match.start(), retrieval_status=RetrievalStatus.SUCCESSFULLY_RETRIEVED )) return citations # ============================================================ # VALIDATOR PROFESSIONNEL # ============================================================ class AnswerValidator: @staticmethod def validate(answer: str, cited_sources: List[CitedSource], source_registry: Dict[str, Any]) -> ValidationResult: """Valide la réponse générée""" hallucinated = [] missing_retrievals = [] errors = [] warnings = [] retrieved_lookup = {} for article, info in source_registry["statutes"].items(): normalized = ArticleNumberNormalizer.normalize(article) retrieved_lookup[normalized] = { "code": info["code"], "original": article, "info": info } for citation in cited_sources: normalized_cite = ArticleNumberNormalizer.normalize(citation.article_number) if normalized_cite in retrieved_lookup: retrieved_info = retrieved_lookup[normalized_cite] if retrieved_info["code"] == citation.code_type.value: citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED else: warnings.append(f"Article {citation.article_number} cited with code {citation.code_type.value} but retrieved from {retrieved_info['code']}") citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED else: hallucinated.append(f"{citation.article_number} ({citation.code_type.value})") citation.retrieval_status = RetrievalStatus.CITED_NOT_RETRIEVED errors.append(f"Article {citation.article_number} from {citation.code_type.value} cited but NOT retrieved") all_article_numbers = ArticleNumberNormalizer.extract_all_numbers(answer) for art_num in all_article_numbers: normalized_art_num = ArticleNumberNormalizer.normalize(art_num) found_in_sources = normalized_art_num in retrieved_lookup already_cited = any( ArticleNumberNormalizer.normalize(citation.article_number) == normalized_art_num for citation in cited_sources ) if not found_in_sources and not already_cited: warnings.append(f"Potential uncited reference to article {art_num}") for article in source_registry.get("cited_articles", []): normalized = ArticleNumberNormalizer.normalize(article) if normalized not in retrieved_lookup: missing_retrievals.append(article) warnings.append(f"Requested article {article} was NOT retrieved") if hallucinated: confidence = 0.0 elif missing_retrievals: confidence = 0.7 elif warnings: confidence = 0.85 else: confidence = 0.95 is_valid = len(hallucinated) == 0 if not is_valid: errors.append(f"CRITICAL: {len(hallucinated)} hallucinated citations detected") return ValidationResult( is_valid=is_valid, confidence_score=confidence, hallucinated_citations=hallucinated, missing_retrievals=missing_retrievals, validation_errors=errors, validation_warnings=warnings ) # ============================================================ # CHATBOT FINAL AMÉLIORÉ # ============================================================ class UltimateLegalChatbot: """Chatbot juridique ultime avec toutes les améliorations""" def __init__(self): self.query_analyzer = EnterpriseQueryAnalyzer() self.search_engine = EnhancedEnterpriseSearchEngine() self.context_builder = EnterpriseContextBuilder() self.answer_generator = EnterpriseAnswerGenerator() self.validator = AnswerValidator() def process_query(self, query: str) -> LegalAnswer: """Traite une requête avec toutes les améliorations""" logger.info(f"\n{'='*120}") logger.info("🚀 ULTIMATE LEGAL CHATBOT - 10 TUNISIAN CODES (HYBRID SEARCH)") logger.info(f"{'='*120}\n") start_time = datetime.now() logger.info("📊 ÉTAPE 1: Analyse de la requête") query_analysis = self.query_analyzer.analyze_query(query) logger.info("\n🔍 ÉTAPE 2: Récupération des sources améliorée avec recherche hybride") search_results = self.search_engine.search(query_analysis) logger.info("\n🧱 ÉTAPE 3: Construction du contexte améliorée") context, source_registry = self.context_builder.build_context(search_results) logger.info("\n✍️ ÉTAPE 4: Génération de la réponse améliorée") answer_text, cited_sources = self.answer_generator.generate_answer( query, context, source_registry, query_analysis ) logger.info("\n✅ ÉTAPE 5: Validation de la réponse") validation_result = self.validator.validate(answer_text, cited_sources, source_registry) all_sources = [] for level in ["high", "medium", "low"]: all_sources.extend(search_results["statute_sources"].get(level, [])) all_sources.extend(search_results["jurisprudence_sources"]) end_time = datetime.now() processing_time = (end_time - start_time).total_seconds() juris_code_distribution = {} for source in search_results["jurisprudence_sources"]: code_fr = source.code_fr code_ar = source.code_ar for code_key, code_info in Config.CODE_NAMES.items(): if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar): if code_key not in juris_code_distribution: juris_code_distribution[code_key] = 0 juris_code_distribution[code_key] += 1 legal_answer = LegalAnswer( answer_text=answer_text, retrieved_sources=all_sources, cited_sources=cited_sources, validation_result=validation_result, query_analysis=query_analysis, processing_metadata={ "processing_time": processing_time, "retrieval_stats": search_results["retrieval_metadata"], "translation_applied": query_analysis.original_language == "ar", "jurisprudence_found": len(search_results["jurisprudence_sources"]), "primary_code": query_analysis.primary_legal_code.value, "secondary_codes": [c.value for c in query_analysis.secondary_codes], "total_codes_available": 10, "search_methods": search_results["retrieval_metadata"]["search_methods"], "hybrid_search_used": True, "jurisprudence_code_distribution": juris_code_distribution, "timestamp": datetime.now().isoformat() }, confidence_score=validation_result.confidence_score * query_analysis.code_confidence ) logger.info(f"\n✅ Traitement terminé en {processing_time:.2f} secondes") logger.info(f"📊 Résumé: {len(all_sources)} sources totales") logger.info(f"🎯 Système 10 codes tunisiens avec recherche hybride opérationnel") return legal_answer # ============================================================ # CHAINLIT APPLICATION (CHAT INTERFACE) # ============================================================ # ... (all previous imports and classes remain exactly the same until the Chainlit part) ... # ============================================================ # CHAINLIT APPLICATION (CHAT INTERFACE) - MODIFIED # ============================================================ chatbot = UltimateLegalChatbot() @cl.on_chat_start async def start(): """Initialise la session de chat avec un message d'accueil minimal""" await cl.Message( content="""Legal Assistant | مساعد قانوني Ask your legal question | اطرح سؤالك القانوني""" ).send() @cl.on_message async def main(message: cl.Message): """Traite le message de l'utilisateur et renvoie la réponse + sources complètes sans termes techniques""" user_query = message.content.strip() if not user_query: await cl.Message(content="Veuillez poser une question valide.").send() return async with cl.Step(name="Analyse et recherche en cours", type="loading"): legal_answer = await asyncio.to_thread(chatbot.process_query, user_query) await cl.Message(content=legal_answer.answer_text).send() sources_message = "" statutes = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.STATUTE] if statutes: sources_message += "### 📜 Articles de loi (texte intégral)\n\n" for idx, stat in enumerate(statutes, 1): if stat.article_metadata: art_num = stat.article_metadata.article_number if legal_answer.query_analysis.original_language == "ar": code_name = stat.article_metadata.code_name_ar else: code_name = stat.article_metadata.code_name_fr sources_message += f"**{idx}. {code_name} – Article {art_num}**\n" sources_message += f"```\n{stat.full_text}\n```\n\n" juris = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.JURISPRUDENCE] if juris: sources_message += "### ⚖️ Décisions de jurisprudence (texte intégral)\n\n" for idx, dec in enumerate(juris, 1): title = f"**{idx}. {dec.juridiction or 'Décision'}**" if dec.date_decision: title += f" – {dec.date_decision}" sources_message += title + "\n" if legal_answer.query_analysis.original_language == "ar": code_display = dec.code_ar else: code_display = dec.code_fr if code_display: sources_message += f"*Code : {code_display}*\n" sources_message += f"```\n{dec.full_text}\n```\n\n" if sources_message: await cl.Message(content=sources_message).send() else: await cl.Message(content="*Aucune source textuelle n'a été trouvée pour cette question.*").send() if __name__ == "__main__": from chainlit.cli import run run()