LegalBot / app.py
Amyneee's picture
Update app.py
c82f889 verified
Raw History Blame Contribute Delete
140 kB
import re
import json
import os
from dotenv import load_dotenv
from pymongo import MongoClient
from openai import OpenAI, AzureOpenAI
import numpy as np
from sklearn.metrics.pairwise import cosine_similarity
from typing import List, Dict, Tuple, Optional, Any, Set
import logging
from functools import lru_cache
from dataclasses import dataclass, field
from datetime import datetime
import hashlib
from enum import Enum
import time
import math
from collections import Counter
import asyncio
import chainlit as cl
# Charger les variables d'environnement
load_dotenv()
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# ============================================================
# CONFIGURATION PROFESSIONNELLE AVEC NOMS EXACTS
# ============================================================
class Config:
"""Professional configuration for all Tunisian codes"""
MONGO_URI = os.getenv("MONGO_URI")
DB_CODE = os.getenv("DB_CODE")
COLLECTIONS = {
"CSP": os.getenv("CSP", "csp"),
"DROITS_REELS": os.getenv("DROITS_REELS", "code_droits_reels"),
"OBLIGATIONS_CONTRATS": os.getenv("OBLIGATIONS_CONTRATS", "code_des_obligations_et_des_contrats"),
"PROCEDURE_CIVILE": os.getenv("PROCEDURE_CIVILE", "code_de_procédure_civile_et_commerciale"),
"CODE_TRAVAIL": os.getenv("CODE_TRAVAIL", "code-de-travail"),
"DROITS_PROCEDURES_FISCAUX": os.getenv("DROITS_PROCEDURES_FISCAUX", "code-des-droits-et-procedures-fiscaux"),
"DROIT_INTERNATIONAL_PRIVE": os.getenv("DROIT_INTERNATIONAL_PRIVE", "code-du-droit-international-privé"),
"CODE_PENAL": os.getenv("CODE_PENAL", "code-pénal"),
"CODE_COMMERCE": os.getenv("CODE_COMMERCE", "code_de_commerce"),
"PROCEDURES_PENALES": os.getenv("PROCEDURES_PENALES", "code_des_procedures_penales")
}
DB_JURIS = os.getenv("DB_JURIS")
COL_JURIS = os.getenv("COL_JURIS")
AZURE_ENDPOINT = os.getenv("AZURE_ENDPOINT")
AZURE_API_KEY = os.getenv("AZURE_API_KEY")
AZURE_API_VERSION = os.getenv("AZURE_API_VERSION")
EMBEDDING_MODEL = os.getenv("EMBEDDING_MODEL")
CHAT_MODEL = os.getenv("CHAT_MODEL")
RELEVANCE_THRESHOLDS = {
"CSP": {"HIGH": 0.78, "MEDIUM": 0.65, "MINIMUM": 0.55},
"DROITS_REELS": {"HIGH": 0.75, "MEDIUM": 0.62, "MINIMUM": 0.52},
"OBLIGATIONS_CONTRATS": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50},
"PROCEDURE_CIVILE": {"HIGH": 0.65, "MEDIUM": 0.55, "MINIMUM": 0.45},
"CODE_TRAVAIL": {"HIGH": 0.73, "MEDIUM": 0.61, "MINIMUM": 0.51},
"DROITS_PROCEDURES_FISCAUX": {"HIGH": 0.68, "MEDIUM": 0.56, "MINIMUM": 0.46},
"DROIT_INTERNATIONAL_PRIVE": {"HIGH": 0.72, "MEDIUM": 0.60, "MINIMUM": 0.50},
"CODE_PENAL": {"HIGH": 0.76, "MEDIUM": 0.64, "MINIMUM": 0.54},
"CODE_COMMERCE": {"HIGH": 0.71, "MEDIUM": 0.59, "MINIMUM": 0.49},
"PROCEDURES_PENALES": {"HIGH": 0.74, "MEDIUM": 0.62, "MINIMUM": 0.52}
}
# Optimized jurisprudence thresholds
JURIS_HIGH_RELEVANCE = 0.68
JURIS_MEDIUM_RELEVANCE = 0.52
JURIS_MINIMUM_RELEVANCE = 0.35
TEMP_ANALYSIS = 0.05
TEMP_CLASSIFICATION = 0.1
TEMP_GENERATION = 0.05
MAX_HIGH_ARTICLES = 12
MAX_MEDIUM_ARTICLES = 8
MAX_LOW_ARTICLES = 4
MAX_JURIS_DOCS = 20 # REDUCED FROM 60
MAX_JURIS_RETRIEVAL = 80 # REDUCED FROM 250
MAX_CROSS_CODE_ARTICLES = 6
MAX_LEXICAL_RESULTS = 30
# Optimized secondary code mapping
SECONDARY_CODE_MAPPING = {
"CSP": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "DROITS_REELS"],
"DROITS_REELS": ["PROCEDURE_CIVILE", "CSP", "OBLIGATIONS_CONTRATS", "CODE_COMMERCE"],
"OBLIGATIONS_CONTRATS": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "CODE_TRAVAIL", "CSP"],
"PROCEDURE_CIVILE": ["CSP", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL"],
"CODE_TRAVAIL": ["PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS", "CSP"],
"DROITS_PROCEDURES_FISCAUX": ["CODE_COMMERCE", "PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"],
"DROIT_INTERNATIONAL_PRIVE": ["CSP", "PROCEDURE_CIVILE", "CODE_COMMERCE", "OBLIGATIONS_CONTRATS"],
"CODE_PENAL": ["PROCEDURES_PENALES", "PROCEDURE_CIVILE", "CSP"],
"CODE_COMMERCE": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS", "CODE_TRAVAIL", "DROITS_REELS"],
"PROCEDURES_PENALES": ["CODE_PENAL", "PROCEDURE_CIVILE", "CSP"]
}
CODE_NAMES = {
"CSP": {
"fr": "Code du Statut Personnel",
"ar": "مجلة الأحوال الشخصية",
"keywords": ["mariage", "divorce", "héritage", "succession", "paternité", "نكاح", "طلاق", "إرث", "ميراث", "نسب", "نفقة", "حضانة", "زواج", "فرقة"]
},
"DROITS_REELS": {
"fr": "Code des Droits Réels",
"ar": "مجلة الحقوق العينية",
"keywords": ["propriété", "immobilier", "hypothèque", "servitude", "usufruit", "ملكية", "عقار", "رهن", "أراضي", "عقارية", "حيازة", "تملك"]
},
"OBLIGATIONS_CONTRATS": {
"fr": "Code des Obligations et des Contrats",
"ar": "مجلة الالتزامات والعقود",
"keywords": ["contrat", "obligation", "responsabilité", "délit", "عقد", "التزام", "مسؤولية", "خطأ", "تعويض", "إبرام", "فسخ", "إبطال"]
},
"PROCEDURE_CIVILE": {
"fr": "Code de Procédure Civile et Commerciale",
"ar": "مجلة المرافعات المدنية والتجارية",
"keywords": ["procédure", "appel", "recours", "jugement", "تعقيب", "نقض", "محكمة", "حكم", "قرار", "طعن", "إجراءات", "استئناف", "طلبات جديدة", "الطلبات الجديدة"]
},
"CODE_TRAVAIL": {
"fr": "Code du Travail",
"ar": "مجلة الشغل",
"keywords": ["travail", "emploi", "licenciement", "salaire", "contrat", "شغل", "عقد شغل", "فصل", "أجير", "أجرة", "إجازة", "تعويض"]
},
"DROITS_PROCEDURES_FISCAUX": {
"fr": "Code des Droits et Procédures Fiscaux",
"ar": "مجلة الحقوق والإجراءات الضريبية",
"keywords": ["fiscal", "impôt", "taxe", "droit fiscal", "procédure fiscale", "ضريبة", "جباية", "ديوان", "غرامة", "تحصيل", "تهرب"]
},
"DROIT_INTERNATIONAL_PRIVE": {
"fr": "Code de Droit International Privé",
"ar": "مجلة القانون الدولي الخاص",
"keywords": ["international", "conflit de lois", "nationalité", "étranger", "تنازع القوانين", "اختصاص دولي", "جنسية", "أجانب", "إقليمية"]
},
"CODE_PENAL": {
"fr": "Code Pénal",
"ar": "المجلة الجزائية",
"keywords": ["pénal", "crime", "délit", "contravention", "peine", "جناية", "جنحة", "مخالفة", "عقوبة", "سجن", "حبس", "جرم"]
},
"CODE_COMMERCE": {
"fr": "Code de Commerce",
"ar": "مجلة التجارية",
"keywords": ["commerce", "commerçant", "entreprise", "société", "faillite", "تاجر", "تجار", "شركة", "سجل تجاري", "إفلاس", "تسوية"]
},
"PROCEDURES_PENALES": {
"fr": "Code des Procédures Pénales",
"ar": "مجلة الإجراءات الجزائية",
"keywords": ["procédure pénale", "enquête", "instruction", "تحقيق", "تحقيق جزائي", "قاضي التحقيق", "إحالة", "نيابة", "محاكمة"]
}
}
# Initialisation des clients
mongo_client = MongoClient(Config.MONGO_URI)
embedding_client = OpenAI(base_url=f"{Config.AZURE_ENDPOINT}openai/v1/", api_key=Config.AZURE_API_KEY)
chat_client = AzureOpenAI(api_key=Config.AZURE_API_KEY, azure_endpoint=Config.AZURE_ENDPOINT, api_version=Config.AZURE_API_VERSION)
# ============================================================
# TRANSLATION SERVICE
# ============================================================
class TranslationService:
"""Service de traduction professionnel"""
_translation_cache = {}
@staticmethod
def translate_text(text: str, source_lang: str, target_lang: str) -> str:
if source_lang == target_lang or not text:
return text
cache_key = f"{source_lang}_{target_lang}_{hashlib.md5(text.encode()).hexdigest()}"
if cache_key in TranslationService._translation_cache:
return TranslationService._translation_cache[cache_key]
try:
if target_lang == "ar":
system_content = "أنت مترجم قانوني محترف متخصص في الترجمة من الفرنسية إلى العربية. حافظ على الدقة القانونية والمصطلحات الفنية."
prompt = f"""ترجم النص القانوني التالي بدقة مع الحفاظ على:
1. المعنى القانوني الدقيق
2. المصطلحات القانونية المتخصصة
3. الأرقام والمراجع القانونية
4. السياق القانوني التونسي
النص الفرنسي: {text}
الترجمة العربية (تجنب الإضافة أو الحذف، كن دقيقًا):"""
else:
system_content = "Tu es un traducteur juridique professionnel spécialisé en droit tunisien. Préserve la précision juridique et la terminologie technique."
prompt = f"""Traduis ce texte juridique avec précision en préservant:
1. Le sens juridique exact
2. La terminologie juridique spécialisée
3. Les chiffres et références légales
4. Le contexte juridique tunisien
Texte arabe: {text}
Traduction française (sans ajout ni omission, sois précis):"""
response = chat_client.chat.completions.create(
model=Config.CHAT_MODEL,
messages=[
{"role": "system", "content": system_content},
{"role": "user", "content": prompt}
],
temperature=0.1,
max_tokens=2000
)
translation = response.choices[0].message.content.strip()
TranslationService._translation_cache[cache_key] = translation
logger.info(f"✅ Traduction {source_lang} → {target_lang} effectuée")
return translation
except Exception as e:
logger.error(f"Erreur de traduction: {e}")
return text
@staticmethod
def translate_ar_to_fr(text: str) -> str:
return TranslationService.translate_text(text, "ar", "fr")
@staticmethod
def translate_fr_to_ar(text: str) -> str:
return TranslationService.translate_text(text, "fr", "ar")
@staticmethod
def detect_and_translate(query: str) -> Dict[str, str]:
arabic_chars = len(re.findall(r'[\u0600-\u06FF]', query))
total_chars = len(query.replace(' ', ''))
if total_chars > 0 and (arabic_chars / total_chars) > 0.2:
translated = TranslationService.translate_ar_to_fr(query)
return {
"original_query": query,
"translated_query": translated,
"original_language": "ar",
"search_language": "fr"
}
else:
return {
"original_query": query,
"translated_query": query,
"original_language": "fr",
"search_language": "fr"
}
# ============================================================
# ENUMS & DATA STRUCTURES
# ============================================================
class SourceType(Enum):
STATUTE = "statute"
JURISPRUDENCE = "jurisprudence"
class LegalCode(Enum):
CSP = "CSP"
DROITS_REELS = "DROITS_REELS"
OBLIGATIONS_CONTRATS = "OBLIGATIONS_CONTRATS"
PROCEDURE_CIVILE = "PROCEDURE_CIVILE"
CODE_TRAVAIL = "CODE_TRAVAIL"
DROITS_PROCEDURES_FISCAUX = "DROITS_PROCEDURES_FISCAUX"
DROIT_INTERNATIONAL_PRIVE = "DROIT_INTERNATIONAL_PRIVE"
CODE_PENAL = "CODE_PENAL"
CODE_COMMERCE = "CODE_COMMERCE"
PROCEDURES_PENALES = "PROCEDURES_PENALES"
@classmethod
def from_string(cls, value: str):
try:
normalized = value.upper().replace("-", "_").replace(" ", "_")
return cls(normalized)
except:
return cls.CSP
class RetrievalStatus(Enum):
SUCCESSFULLY_RETRIEVED = "retrieved"
CITED_NOT_RETRIEVED = "cited_not_retrieved"
@dataclass
class ArticleMetadata:
article_number: str
normalized_number: str
article_text_fr: str
article_text_ar: str
code_type: LegalCode
code_name_fr: str
code_name_ar: str
chapter: Optional[str] = None
section: Optional[str] = None
pdf_source: Optional[str] = None
@dataclass
class RelevanceScore:
similarity: float
topic_overlap: float
entity_match: float
article_match: float
code_relevance: float
combined_score: float
relevance_level: str
confidence: float
@dataclass
class RetrievedSource:
source_id: str
source_type: SourceType
article_metadata: Optional[ArticleMetadata]
content: str
relevance: RelevanceScore
retrieval_timestamp: datetime
retrieval_method: str
summary: str = ""
full_text: str = ""
primary_code: bool = True
tags: List[str] = field(default_factory=list)
code_fr: str = ""
code_ar: str = ""
juridiction: str = ""
date_decision: str = ""
@dataclass
class CitedSource:
article_number: str
code_type: LegalCode
citation_context: str
citation_position: int
retrieval_status: RetrievalStatus
@dataclass
class ValidationResult:
is_valid: bool
confidence_score: float
hallucinated_citations: List[str]
missing_retrievals: List[str]
validation_errors: List[str]
validation_warnings: List[str]
@dataclass
class QueryAnalysis:
original_query: str
translated_query: str
original_language: str
search_language: str
extracted_topics: List[str]
legal_entities: List[str]
cited_articles: List[str]
primary_legal_code: LegalCode
secondary_codes: List[LegalCode]
question_type: str
complexity_score: float
search_queries: List[str]
requires_statutory_law: bool
code_confidence: float
@dataclass
class LegalAnswer:
answer_text: str
retrieved_sources: List[RetrievedSource]
cited_sources: List[CitedSource]
validation_result: ValidationResult
query_analysis: QueryAnalysis
processing_metadata: Dict[str, Any]
confidence_score: float
# ============================================================
# ARTICLE NUMBER NORMALIZER PROFESSIONNEL
# ============================================================
class ArticleNumberNormalizer:
@staticmethod
def normalize(article_ref: str) -> str:
"""Normalise une référence d'article pour obtenir uniquement le numéro"""
if not article_ref or not isinstance(article_ref, str):
return ""
cleaned = re.sub(r'[^\d\s]', '', article_ref.strip())
cleaned = re.sub(r'\s+', ' ', cleaned)
match = re.search(r'(\d+)(?:\s*(?:bis|ter|quater))?', cleaned, re.IGNORECASE)
if match:
return match.group(0).strip().lower()
return cleaned
@staticmethod
def extract_all_numbers(text: str) -> Set[str]:
"""Extrait uniquement les numéros d'articles valides avec mots-clés spécifiques"""
if not text:
return set()
patterns = [
r'\b(?:Article|article|art\.|Art\.)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b',
r'\b(?:الفصل|فصل|المادة|مادة|المادّة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b',
r'\b(?:n°|N°|numéro|رقم)\s*(\d+(?:\s*(?:bis|ter|quater))?)\b',
r'[\(\[]\s*(?:article|Article|art\.|الفصل|المادة)\s*(\d+(?:\s*(?:bis|ter|quater))?)\s*[\)\]]'
]
numbers = set()
for pattern in patterns:
matches = re.findall(pattern, text, re.IGNORECASE | re.UNICODE)
for match in matches:
if isinstance(match, tuple):
num = match[0]
else:
num = match
if num.isdigit():
num_int = int(num)
if (1900 <= num_int <= 2100) or num_int > 9999:
continue
normalized = ArticleNumberNormalizer.normalize(num)
if normalized and re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE):
numbers.add(normalized)
return numbers
@staticmethod
def is_valid_article_number(art_num: str) -> bool:
"""Vérifie si un numéro d'article est valide"""
if not art_num or not isinstance(art_num, str):
return False
normalized = ArticleNumberNormalizer.normalize(art_num)
if not normalized:
return False
if not re.match(r'^\d+(?:\s*(?:bis|ter|quater))?$', normalized, re.IGNORECASE):
return False
match = re.search(r'(\d+)', normalized)
if not match:
return False
num = int(match.group(1))
if 1900 <= num <= 2100:
return False
if num > 9999:
return False
return True
# ============================================================
# LEXICAL SEARCH ENGINE
# ============================================================
class LexicalSearchEngine:
"""Moteur de recherche lexicale avancée pour la jurisprudence"""
@staticmethod
def create_lexical_queries(query_analysis: QueryAnalysis) -> List[Dict]:
"""Crée des requêtes lexicales pour la recherche dans la jurisprudence"""
queries = []
keywords_fr = []
keywords_ar = []
for topic in query_analysis.extracted_topics:
if topic and len(topic) > 2:
keywords_fr.append(topic.lower())
if not any(char in topic for char in 'اأإآبتثجحخدذرزسشصضطظعغفقكلمنهوي'):
try:
translated = TranslationService.translate_fr_to_ar(topic)
keywords_ar.append(translated)
except:
pass
query_terms = re.findall(r'\b\w+\b', query_analysis.translated_query.lower())
keywords_fr.extend([term for term in query_terms if len(term) > 3])
if query_analysis.original_language == "ar":
arabic_terms = re.findall(r'[\u0600-\u06FF]+', query_analysis.original_query)
keywords_ar.extend([term for term in arabic_terms if len(term) > 2])
keywords_fr = list(set([k for k in keywords_fr if len(k) > 2]))[:30]
keywords_ar = list(set([k for k in keywords_ar if len(k) > 2]))[:30]
if keywords_fr:
queries.append({
"language": "fr",
"keywords": keywords_fr,
"search_fields": ["resume_fr", "faits_fr", "decision_fr", "tags_fr", "text_to_vector_fr.principe"],
"boost_fields": {
"resume_fr": 2.0,
"faits_fr": 1.5,
"text_to_vector_fr.principe": 2.5,
"tags_fr": 3.0
}
})
if keywords_ar:
queries.append({
"language": "ar",
"keywords": keywords_ar,
"search_fields": ["resume_ar", "faits_ar", "decision_ar", "tags_ar", "text_to_vector_ar.principe"],
"boost_fields": {
"resume_ar": 2.0,
"faits_ar": 1.5,
"text_to_vector_ar.principe": 2.5,
"tags_ar": 3.0
}
})
all_codes = [query_analysis.primary_legal_code] + query_analysis.secondary_codes
for code in all_codes:
code_keywords = Config.CODE_NAMES.get(code.value, {}).get("keywords", [])
if code_keywords:
fr_keywords = [kw for kw in code_keywords if kw.isascii()]
ar_keywords = [kw for kw in code_keywords if not kw.isascii()]
if fr_keywords:
queries.append({
"language": "fr",
"keywords": fr_keywords[:15],
"search_fields": ["tags_fr", "resume_fr", "code_fr"],
"boost_fields": {"tags_fr": 3.0, "code_fr": 2.0},
"code_filter": code.value
})
if ar_keywords:
queries.append({
"language": "ar",
"keywords": ar_keywords[:15],
"search_fields": ["tags_ar", "resume_ar", "code_ar"],
"boost_fields": {"tags_ar": 3.0, "code_ar": 2.0},
"code_filter": code.value
})
if query_analysis.cited_articles:
for article in query_analysis.cited_articles:
if ArticleNumberNormalizer.is_valid_article_number(article):
queries.append({
"language": "both",
"keywords": [f"article {article}", f"الفصل {article}"],
"search_fields": ["articles_cites.article_num", "text_to_vector_ar.principe", "text_to_vector_fr.principe"],
"boost_fields": {"articles_cites.article_num": 5.0}
})
return queries
@staticmethod
def search_lexical(collection, queries: List[Dict], limit: int = 100) -> List[Dict]:
"""Exécute une recherche lexicale dans la collection"""
all_results = []
for query_config in queries:
try:
mongo_query = LexicalSearchEngine._build_mongo_query(query_config)
results = list(collection.find(mongo_query).limit(limit))
for doc in results:
lexical_score = LexicalSearchEngine._calculate_lexical_score(doc, query_config)
doc["_lexical_score"] = lexical_score
doc["_lexical_query"] = query_config
doc["_search_method"] = "lexical_search"
found = False
for existing in all_results:
if existing.get("_id") == doc.get("_id"):
found = True
if lexical_score > existing.get("_lexical_score", 0):
existing.update(doc)
break
if not found:
all_results.append(doc)
except Exception as e:
logger.error(f"Erreur recherche lexicale: {e}")
continue
all_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True)
return all_results[:limit]
@staticmethod
def _build_mongo_query(query_config: Dict) -> Dict:
"""Construit une requête MongoDB pour la recherche lexicale"""
language = query_config.get("language", "fr")
keywords = query_config.get("keywords", [])
search_fields = query_config.get("search_fields", [])
code_filter = query_config.get("code_filter")
if not keywords or not search_fields:
return {}
or_conditions = []
for keyword in keywords:
if not keyword or len(keyword) < 2:
continue
escaped_keyword = re.escape(keyword)
for field in search_fields:
field_parts = field.split('.')
if len(field_parts) > 1:
nested_field = field_parts[0]
nested_subfield = field_parts[1]
condition = {
f"{nested_field}.{nested_subfield}": {
"$regex": escaped_keyword,
"$options": "i"
}
}
else:
condition = {
field: {
"$regex": escaped_keyword,
"$options": "i"
}
}
or_conditions.append(condition)
if not or_conditions:
return {}
mongo_query = {"$or": or_conditions}
if code_filter:
code_names = Config.CODE_NAMES.get(code_filter, {})
if code_names:
code_fr = code_names.get("fr", "")
code_ar = code_names.get("ar", "")
code_conditions = []
if code_fr:
code_conditions.append({"code_fr": {"$regex": code_fr, "$options": "i"}})
if code_ar:
code_conditions.append({"code_ar": {"$regex": code_ar, "$options": "i"}})
if code_conditions:
mongo_query["$and"] = [{"$or": code_conditions}]
return mongo_query
@staticmethod
def _calculate_lexical_score(doc: Dict, query_config: Dict) -> float:
"""Calcule un score lexical basé sur la pertinence"""
keywords = query_config.get("keywords", [])
search_fields = query_config.get("search_fields", [])
boost_fields = query_config.get("boost_fields", {})
if not keywords:
return 0.0
total_score = 0.0
keyword_count = 0
for keyword in keywords:
keyword_lower = keyword.lower()
keyword_score = 0.0
for field in search_fields:
field_value = LexicalSearchEngine._get_field_value(doc, field)
if not field_value:
continue
if keyword_lower in field_value.lower():
base_score = 1.0
boost = boost_fields.get(field, 1.0)
if re.search(rf'\b{re.escape(keyword_lower)}\b', field_value.lower()):
base_score *= 1.5
keyword_score = max(keyword_score, base_score * boost)
if keyword_score > 0:
total_score += keyword_score
keyword_count += 1
if keyword_count == 0:
return 0.0
average_score = total_score / keyword_count
coverage_bonus = keyword_count / len(keywords) * 0.5
tags_fr = doc.get("tags_fr", [])
tags_ar = doc.get("tags_ar", [])
all_tags = tags_fr + tags_ar
tag_bonus = 0.0
for tag in all_tags:
for keyword in keywords:
if keyword.lower() in tag.lower():
tag_bonus += 0.2
final_score = min(1.0, average_score + coverage_bonus + tag_bonus)
return final_score
@staticmethod
def _get_field_value(doc: Dict, field_path: str) -> str:
"""Obtient la valeur d'un champ, gère les champs imbriqués"""
if not field_path:
return ""
parts = field_path.split('.')
current = doc
for part in parts:
if isinstance(current, dict):
current = current.get(part, {})
else:
return ""
if isinstance(current, str):
return current
elif isinstance(current, list):
return " ".join([str(item) for item in current])
elif current:
return str(current)
return ""
# ============================================================
# ADVANCED SEARCH ENGINES (BM25, TF-IDF, EXACT MATCH)
# ============================================================
class BM25SearchEngine:
"""Moteur de recherche BM25 avancé"""
def __init__(self):
self.k1 = 1.5
self.b = 0.75
self.avgdl = 0
self.doc_freqs = {}
self.idf = {}
self.doc_lengths = []
self.corpus_size = 0
self.corpus = []
self.doc_ids = []
self.fields_weights = {
"resume_ar": 2.0, "resume_fr": 2.0,
"faits_ar": 1.5, "faits_fr": 1.5,
"decision_ar": 1.8, "decision_fr": 1.8,
"tags_ar": 3.0, "tags_fr": 3.0,
"text_to_vector_ar.principe": 2.5,
"text_to_vector_fr.principe": 2.5,
"code_ar": 1.2, "code_fr": 1.2
}
def preprocess_text(self, text: str, language: str = "ar") -> List[str]:
"""Prétraitement avancé du texte"""
if not text:
return []
text = text.lower()
text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s]', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
if language == "ar":
tokens = re.findall(r'[\u0600-\u06FF]+', text)
arabic_stopwords = {
'في', 'من', 'إلى', 'على', 'أن', 'إن', 'ما', 'هو', 'هي', 'كان',
'يكون', 'كانت', 'ليس', 'لا', 'ولكن', 'أو', 'و', 'لكن', 'إذا',
'ذلك', 'هذا', 'هذه', 'تلك', 'التي', 'الذي', 'الذين', 'قد', 'حيث',
'عن', 'مع', 'بين', 'فيما', 'كل', 'بعض', 'أي', 'كل', 'مادة', 'فصل',
'المادة', 'الفصل', 'قانون', 'القانون', 'مجلة', 'المجلة'
}
tokens = [token for token in tokens if token not in arabic_stopwords and len(token) > 2]
else:
tokens = re.findall(r'\b\w+\b', text)
french_stopwords = {
'le', 'la', 'les', 'de', 'des', 'du', 'et', 'est', 'une', 'un',
'dans', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'article',
'articles', 'code', 'loi', 'droit', 'juridique', 'tribunal',
'cour', 'jugement', 'décision', 'affaire', 'procédure'
}
tokens = [token for token in tokens if token not in french_stopwords and len(token) > 2]
return tokens
def build_index(self, documents: List[Dict], language: str = "ar"):
"""Construit l'index BM25"""
self.corpus = []
self.doc_ids = []
self.doc_lengths = []
logger.info(f"🔨 Construction index BM25 pour {len(documents)} documents...")
for doc in documents:
doc_id = str(doc.get("_id", ""))
if not doc_id:
continue
full_text_tokens = []
for field, weight in self.fields_weights.items():
field_parts = field.split('.')
field_value = doc
for part in field_parts:
if isinstance(field_value, dict):
field_value = field_value.get(part, "")
else:
field_value = ""
break
if field_value:
if isinstance(field_value, list):
field_value = " ".join(field_value)
tokens = self.preprocess_text(str(field_value), language)
for _ in range(int(weight)):
full_text_tokens.extend(tokens)
if full_text_tokens:
self.corpus.append(full_text_tokens)
self.doc_ids.append(doc_id)
self.doc_lengths.append(len(full_text_tokens))
if not self.corpus:
logger.warning("⚠️ Aucun document indexé pour BM25")
return
self.corpus_size = len(self.corpus)
self.avgdl = sum(self.doc_lengths) / self.corpus_size
self.doc_freqs = {}
for doc_tokens in self.corpus:
unique_tokens = set(doc_tokens)
for token in unique_tokens:
self.doc_freqs[token] = self.doc_freqs.get(token, 0) + 1
self.idf = {}
for token, freq in self.doc_freqs.items():
self.idf[token] = math.log((self.corpus_size - freq + 0.5) / (freq + 0.5) + 1)
logger.info(f"✅ Index BM25 construit: {self.corpus_size} documents, {len(self.idf)} tokens uniques")
def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]:
"""Recherche BM25 avec la requête"""
if not self.corpus:
return []
query_tokens = self.preprocess_text(query, language)
if not query_tokens:
return []
scores = np.zeros(self.corpus_size)
for i, doc_tokens in enumerate(self.corpus):
doc_len = self.doc_lengths[i]
for token in query_tokens:
if token in self.idf:
f = doc_tokens.count(token)
idf_score = self.idf[token]
numerator = f * (self.k1 + 1)
denominator = f + self.k1 * (1 - self.b + self.b * doc_len / self.avgdl)
scores[i] += idf_score * numerator / denominator if denominator != 0 else 0
sorted_indices = np.argsort(scores)[::-1][:top_k]
results = []
for idx in sorted_indices:
if scores[idx] > 0:
results.append((self.doc_ids[idx], float(scores[idx])))
logger.info(f"🔍 Recherche BM25: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})")
return results
class TFIDFSearchEngine:
"""Moteur de recherche TF-IDF"""
def __init__(self):
self.vocabulary = {}
self.idf = {}
self.tfidf_matrix = None
self.doc_ids = []
def build_index(self, documents: List[Dict], language: str = "ar"):
"""Construit l'index TF-IDF"""
self.doc_ids = []
all_docs_tokens = []
logger.info(f"🔨 Construction index TF-IDF pour {len(documents)} documents...")
for doc in documents:
doc_id = str(doc.get("_id", ""))
if not doc_id:
continue
text_parts = []
priority_fields = [
"resume_ar", "resume_fr",
"text_to_vector_ar.principe", "text_to_vector_fr.principe",
"tags_ar", "tags_fr", "faits_ar", "faits_fr"
]
for field in priority_fields:
field_parts = field.split('.')
field_value = doc
for part in field_parts:
if isinstance(field_value, dict):
field_value = field_value.get(part, "")
else:
field_value = ""
break
if field_value:
if isinstance(field_value, list):
field_value = " ".join(field_value)
text_parts.append(str(field_value))
full_text = " ".join(text_parts)
if language == "ar":
tokens = re.findall(r'[\u0600-\u06FF]{3,}', full_text.lower())
else:
tokens = re.findall(r'\b\w{3,}\b', full_text.lower())
if tokens:
all_docs_tokens.append(tokens)
self.doc_ids.append(doc_id)
if not all_docs_tokens:
return
all_tokens = set()
for doc_tokens in all_docs_tokens:
all_tokens.update(doc_tokens)
self.vocabulary = {token: idx for idx, token in enumerate(sorted(all_tokens))}
tf_matrix = np.zeros((len(all_docs_tokens), len(self.vocabulary)))
for i, doc_tokens in enumerate(all_docs_tokens):
token_counts = Counter(doc_tokens)
total_tokens = len(doc_tokens)
for token, count in token_counts.items():
if token in self.vocabulary:
idx = self.vocabulary[token]
tf_matrix[i, idx] = count / total_tokens if total_tokens > 0 else 0
doc_count = len(all_docs_tokens)
df = np.sum(tf_matrix > 0, axis=0)
self.idf = np.log((doc_count + 1) / (df + 1)) + 1
self.tfidf_matrix = tf_matrix * self.idf
logger.info(f"✅ Index TF-IDF construit: {len(self.doc_ids)} documents, {len(self.vocabulary)} tokens")
def search(self, query: str, language: str = "ar", top_k: int = 100) -> List[Tuple[str, float]]:
"""Recherche TF-IDF"""
if self.tfidf_matrix is None or not self.doc_ids:
return []
if language == "ar":
query_tokens = re.findall(r'[\u0600-\u06FF]{3,}', query.lower())
else:
query_tokens = re.findall(r'\b\w{3,}\b', query.lower())
if not query_tokens:
return []
query_vector = np.zeros(len(self.vocabulary))
query_counts = Counter(query_tokens)
total_tokens = len(query_tokens)
for token, count in query_counts.items():
if token in self.vocabulary:
idx = self.vocabulary[token]
query_vector[idx] = count / total_tokens if total_tokens > 0 else 0
query_vector = query_vector * self.idf
norm_docs = np.linalg.norm(self.tfidf_matrix, axis=1, keepdims=True)
norm_query = np.linalg.norm(query_vector)
similarities = np.dot(self.tfidf_matrix, query_vector) / (norm_docs.flatten() * norm_query + 1e-8)
sorted_indices = np.argsort(similarities)[::-1][:top_k]
results = []
for idx in sorted_indices:
if similarities[idx] > 0:
results.append((self.doc_ids[idx], float(similarities[idx])))
logger.info(f"🔍 Recherche TF-IDF: {len(results)} résultats (score max: {max([r[1] for r in results]) if results else 0:.3f})")
return results
class ExactMatchSearchEngine:
"""Moteur de recherche par correspondance exacte"""
def __init__(self):
self.inverted_index = {}
self.doc_metadata = {}
def build_index(self, documents: List[Dict], language: str = "ar"):
"""Construit un index inversé pour la recherche exacte"""
self.inverted_index = {}
self.doc_metadata = {}
logger.info(f"🔨 Construction index exact pour {len(documents)} documents...")
legal_keywords = {
"ar": [
"كراء تجاري", "تجديد الكراء", "مؤسسات التعليم الخاص",
"تعليم خاص", "الأكرية التجارية", "تأجير", "عقد كراء",
"مدة الكراء", "حق التجديد", "المحلات التجارية",
"المادة 532", "الفصل 532", "مجلة الالتزامات والعقود",
"محكمة التعقيب", "المحكمة التجارية", "عقود التسويغ"
],
"fr": [
"bail commercial", "renouvellement bail", "institutions enseignement privé",
"enseignement privé", "baux commerciaux", "location", "contrat bail",
"durée bail", "droit renouvellement", "locaux commerciaux",
"article 532", "code obligations contrats", "cour cassation",
"tribunal commerce", "contrats location"
]
}
keywords = legal_keywords.get(language, [])
for doc in documents:
doc_id = str(doc.get("_id", ""))
if not doc_id:
continue
self.doc_metadata[doc_id] = doc
text_fields = []
if language == "ar":
fields_to_check = ["resume_ar", "faits_ar", "decision_ar", "text_to_vector_ar.principe"]
else:
fields_to_check = ["resume_fr", "faits_fr", "decision_fr", "text_to_vector_fr.principe"]
for field in fields_to_check:
field_parts = field.split('.')
field_value = doc
for part in field_parts:
if isinstance(field_value, dict):
field_value = field_value.get(part, "")
else:
field_value = ""
break
if field_value:
if isinstance(field_value, list):
field_value = " ".join(field_value)
text_fields.append(str(field_value).lower())
full_text = " ".join(text_fields)
for keyword in keywords:
if keyword.lower() in full_text:
if keyword not in self.inverted_index:
self.inverted_index[keyword] = []
occurrences = full_text.count(keyword.lower())
score = occurrences * 2.0
if field_parts[0] in ["resume", "text_to_vector"]:
score += 1.5
self.inverted_index[keyword].append((doc_id, score))
logger.info(f"✅ Index exact construit: {len(self.inverted_index)} mots-clés indexés")
def search(self, query: str, language: str = "ar", top_k: int = 50) -> List[Tuple[str, float]]:
"""Recherche par correspondance exacte"""
if not self.inverted_index:
return []
query_lower = query.lower()
found_keywords = []
for keyword in self.inverted_index.keys():
if keyword.lower() in query_lower:
found_keywords.append(keyword)
if not found_keywords:
return []
doc_scores = {}
for keyword in found_keywords:
for doc_id, score in self.inverted_index.get(keyword, []):
if doc_id not in doc_scores:
doc_scores[doc_id] = 0
doc_scores[doc_id] += score
sorted_docs = sorted(doc_scores.items(), key=lambda x: x[1], reverse=True)[:top_k]
results = [(doc_id, score) for doc_id, score in sorted_docs if score > 0]
logger.info(f"🔍 Recherche exacte: {len(results)} résultats (mots-clés trouvés: {found_keywords})")
return results
class HybridJurisprudenceSearch:
"""Recherche hybride de jurisprudence avec multiples méthodes"""
def __init__(self, db_manager):
self.db_manager = db_manager
self.bm25_searcher = BM25SearchEngine()
self.tfidf_searcher = TFIDFSearchEngine()
self.exact_searcher = ExactMatchSearchEngine()
self.cached_docs = {}
def load_jurisprudence_docs(self, language: str = "ar", limit: int = 2000) -> List[Dict]:
"""Charge les documents de jurisprudence depuis MongoDB"""
cache_key = f"juris_docs_{language}"
if cache_key in self.cached_docs:
cached_time, docs = self.cached_docs[cache_key]
if datetime.now().timestamp() - cached_time < 300:
logger.info(f"📦 Utilisation du cache pour {len(docs)} documents")
return docs
logger.info(f"📥 Chargement des documents de jurisprudence ({language})...")
query = {}
if language == "ar":
query = {"resume_ar": {"$exists": True, "$ne": ""}}
else:
query = {"resume_fr": {"$exists": True, "$ne": ""}}
docs = list(self.db_manager.juris_collection.find(query).limit(limit))
self.cached_docs[cache_key] = (datetime.now().timestamp(), docs)
logger.info(f"✅ {len(docs)} documents chargés")
return docs
def search_hybrid(self, query: str, query_analysis: QueryAnalysis,
limit: int = 150) -> List[Dict]:
"""Recherche hybride avec multiples méthodes"""
language = query_analysis.search_language
docs = self.load_jurisprudence_docs(language, limit=2000)
if not docs:
logger.warning("⚠️ Aucun document de jurisprudence disponible")
return []
logger.info("🏗️ Construction des index de recherche...")
self.bm25_searcher.build_index(docs, language)
self.tfidf_searcher.build_index(docs, language)
self.exact_searcher.build_index(docs, language)
logger.info("🔍 Exécution des recherches hybrides...")
bm25_results = self.bm25_searcher.search(query, language, top_k=limit)
tfidf_results = self.tfidf_searcher.search(query, language, top_k=limit)
exact_results = self.exact_searcher.search(query, language, top_k=limit)
article_results = self._search_by_articles(query_analysis.cited_articles, docs)
all_results = self._merge_results(
bm25_results, tfidf_results, exact_results, article_results,
docs, limit
)
logger.info(f"✅ Recherche hybride: {len(all_results)} résultats combinés")
return all_results
def _search_by_articles(self, articles: List[str], docs: List[Dict]) -> List[Tuple[str, float]]:
"""Recherche basée sur les articles cités"""
if not articles:
return []
results = []
article_set = set(articles)
for doc in docs:
doc_id = str(doc.get("_id", ""))
cited_articles = doc.get("articles_cites", [])
score = 0
for cited in cited_articles:
article_num = cited.get("article_num", "")
if article_num in article_set:
score += 3.0
if score > 0:
results.append((doc_id, score))
return results
def _merge_results(self, bm25_results: List, tfidf_results: List,
exact_results: List, article_results: List,
docs: List[Dict], limit: int) -> List[Dict]:
"""Fusionne les résultats de toutes les méthodes"""
doc_map = {str(doc.get("_id", "")): doc for doc in docs}
combined_scores = {}
method_weights = {
"bm25": 0.3,
"tfidf": 0.25,
"exact": 0.3,
"articles": 0.15
}
for doc_id, score in bm25_results:
if doc_id not in combined_scores:
combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0}
combined_scores[doc_id]["bm25"] = score
for doc_id, score in tfidf_results:
if doc_id not in combined_scores:
combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0}
combined_scores[doc_id]["tfidf"] = score
for doc_id, score in exact_results:
if doc_id not in combined_scores:
combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0}
combined_scores[doc_id]["exact"] = score
for doc_id, score in article_results:
if doc_id not in combined_scores:
combined_scores[doc_id] = {"bm25": 0, "tfidf": 0, "exact": 0, "articles": 0}
combined_scores[doc_id]["articles"] = score
final_results = []
for doc_id, scores in combined_scores.items():
if doc_id in doc_map:
final_score = (
scores["bm25"] * method_weights["bm25"] +
scores["tfidf"] * method_weights["tfidf"] +
scores["exact"] * method_weights["exact"] +
scores["articles"] * method_weights["articles"]
)
if final_score > 0:
doc = doc_map[doc_id].copy()
doc["_hybrid_score"] = final_score
doc["_score_details"] = scores
final_results.append(doc)
final_results.sort(key=lambda x: x.get("_hybrid_score", 0), reverse=True)
return final_results[:limit]
# ============================================================
# LEGAL CODE CLASSIFIER PROFESSIONNEL
# ============================================================
class LegalCodeClassifier:
@staticmethod
def classify_query(query: str, language: str) -> Dict[str, Any]:
"""Classification avancée des codes juridiques tunisiens"""
classification_prompt = f"""Analyze this Tunisian legal query to identify relevant codes from ALL 10 Tunisian codes.
TUNISIAN LEGAL CODES AVAILABLE:
1. CSP (Code du Statut Personnel) - Family law, marriage, divorce, inheritance, personal status
2. DROITS_REELS (Code des Droits Réels) - Property law, real estate, ownership, mortgages, real rights
3. OBLIGATIONS_CONTRATS (Code des Obligations et des Contrats) - Contract law, obligations, torts, civil liability
4. PROCEDURE_CIVILE (Code de Procédure Civile et Commerciale) - Civil procedure, appeals, judicial process
5. CODE_TRAVAIL (Code du Travail) - Labor law, employment contracts, termination, labor disputes
6. DROITS_PROCEDURES_FISCAUX (Code des Droits et Procédures Fiscaux) - Tax law, fiscal procedures, tax disputes
7. DROIT_INTERNATIONAL_PRIVE (Code de Droit International Privé) - Private international law, conflicts of law
8. CODE_PENAL (Code Pénal) - Criminal law, offenses, penalties, criminal procedure
9. CODE_COMMERCE (Code de Commerce) - Commercial law, companies, bankruptcy, commercial contracts
10. PROCEDURES_PENALES (Code des Procédures Pénales) - Criminal procedure, investigation, prosecution
QUERY: {query}
LANGUAGE: {language}
ANALYSIS INSTRUCTIONS:
1. Identify the PRIMARY code (most relevant)
2. Identify SECONDARY codes (relevant but less direct)
3. Provide confidence score (0.0-1.0)
4. Extract key legal concepts
5. DO NOT extract article numbers here
OUTPUT ONLY VALID JSON:
{{
"primary_code": "CODE_TRAVAIL",
"secondary_codes": ["PROCEDURE_CIVILE", "OBLIGATIONS_CONTRATS"],
"confidence": 0.92,
"key_concepts": ["labor contract", "termination", "CDD renewal"],
"reasoning": "Query focuses on employment contract renewal and termination issues",
"priority_factors": ["employment", "contract", "termination", "labor rights"]
}}"""
try:
response = chat_client.chat.completions.create(
model=Config.CHAT_MODEL,
messages=[
{"role": "system", "content": "Expert Tunisian legal code classifier. Analyze queries and identify relevant codes accurately. Output ONLY valid JSON."},
{"role": "user", "content": classification_prompt}
],
temperature=Config.TEMP_CLASSIFICATION,
max_tokens=800
)
result = response.choices[0].message.content.strip()
result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE)
classification = json.loads(result)
primary_code = LegalCodeClassifier._validate_and_normalize_code(classification["primary_code"])
secondary_codes = [LegalCodeClassifier._validate_and_normalize_code(code) for code in classification.get("secondary_codes", [])]
logger.info(f"✅ Classification réussie: {primary_code.value} (confiance: {classification['confidence']:.2f})")
return {
"primary_code": primary_code,
"secondary_codes": secondary_codes,
"confidence": float(classification.get("confidence", 0.5)),
"key_concepts": classification.get("key_concepts", []),
"reasoning": classification.get("reasoning", ""),
"priority_factors": classification.get("priority_factors", [])
}
except Exception as e:
logger.error(f"Erreur classification GPT: {e}")
return LegalCodeClassifier.fallback_classification(query, language)
@staticmethod
def _validate_and_normalize_code(code_str: str) -> LegalCode:
"""Valide et normalise un code"""
try:
normalized = code_str.upper().strip()
variations = {
"TRAVAIL": "CODE_TRAVAIL",
"CODE TRAVAIL": "CODE_TRAVAIL",
"CODETRAVAIL": "CODE_TRAVAIL",
"DROIT_REEL": "DROITS_REELS",
"DROIT REEL": "DROITS_REELS",
"DROITS_REEL": "DROITS_REELS",
"PENAL": "CODE_PENAL",
"CODEPENAL": "CODE_PENAL",
"COMMERCE": "CODE_COMMERCE",
"CODECOMMERCE": "CODE_COMMERCE",
"PROCEDURE_PENALE": "PROCEDURES_PENALES",
"PROCEDURE PENALE": "PROCEDURES_PENALES"
}
if normalized in variations:
normalized = variations[normalized]
return LegalCode.from_string(normalized)
except:
logger.warning(f"Code non reconnu: {code_str}, utilisation CSP par défaut")
return LegalCode.CSP
@staticmethod
def fallback_classification(query: str, language: str) -> Dict[str, Any]:
"""Classification par mots-clés comme fallback"""
query_lower = query.lower()
scores = {code: 0 for code in LegalCode}
for code in LegalCode:
if code.value in Config.CODE_NAMES:
keywords = Config.CODE_NAMES[code.value].get("keywords", [])
for keyword in keywords:
if keyword.lower() in query_lower:
scores[code] += 2
special_rules = [
(["contrat", "travail", "licenciement"], LegalCode.CODE_TRAVAIL, 3),
(["contrat", "travail", "salaire"], LegalCode.CODE_TRAVAIL, 2),
(["propriété", "succession", "héritage"], LegalCode.CSP, 2),
(["propriété", "hypothèque", "immobilier"], LegalCode.DROITS_REELS, 2),
(["contrat", "obligation", "responsabilité"], LegalCode.OBLIGATIONS_CONTRATS, 2),
(["procédure", "appel", "jugement"], LegalCode.PROCEDURE_CIVILE, 2),
(["fiscal", "impôt", "taxe"], LegalCode.DROITS_PROCEDURES_FISCAUX, 2),
(["international", "étranger", "nationalité"], LegalCode.DROIT_INTERNATIONAL_PRIVE, 2),
(["crime", "peine", "prison"], LegalCode.CODE_PENAL, 2),
(["commerce", "société", "faillite"], LegalCode.CODE_COMMERCE, 2),
(["enquête", "instruction", "procédure pénale"], LegalCode.PROCEDURES_PENALES, 2)
]
for keywords, code, bonus in special_rules:
if all(keyword in query_lower for keyword in keywords):
scores[code] += bonus
sorted_codes = sorted(scores.items(), key=lambda x: x[1], reverse=True)
primary_code = sorted_codes[0][0] if sorted_codes else LegalCode.CSP
secondary_codes = []
for code, score in sorted_codes[1:]:
if score > 0 and len(secondary_codes) < 3:
secondary_codes.append(code)
total_score = sum(scores.values())
confidence = min(scores[primary_code] / max(total_score, 1) * 1.2, 0.85)
logger.info(f"⚠️ Classification fallback: {primary_code.value} (confiance: {confidence:.2f})")
return {
"primary_code": primary_code,
"secondary_codes": secondary_codes,
"confidence": confidence,
"key_concepts": [],
"reasoning": "Fallback keyword-based classification",
"priority_factors": []
}
# ============================================================
# EMBEDDING SERVICE PROFESSIONNEL
# ============================================================
class EnterpriseEmbeddingService:
_cache = {}
_failed_embeddings = set()
@staticmethod
def get_embedding(text: str) -> Optional[List[float]]:
cache_key = hashlib.md5(text.encode()).hexdigest()
if cache_key in EnterpriseEmbeddingService._cache:
return EnterpriseEmbeddingService._cache[cache_key]
if cache_key in EnterpriseEmbeddingService._failed_embeddings:
return None
try:
cleaned_text = EnterpriseEmbeddingService.preprocess_text(text)
if not cleaned_text or len(cleaned_text) < 3:
return None
response = embedding_client.embeddings.create(
input=cleaned_text,
model=Config.EMBEDDING_MODEL
)
embedding = response.data[0].embedding
EnterpriseEmbeddingService._cache[cache_key] = embedding
return embedding
except Exception as e:
logger.error(f"Erreur génération embedding: {e}")
EnterpriseEmbeddingService._failed_embeddings.add(cache_key)
return None
@staticmethod
def preprocess_text(text: str) -> str:
"""Prétraitement avancé du texte"""
if not text:
return ""
text = re.sub(r'\s+', ' ', text)
text = re.sub(r'[^\w\u0600-\u06FF\u00C0-\u017F\s.,;:!?()\[\]-]', ' ', text)
text = text.strip()
if len(text) > 8000:
text = text[:8000]
return text
@staticmethod
def calculate_similarity(emb1: Optional[List[float]], emb2: Optional[List[float]]) -> float:
if not emb1 or not emb2:
return 0.0
try:
arr1 = np.array(emb1).reshape(1, -1)
arr2 = np.array(emb2).reshape(1, -1)
similarity = cosine_similarity(arr1, arr2)[0][0]
return max(0.0, min(1.0, similarity))
except Exception as e:
logger.error(f"Erreur calcul similarité: {e}")
return 0.0
# ============================================================
# QUERY ANALYZER PROFESSIONNEL
# ============================================================
class EnterpriseQueryAnalyzer:
@staticmethod
def analyze_query(query: str) -> QueryAnalysis:
"""Analyse complète de la requête"""
translation_result = TranslationService.detect_and_translate(query)
original_query = translation_result["original_query"]
translated_query = translation_result["translated_query"]
original_language = translation_result["original_language"]
search_language = translation_result["search_language"]
logger.info(f"🌐 Langue détectée: {original_language}")
if original_language == "ar":
logger.info(f"📝 Requête traduite pour recherche")
classification = LegalCodeClassifier.classify_query(translated_query, search_language)
cited_articles_original = ArticleNumberNormalizer.extract_all_numbers(original_query)
cited_articles_translated = ArticleNumberNormalizer.extract_all_numbers(translated_query)
all_articles = cited_articles_original.union(cited_articles_translated)
analysis_prompt = f"""Analyze this Tunisian legal query in detail.
PRIMARY CODE IDENTIFIED: {classification['primary_code'].value}
QUERY: {translated_query}
LANGUAGE: {search_language}
CITED ARTICLES DETECTED: {list(all_articles)}
ANALYSIS TASKS:
1. Validate and filter article numbers (keep only valid Tunisian law article references)
2. Extract key legal topics
3. Identify relevant legal entities
4. Determine question type and complexity
5. Generate search queries for semantic search
6. Identify cross-code implications
RULES FOR ARTICLE EXTRACTION:
- Keep only valid article numbers (e.g., "123", "45 bis")
- Remove years, dates, page numbers, etc.
- If query mentions articles by topic without numbers, don't list them
OUTPUT ONLY VALID JSON:
{{
"validated_articles": [],
"topics": ["topic1", "topic2"],
"legal_entities": ["entity1", "entity2"],
"question_type": "substantive/procedural/hybrid",
"complexity_score": 0.85,
"requires_statutory_law": true,
"requires_procedural_law": false,
"requires_commercial_law": false,
"requires_criminal_law": false,
"requires_labor_law": true,
"requires_tax_law": false,
"requires_international_law": false,
"requires_family_law": false,
"requires_property_law": false,
"search_queries": ["query1", "query2"],
"cross_code_references": []
}}"""
try:
response = chat_client.chat.completions.create(
model=Config.CHAT_MODEL,
messages=[
{"role": "system", "content": "Advanced Tunisian legal query analyzer. Focus on accuracy and precision. Output ONLY valid JSON."},
{"role": "user", "content": analysis_prompt}
],
temperature=Config.TEMP_ANALYSIS,
max_tokens=800
)
result = response.choices[0].message.content.strip()
result = re.sub(r'^```json\s*|\s*```$', '', result, flags=re.MULTILINE)
analysis_data = json.loads(result)
validated_articles = []
for art in analysis_data.get('validated_articles', []):
if ArticleNumberNormalizer.is_valid_article_number(str(art)):
validated_articles.append(str(art))
else:
logger.warning(f"Article ignoré (invalide): {art}")
final_articles = list(set(validated_articles + list(all_articles)))
primary_code = classification['primary_code']
secondary_codes = classification['secondary_codes'].copy()
code_mappings = {
'requires_procedural_law': LegalCode.PROCEDURE_CIVILE,
'requires_commercial_law': LegalCode.CODE_COMMERCE,
'requires_criminal_law': LegalCode.CODE_PENAL,
'requires_labor_law': LegalCode.CODE_TRAVAIL,
'requires_tax_law': LegalCode.DROITS_PROCEDURES_FISCAUX,
'requires_international_law': LegalCode.DROIT_INTERNATIONAL_PRIVE,
'requires_family_law': LegalCode.CSP,
'requires_property_law': LegalCode.DROITS_REELS
}
for key, code in code_mappings.items():
if analysis_data.get(key, False) and code not in secondary_codes and code != primary_code:
secondary_codes.append(code)
if not secondary_codes and primary_code.value in Config.SECONDARY_CODE_MAPPING:
default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value]
secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]]
secondary_codes = list(dict.fromkeys(secondary_codes))[:5]
logger.info(f"📊 Classification: {primary_code.value}")
logger.info(f"📋 Codes secondaires: {[c.value for c in secondary_codes]}")
logger.info(f"📄 Articles cités: {final_articles}")
return QueryAnalysis(
original_query=original_query,
translated_query=translated_query,
original_language=original_language,
search_language=search_language,
extracted_topics=analysis_data.get('topics', []),
legal_entities=analysis_data.get('legal_entities', []),
cited_articles=final_articles,
primary_legal_code=primary_code,
secondary_codes=secondary_codes,
question_type=analysis_data.get('question_type', 'substantive'),
complexity_score=float(analysis_data.get('complexity_score', 0.5)),
search_queries=analysis_data.get('search_queries', [translated_query]),
requires_statutory_law=analysis_data.get('requires_statutory_law', True),
code_confidence=classification['confidence']
)
except Exception as e:
logger.error(f"Erreur analyse requête: {e}")
return EnterpriseQueryAnalyzer._create_fallback_analysis(
original_query, translated_query, original_language,
search_language, classification, all_articles
)
@staticmethod
def _create_fallback_analysis(original_query, translated_query, original_language,
search_language, classification, all_articles):
"""Créer une analyse de fallback"""
primary_code = classification['primary_code']
if primary_code.value in Config.SECONDARY_CODE_MAPPING:
default_secondary = Config.SECONDARY_CODE_MAPPING[primary_code.value]
secondary_codes = [LegalCode.from_string(code) for code in default_secondary[:3]]
else:
secondary_codes = []
if primary_code != LegalCode.PROCEDURE_CIVILE and LegalCode.PROCEDURE_CIVILE not in secondary_codes:
secondary_codes.append(LegalCode.PROCEDURE_CIVILE)
return QueryAnalysis(
original_query=original_query,
translated_query=translated_query,
original_language=original_language,
search_language=search_language,
extracted_topics=[],
legal_entities=[],
cited_articles=list(all_articles),
primary_legal_code=primary_code,
secondary_codes=secondary_codes,
question_type="substantive",
complexity_score=0.5,
search_queries=[translated_query],
requires_statutory_law=True,
code_confidence=classification['confidence']
)
# ============================================================
# DATABASE MANAGER PROFESSIONNEL
# ============================================================
class EnterpriseDatabaseManager:
def __init__(self):
self.code_db = mongo_client[Config.DB_CODE]
self.juris_collection = mongo_client[Config.DB_JURIS][Config.COL_JURIS]
self._article_cache = {}
self._collection_cache = {}
self._verify_collections()
def _verify_collections(self):
"""Vérifier que toutes les collections existent"""
logger.info("🔍 Vérification des collections...")
available_collections = self.code_db.list_collection_names()
for code_key, collection_name in Config.COLLECTIONS.items():
if collection_name in available_collections:
logger.info(f" ✓ {code_key}: {collection_name}")
else:
logger.warning(f" ✗ {code_key}: {collection_name} - COLLECTION NON TROUVÉE")
def get_collection(self, code_type: LegalCode):
"""Récupère la collection MongoDB pour un type de code"""
if code_type in self._collection_cache:
return self._collection_cache[code_type]
collection_name = Config.COLLECTIONS.get(code_type.value)
if not collection_name:
logger.error(f"❌ Collection non configurée pour: {code_type.value}")
return None
try:
collection = self.code_db[collection_name]
self._collection_cache[code_type] = collection
return collection
except Exception as e:
logger.error(f"❌ Erreur accès collection {collection_name}: {e}")
return None
def get_article_by_number(self, article_number: str, code_type: LegalCode) -> Optional[Dict]:
"""Récupère un article par son numéro"""
if not ArticleNumberNormalizer.is_valid_article_number(article_number):
logger.warning(f"Numéro d'article invalide: {article_number}")
return None
cache_key = f"{code_type.value}_{article_number}"
if cache_key in self._article_cache:
return self._article_cache[cache_key]
collection = self.get_collection(code_type)
if collection is None:
return None
normalized = ArticleNumberNormalizer.normalize(article_number)
search_strategies = [
lambda: collection.find_one({"article_num": normalized}),
lambda: collection.find_one({"article_number": normalized}),
lambda: collection.find_one({"art": normalized}),
lambda: collection.find_one({"$or": [
{"article_text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}},
{"text": {"$regex": f"\\b{normalized}\\b", "$options": "i"}},
{"contenu": {"$regex": f"\\b{normalized}\\b", "$options": "i"}}
]})
]
article = None
for strategy in search_strategies:
article = strategy()
if article:
break
if article:
self._article_cache[cache_key] = article
logger.info(f"✅ Article {normalized} trouvé dans {code_type.value}")
else:
logger.warning(f"❌ Article {normalized} NON TROUVÉ dans {code_type.value}")
return article
def get_multiple_articles(self, article_numbers: List[str], primary_code: LegalCode,
secondary_codes: List[LegalCode] = None, user_language: str = "fr") -> List[RetrievedSource]:
"""Récupère plusieurs articles - MÉTHODE MANQUANTE AJOUTÉE"""
retrieved = []
valid_articles = [art for art in article_numbers if ArticleNumberNormalizer.is_valid_article_number(art)]
if not valid_articles:
logger.info("Aucun numéro d'article valide")
return retrieved
logger.info(f"Recherche de {len(valid_articles)} articles...")
for art_num in valid_articles:
article = self.get_article_by_number(art_num, primary_code)
if article:
retrieved.append(self._create_source_from_article(article, art_num, primary_code, True, user_language))
if secondary_codes:
for art_num in valid_articles:
already_found = any(
source.article_metadata.normalized_number == ArticleNumberNormalizer.normalize(art_num)
for source in retrieved
)
if not already_found:
for code in secondary_codes:
article = self.get_article_by_number(art_num, code)
if article:
retrieved.append(self._create_source_from_article(article, art_num, code, False, user_language))
break
logger.info(f"✅ {len(retrieved)} articles récupérés")
return retrieved
def _create_source_from_article(self, article: Dict, art_num: str, code_type: LegalCode,
primary: bool, user_language: str) -> RetrievedSource:
"""Crée un objet RetrievedSource à partir d'un article"""
article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or ''
article_text_ar = ""
if user_language == "ar" and article_text_fr:
article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr)
metadata = ArticleMetadata(
article_number=article.get('article_num', art_num) or article.get('article_number', art_num),
normalized_number=ArticleNumberNormalizer.normalize(art_num),
article_text_fr=article_text_fr,
article_text_ar=article_text_ar,
code_type=code_type,
code_name_fr=Config.CODE_NAMES[code_type.value]["fr"],
code_name_ar=Config.CODE_NAMES[code_type.value]["ar"],
chapter=article.get('chapter'),
section=article.get('section'),
pdf_source=article.get('pdf_source')
)
if user_language == "ar" and metadata.article_text_ar:
content = metadata.article_text_ar
code_name = metadata.code_name_ar
else:
content = metadata.article_text_fr
code_name = metadata.code_name_fr
full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}"
return RetrievedSource(
source_id=str(article.get('_id', '')),
source_type=SourceType.STATUTE,
article_metadata=metadata,
content=full_content,
relevance=RelevanceScore(
similarity=1.0, topic_overlap=1.0, entity_match=1.0, article_match=1.0,
code_relevance=1.0 if primary else 0.7,
combined_score=1.0 if primary else 0.7,
relevance_level="high", confidence=1.0
),
retrieval_timestamp=datetime.now(),
retrieval_method='direct_lookup',
summary=content[:500] + "..." if len(content) > 500 else content,
full_text=content,
primary_code=primary
)
def search_similar_articles(self, query_embedding: List[float], code_type: LegalCode,
limit: int = 20, user_language: str = "fr") -> List[Dict]:
"""Recherche sémantique d'articles similaires"""
try:
collection = self.get_collection(code_type)
if collection is None:
return []
articles = list(collection.find({"embedding": {"$exists": True}}).limit(200))
if not articles:
return []
for article in articles:
embedding = article.get("embedding")
if embedding:
similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding)
article["_similarity"] = similarity
else:
article["_similarity"] = 0.0
articles.sort(key=lambda x: x.get("_similarity", 0), reverse=True)
return articles[:limit]
except Exception as e:
logger.error(f"Erreur recherche sémantique {code_type.value}: {e}")
return []
def search_jurisprudence_enhanced(self, query_embedding: List[float], language: str,
topics: List[str] = None, primary_code: str = None,
limit: int = 100) -> List[Dict]:
"""Recherche améliorée de jurisprudence avec filtres avancés"""
try:
embedding_field = "embedding_ar" if language == "ar" else "embedding_fr"
sample = self.juris_collection.find_one({embedding_field: {"$exists": True}})
if not sample:
embedding_field = "embedding"
logger.warning(f"Champ {embedding_field} non trouvé, utilisation du champ générique 'embedding'")
base_query = {embedding_field: {"$exists": True}}
if primary_code:
code_name_ar = Config.CODE_NAMES.get(primary_code, {}).get("ar", "")
code_name_fr = Config.CODE_NAMES.get(primary_code, {}).get("fr", "")
base_query["$or"] = [
{"code_ar": {"$regex": code_name_ar, "$options": "i"}},
{"code_fr": {"$regex": code_name_fr, "$options": "i"}},
{"tags_ar": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", []) if tag.isascii() is False]}},
{"tags_fr": {"$in": [tag for tag in Config.CODE_NAMES.get(primary_code, {}).get("keywords", [])]}}
]
docs = list(self.juris_collection.find(base_query).limit(limit * 2))
if not docs:
logger.warning("Aucun document de jurisprudence trouvé")
return []
for doc in docs:
embedding = doc.get(embedding_field)
if embedding:
similarity = EnterpriseEmbeddingService.calculate_similarity(query_embedding, embedding)
bonus = 0.0
if topics:
tags_ar = doc.get("tags_ar", [])
tags_fr = doc.get("tags_fr", [])
all_tags = tags_ar + tags_fr
for topic in topics:
topic_lower = topic.lower()
for tag in all_tags:
if topic_lower in tag.lower() or tag.lower() in topic_lower:
bonus += 0.05
doc["_similarity"] = min(1.0, similarity + bonus)
doc["_search_method"] = "enhanced_semantic_search"
else:
doc["_similarity"] = 0.0
docs.sort(key=lambda x: x.get("_similarity", 0), reverse=True)
filtered_docs = [doc for doc in docs if doc.get("_similarity", 0) >= Config.JURIS_MINIMUM_RELEVANCE]
logger.info(f"Jurisprudence: {len(filtered_docs)} documents après filtrage (sur {len(docs)})")
return filtered_docs[:limit]
except Exception as e:
logger.error(f"Erreur recherche jurisprudence améliorée: {e}")
return []
def search_jurisprudence_by_keywords(self, keywords: List[str], language: str,
primary_code: str = None, limit: int = 40) -> List[Dict]:
"""Recherche de jurisprudence par mots-clés"""
try:
query = {}
if language == "ar":
text_fields = ["resume_ar", "faits_ar", "text_to_vector_ar.principe", "decision_ar"]
tag_field = "tags_ar"
else:
text_fields = ["resume_fr", "faits_fr", "text_to_vector_fr.principe", "decision_fr"]
tag_field = "tags_fr"
keyword_queries = []
for keyword in keywords:
if keyword:
for field in text_fields:
keyword_queries.append({field: {"$regex": keyword, "$options": "i"}})
if keyword_queries:
query["$or"] = keyword_queries
if primary_code:
code_keywords = Config.CODE_NAMES.get(primary_code, {}).get("keywords", [])
if code_keywords:
code_query = {"tags_fr": {"$in": code_keywords}}
if language == "ar":
arabic_keywords = [kw for kw in code_keywords if not kw.isascii()]
if arabic_keywords:
code_query["$or"] = [{"tags_ar": {"$in": arabic_keywords}}]
if "$or" in query:
query["$and"] = [{"$or": query.pop("$or")}, code_query]
else:
query.update(code_query)
docs = list(self.juris_collection.find(query).limit(limit))
for doc in docs:
score = 0.0
for keyword in keywords:
for field in text_fields:
field_value = doc
for part in field.split('.'):
field_value = field_value.get(part, {}) if isinstance(field_value, dict) else ""
if isinstance(field_value, str) and keyword.lower() in field_value.lower():
score += 0.1
tags = doc.get(tag_field, [])
for tag in tags:
for keyword in keywords:
if keyword.lower() in tag.lower():
score += 0.15
doc["_keyword_score"] = min(1.0, score)
doc["_search_method"] = "keyword_search"
docs.sort(key=lambda x: x.get("_keyword_score", 0), reverse=True)
logger.info(f"Jurisprudence par mots-clés: {len(docs)} documents trouvés")
return docs[:limit]
except Exception as e:
logger.error(f"Erreur recherche par mots-clés: {e}")
return []
def search_jurisprudence_lexical(self, query_analysis: QueryAnalysis, limit: int = 100) -> List[Dict]:
"""Recherche lexicale avancée dans la jurisprudence"""
try:
lexical_queries = LexicalSearchEngine.create_lexical_queries(query_analysis)
if not lexical_queries:
logger.warning("Aucune requête lexicale générée")
return []
lexical_results = LexicalSearchEngine.search_lexical(
self.juris_collection,
lexical_queries,
limit=limit
)
logger.info(f"🔍 Recherche lexicale: {len(lexical_results)} documents trouvés")
filtered_results = []
for doc in lexical_results:
lexical_score = doc.get("_lexical_score", 0)
if lexical_score >= 0.25:
code_relevant = False
all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes]
code_fr = doc.get("code_fr", "")
code_ar = doc.get("code_ar", "")
for code in all_codes:
code_info = Config.CODE_NAMES.get(code, {})
if code_info:
if (code_info.get("fr") and code_info["fr"] in code_fr) or \
(code_info.get("ar") and code_info["ar"] in code_ar):
code_relevant = True
break
if code_relevant:
lexical_score = min(1.0, lexical_score + 0.2)
doc["_lexical_score"] = lexical_score
filtered_results.append(doc)
filtered_results.sort(key=lambda x: x.get("_lexical_score", 0), reverse=True)
logger.info(f"✅ Recherche lexicale filtrée: {len(filtered_results)} documents pertinents")
return filtered_results[:limit]
except Exception as e:
logger.error(f"Erreur recherche lexicale: {e}")
return []
def search_jurisprudence_by_all_codes(self, query_analysis: QueryAnalysis, query_embedding: List[float],
limit: int = 150) -> List[Dict]:
"""Recherche de jurisprudence dans tous les codes pertinents"""
all_results = []
all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes]
for code in all_codes:
try:
code_results = self.search_jurisprudence_enhanced(
query_embedding,
query_analysis.search_language,
topics=query_analysis.extracted_topics,
primary_code=code,
limit=limit // len(all_codes)
)
for doc in code_results:
doc_code_fr = doc.get("code_fr", "")
doc_code_ar = doc.get("code_ar", "")
code_info = Config.CODE_NAMES.get(code, {})
if code_info:
code_fr = code_info.get("fr", "")
code_ar = code_info.get("ar", "")
if (code_fr and code_fr in doc_code_fr) or (code_ar and code_ar in doc_code_ar):
doc["_similarity"] = min(1.0, doc.get("_similarity", 0) + 0.15)
doc["_code_filter"] = code
all_results.extend(code_results)
except Exception as e:
logger.error(f"Erreur recherche jurisprudence pour code {code}: {e}")
continue
seen_ids = set()
unique_results = []
for doc in all_results:
doc_id = doc.get("_id")
if doc_id and doc_id not in seen_ids:
seen_ids.add(doc_id)
unique_results.append(doc)
unique_results.sort(key=lambda x: x.get("_similarity", 0), reverse=True)
logger.info(f"🌐 Recherche multi-code: {len(unique_results)} documents uniques")
return unique_results[:limit]
# ============================================================
# ENHANCED SEARCH ENGINE WITH HYBRID SEARCH
# ============================================================
class EnhancedEnterpriseSearchEngine:
def __init__(self):
self.db_manager = EnterpriseDatabaseManager()
self.hybrid_searcher = HybridJurisprudenceSearch(self.db_manager)
def search(self, query_analysis: QueryAnalysis) -> Dict[str, Any]:
"""Exécute une recherche complète avec méthode hybride"""
logger.info(f"🔍 Lancement recherche hybride multi-méthodes")
# 1. Récupération directe des articles cités
direct_sources = self.db_manager.get_multiple_articles(
query_analysis.cited_articles,
query_analysis.primary_legal_code,
query_analysis.secondary_codes,
query_analysis.original_language
)
# 2. Recherche sémantique dans les codes
query_embedding = EnterpriseEmbeddingService.get_embedding(query_analysis.translated_query)
primary_semantic = []
secondary_semantic = []
if query_embedding:
# Recherche sémantique dans le code principal
primary_articles = self.db_manager.search_similar_articles(
query_embedding, query_analysis.primary_legal_code,
limit=30, user_language=query_analysis.original_language
)
primary_semantic = self._process_semantic_results(
primary_articles, query_analysis.primary_legal_code,
True, query_analysis.original_language
)
# Recherche sémantique dans les codes secondaires
for code in query_analysis.secondary_codes:
articles = self.db_manager.search_similar_articles(
query_embedding, code,
limit=Config.MAX_CROSS_CODE_ARTICLES,
user_language=query_analysis.original_language
)
secondary_results = self._process_semantic_results(
articles, code, False, query_analysis.original_language
)
secondary_semantic.extend(secondary_results)
# 3. Recherche hybride de jurisprudence
logger.info(" 🔍 Recherche jurisprudence hybride...")
hybrid_juris_docs = self.hybrid_searcher.search_hybrid(
query_analysis.translated_query,
query_analysis,
limit=Config.MAX_JURIS_RETRIEVAL
)
# 4. Recherches traditionnelles (pour complément)
juris_docs_semantic = []
juris_docs_lexical = []
juris_docs_keywords = []
if query_embedding:
juris_docs_semantic = self.db_manager.search_jurisprudence_by_all_codes(
query_analysis,
query_embedding,
limit=Config.MAX_JURIS_RETRIEVAL // 3
)
juris_docs_lexical = self.db_manager.search_jurisprudence_lexical(
query_analysis,
limit=Config.MAX_JURIS_RETRIEVAL // 3
)
keywords = self._extract_juris_keywords(query_analysis)
juris_docs_keywords = self.db_manager.search_jurisprudence_by_keywords(
keywords,
query_analysis.search_language,
primary_code=query_analysis.primary_legal_code.value,
limit=Config.MAX_JURIS_RETRIEVAL // 3
)
# 5. Fusionner TOUS les résultats de jurisprudence
all_juris_docs = self._merge_all_jurisprudence_results(
hybrid_juris_docs,
juris_docs_semantic,
juris_docs_lexical,
juris_docs_keywords
)
juris_sources = self._process_jurisprudence_results(all_juris_docs, query_analysis.original_language)
# 6. Fusionner et organiser
all_statute_sources = self._merge_sources(direct_sources, primary_semantic, secondary_semantic)
statute_results = self._organize_by_relevance(all_statute_sources, query_analysis.primary_legal_code)
missing_articles = self._identify_missing_articles(query_analysis.cited_articles, all_statute_sources)
retrieval_metadata = {
"primary_code": query_analysis.primary_legal_code.value,
"secondary_codes": [c.value for c in query_analysis.secondary_codes],
"direct_retrievals": len(direct_sources),
"primary_semantic": len(primary_semantic),
"secondary_semantic": len(secondary_semantic),
"total_statutes": len(all_statute_sources),
"total_jurisprudence": len(juris_sources),
"missing_articles": missing_articles,
"search_methods": ["direct", "semantic", "hybrid", "lexical", "keywords", "bm25", "tfidf", "exact"]
}
return {
"statute_sources": statute_results,
"jurisprudence_sources": juris_sources[:Config.MAX_JURIS_DOCS],
"query_analysis": query_analysis,
"retrieval_metadata": retrieval_metadata
}
def _extract_juris_keywords(self, query_analysis: QueryAnalysis) -> List[str]:
"""Extrait les mots-clés pour la recherche de jurisprudence"""
keywords = set()
keywords.update(query_analysis.extracted_topics)
all_codes = [query_analysis.primary_legal_code.value] + [c.value for c in query_analysis.secondary_codes]
for code in all_codes:
if code in Config.CODE_NAMES:
code_keywords = Config.CODE_NAMES[code].get("keywords", [])
keywords.update(code_keywords[:15])
keywords.update([
"طلبات جديدة", "الطلبات الجديدة", "demandes nouvelles", "new claims",
"استئناف", "appel", "appeal", "تعقيب", "cassation",
"طلاق إنشاء", "divorce création", "طلاق الضرر", "divorce préjudice",
"جراية عمرية", "pension viagère", "lifetime pension",
"تعويض", "indemnité", "compensation", "ضرر", "préjudice",
"محكمة البداية", "first instance court", "محكمة الاستئناف", "appeal court",
"الفصل 147", "147", "article 147", "مرفوض شكلاً", "non recevable",
"إجراءات", "procédure", "شكلية", "formalité", "تقاضي", "litigation",
"مبدأ التقاضي", "principe de procès", "درجة التقاضي", "degré de juridiction"
])
cleaned_keywords = []
for kw in keywords:
if kw and len(kw) > 2:
cleaned_keywords.append(kw.strip().lower())
return list(set(cleaned_keywords))[:40]
def _merge_all_jurisprudence_results(self, hybrid_results: List[Dict],
semantic_results: List[Dict],
lexical_results: List[Dict],
keyword_results: List[Dict]) -> List[Dict]:
"""Fusionne tous les types de résultats de jurisprudence"""
seen_ids = set()
merged = []
def add_document(doc, score_field, method):
doc_id = doc.get("_id")
if doc_id and doc_id not in seen_ids:
seen_ids.add(doc_id)
if score_field in doc:
doc["_similarity"] = doc[score_field]
doc["_search_method"] = method
merged.append(doc)
for doc in hybrid_results:
add_document(doc, "_hybrid_score", "hybrid_search")
for doc in semantic_results:
if doc.get("_id") not in seen_ids:
add_document(doc, "_similarity", "semantic_search")
for doc in lexical_results:
if doc.get("_id") not in seen_ids:
add_document(doc, "_lexical_score", "lexical_search")
for doc in keyword_results:
if doc.get("_id") not in seen_ids:
add_document(doc, "_keyword_score", "keyword_search")
merged.sort(key=lambda x: self._calculate_combined_juris_score(x), reverse=True)
logger.info(f"Fusion complète jurisprudence: {len(merged)} documents uniques")
return merged
def _calculate_combined_juris_score(self, doc: Dict) -> float:
"""Calcule un score combiné pour la jurisprudence"""
hybrid_score = doc.get("_hybrid_score", 0.0)
if hybrid_score > 0:
return hybrid_score * 1.2
semantic_score = doc.get("_similarity", 0.0)
lexical_score = doc.get("_lexical_score", 0.0)
keyword_score = doc.get("_keyword_score", 0.0)
method = doc.get("_search_method", "")
if method == "semantic_search":
return semantic_score
elif method == "lexical_search":
return lexical_score * 0.7 + semantic_score * 0.3
elif method == "keyword_search":
return keyword_score * 0.6 + semantic_score * 0.4
else:
return max(semantic_score, lexical_score, keyword_score)
def _process_semantic_results(self, articles: List[Dict], code_type: LegalCode,
primary: bool, user_language: str) -> List[RetrievedSource]:
"""Traite les résultats de recherche sémantique"""
processed = []
thresholds = Config.RELEVANCE_THRESHOLDS.get(code_type.value, {"HIGH": 0.7, "MEDIUM": 0.6, "MINIMUM": 0.5})
for article in articles:
similarity = article.get('_similarity', 0.0)
if similarity >= thresholds["MINIMUM"]:
art_num = article.get('article_num') or article.get('article_number') or article.get('art') or 'N/A'
article_text_fr = article.get('article_text', '') or article.get('text', '') or article.get('contenu', '') or ''
article_text_ar = ""
if user_language == "ar" and article_text_fr:
article_text_ar = TranslationService.translate_fr_to_ar(article_text_fr)
metadata = ArticleMetadata(
article_number=art_num,
normalized_number=ArticleNumberNormalizer.normalize(art_num),
article_text_fr=article_text_fr,
article_text_ar=article_text_ar,
code_type=code_type,
code_name_fr=Config.CODE_NAMES[code_type.value]["fr"],
code_name_ar=Config.CODE_NAMES[code_type.value]["ar"],
chapter=article.get('chapter'),
section=article.get('section'),
pdf_source=article.get('pdf_source')
)
if similarity >= thresholds["HIGH"]:
level = "high"
elif similarity >= thresholds["MEDIUM"]:
level = "medium"
else:
level = "low"
if user_language == "ar" and metadata.article_text_ar:
content = metadata.article_text_ar
code_name = metadata.code_name_ar
else:
content = metadata.article_text_fr
code_name = metadata.code_name_fr
full_content = f"{code_name} - Article {metadata.article_number}\n\n{content}"
code_relevance = 1.0 if primary else 0.7
combined_score = similarity * code_relevance
processed.append(RetrievedSource(
source_id=str(article.get('_id', '')),
source_type=SourceType.STATUTE,
article_metadata=metadata,
content=full_content,
relevance=RelevanceScore(
similarity=similarity,
topic_overlap=0.0,
entity_match=0.0,
article_match=0.0,
code_relevance=code_relevance,
combined_score=combined_score,
relevance_level=level,
confidence=similarity
),
retrieval_timestamp=datetime.now(),
retrieval_method='semantic_search',
summary=content[:500] + "..." if len(content) > 500 else content,
full_text=content,
primary_code=primary
))
return processed
def _merge_sources(self, direct: List[RetrievedSource], primary_semantic: List[RetrievedSource],
secondary_semantic: List[RetrievedSource]) -> List[RetrievedSource]:
"""Fusionne les sources en évitant les doublons"""
seen_articles = set()
merged = []
for source in direct + primary_semantic + secondary_semantic:
if source.article_metadata:
article_id = f"{source.article_metadata.code_type.value}_{source.article_metadata.normalized_number}"
if article_id not in seen_articles:
seen_articles.add(article_id)
merged.append(source)
return merged
def _organize_by_relevance(self, sources: List[RetrievedSource], primary_code: LegalCode) -> Dict[str, List[RetrievedSource]]:
"""Organise les sources par niveau de pertinence"""
organized = {"high": [], "medium": [], "low": []}
for source in sources:
level = source.relevance.relevance_level
if level in organized:
organized[level].append(source)
for level in organized:
organized[level].sort(key=lambda x: (x.primary_code, x.relevance.combined_score), reverse=True)
organized["high"] = organized["high"][:Config.MAX_HIGH_ARTICLES]
organized["medium"] = organized["medium"][:Config.MAX_MEDIUM_ARTICLES]
organized["low"] = organized["low"][:Config.MAX_LOW_ARTICLES]
return organized
def _identify_missing_articles(self, cited: List[str], retrieved: List[RetrievedSource]) -> List[str]:
"""Identifie les articles cités mais non récupérés"""
retrieved_numbers = {
source.article_metadata.normalized_number
for source in retrieved if source.article_metadata
}
missing = []
for article in cited:
normalized = ArticleNumberNormalizer.normalize(article)
if normalized not in retrieved_numbers and ArticleNumberNormalizer.is_valid_article_number(article):
missing.append(article)
return missing
def _process_jurisprudence_results(self, juris_docs: List[Dict], user_language: str) -> List[RetrievedSource]:
"""Traite les résultats de jurisprudence de manière améliorée"""
processed = []
for doc in juris_docs:
similarity = self._calculate_combined_juris_score(doc)
if similarity >= Config.JURIS_HIGH_RELEVANCE:
relevance_level = "high"
elif similarity >= Config.JURIS_MEDIUM_RELEVANCE:
relevance_level = "medium"
elif similarity >= Config.JURIS_MINIMUM_RELEVANCE:
relevance_level = "low"
else:
continue
if user_language == "ar":
resume = doc.get('resume_ar') or doc.get('summary_ar') or doc.get('description_ar') or doc.get('contenu_ar') or ''
faits = doc.get('faits_ar') or ''
decision = doc.get('decision_ar') or ''
code_name = doc.get('code_ar', '')
juridiction = doc.get('juridiction', '')
tags = doc.get('tags_ar', [])
if not resume:
resume_fr = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or ''
if resume_fr:
resume = TranslationService.translate_fr_to_ar(resume_fr)
if not faits:
faits_fr = doc.get('faits_fr') or ''
if faits_fr:
faits = TranslationService.translate_fr_to_ar(faits_fr)
if not decision:
decision_fr = doc.get('decision_fr') or ''
if decision_fr:
decision = TranslationService.translate_fr_to_ar(decision_fr)
else:
resume = doc.get('resume_fr') or doc.get('summary_fr') or doc.get('description_fr') or doc.get('contenu_fr') or ''
faits = doc.get('faits_fr') or ''
decision = doc.get('decision_fr') or ''
code_name = doc.get('code_fr', '')
juridiction = doc.get('juridiction', '')
tags = doc.get('tags_fr', [])
case_number = doc.get('numero_dossier') or doc.get('case_number') or ''
date_decision = doc.get('date') or doc.get('date_jugement') or doc.get('date_decision') or ''
full_content = ""
if user_language == "ar":
full_content += f"رقم القضية: {case_number}\n" if case_number else ""
full_content += f"المحكمة: {juridiction}\n" if juridiction else ""
full_content += f"التاريخ: {date_decision}\n" if date_decision else ""
full_content += f"المجلة: {code_name}\n" if code_name else ""
full_content += f"طريقة البحث: {doc.get('_search_method', '')}\n"
full_content += f"درجة الصلة: {similarity:.3f}\n"
full_content += f"\nالملخص:\n{resume}\n" if resume else ""
full_content += f"\nالوقائع:\n{faits}\n" if faits else ""
full_content += f"\nالقرار:\n{decision}\n" if decision else ""
else:
full_content += f"N° Affaire: {case_number}\n" if case_number else ""
full_content += f"Juridiction: {juridiction}\n" if juridiction else ""
full_content += f"Date: {date_decision}\n" if date_decision else ""
full_content += f"Code: {code_name}\n" if code_name else ""
full_content += f"Méthode de recherche: {doc.get('_search_method', '')}\n"
full_content += f"Score de pertinence: {similarity:.3f}\n"
full_content += f"\nRésumé:\n{resume}\n" if resume else ""
full_content += f"\nFaits:\n{faits}\n" if faits else ""
full_content += f"\nDécision:\n{decision}\n" if decision else ""
summary = resume[:200] + "..." if len(resume) > 200 else resume
processed.append(RetrievedSource(
source_id=str(doc.get('_id', '')),
source_type=SourceType.JURISPRUDENCE,
article_metadata=None,
content=full_content,
relevance=RelevanceScore(
similarity=similarity,
topic_overlap=0.0,
entity_match=0.0,
article_match=0.0,
code_relevance=0.9,
combined_score=similarity,
relevance_level=relevance_level,
confidence=similarity
),
retrieval_timestamp=datetime.now(),
retrieval_method=doc.get('_search_method', 'jurisprudence_search'),
summary=summary,
full_text=full_content,
primary_code=False,
tags=tags[:10],
code_fr=doc.get('code_fr', ''),
code_ar=doc.get('code_ar', ''),
juridiction=juridiction,
date_decision=date_decision
))
processed.sort(key=lambda x: x.relevance.combined_score, reverse=True)
return processed
# ============================================================
# CONTEXT BUILDER PROFESSIONNEL
# ============================================================
class EnterpriseContextBuilder:
@staticmethod
def build_context(search_results: Dict[str, Any]) -> Tuple[str, Dict[str, Any]]:
"""Construit le contexte pour la génération avec améliorations"""
statute_results = search_results["statute_sources"]
juris_sources = search_results["jurisprudence_sources"]
query_analysis = search_results["query_analysis"]
retrieval_meta = search_results["retrieval_metadata"]
context_parts = []
source_registry = {
"statutes": {},
"jurisprudence": {},
"primary_code": query_analysis.primary_legal_code.value,
"cited_articles": query_analysis.cited_articles,
"missing_articles": retrieval_meta["missing_articles"],
"search_methods": retrieval_meta.get("search_methods", [])
}
statutes_by_code = {}
for level in ["high", "medium", "low"]:
for source in statute_results.get(level, []):
if source.article_metadata:
code = source.article_metadata.code_type.value
if code not in statutes_by_code:
statutes_by_code[code] = []
statutes_by_code[code].append(source)
statute_count = 0
primary_code = query_analysis.primary_legal_code.value
if primary_code in statutes_by_code:
context_parts.append(f"\n=== {Config.CODE_NAMES[primary_code]['fr'].upper()} ===")
context_parts.append(f"=== {Config.CODE_NAMES[primary_code]['ar']} ===")
for source in statutes_by_code[primary_code]:
statute_count += 1
art_num = source.article_metadata.article_number
context_parts.append(f"\n[STATUTE_{statute_count}]")
context_parts.append(f"Article: {art_num}")
context_parts.append(f"Code: {primary_code}")
context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}")
context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}")
context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n")
source_registry["statutes"][art_num] = {
"text": source.content,
"full_text": source.full_text if source.full_text else source.content,
"code": primary_code,
"code_name_fr": Config.CODE_NAMES[primary_code]["fr"],
"code_name_ar": Config.CODE_NAMES[primary_code]["ar"],
"relevance": source.relevance.combined_score,
"primary_code": True
}
for code, sources in statutes_by_code.items():
if code != primary_code:
context_parts.append(f"\n=== {Config.CODE_NAMES[code]['fr'].upper()} (Contextual) ===")
context_parts.append(f"=== {Config.CODE_NAMES[code]['ar']} (سياقي) ===")
for source in sources[:Config.MAX_CROSS_CODE_ARTICLES]:
statute_count += 1
art_num = source.article_metadata.article_number
context_parts.append(f"\n[STATUTE_{statute_count}]")
context_parts.append(f"Article: {art_num}")
context_parts.append(f"Code: {code}")
context_parts.append(f"Full Text:\n{source.full_text if source.full_text else source.content}")
context_parts.append(f"Relevance: {source.relevance.combined_score:.3f}")
context_parts.append(f"Primary: {'Yes' if source.primary_code else 'No'}\n")
source_registry["statutes"][art_num] = {
"text": source.content,
"full_text": source.full_text if source.full_text else source.content,
"code": code,
"code_name_fr": Config.CODE_NAMES[code]["fr"],
"code_name_ar": Config.CODE_NAMES[code]["ar"],
"relevance": source.relevance.combined_score,
"primary_code": False
}
if juris_sources:
if query_analysis.original_language == "ar":
context_parts.append("\n=== JURISPRUDENCE (الأحكام القضائية) ===")
else:
context_parts.append("\n=== JURISPRUDENCE (CASE LAW SUPPORT) ===")
juris_by_level = {"high": [], "medium": [], "low": []}
for source in juris_sources:
level = source.relevance.relevance_level
if level in juris_by_level:
juris_by_level[level].append(source)
for level in ["high", "medium", "low"]:
if juris_by_level[level]:
level_display = level.upper()
if query_analysis.original_language == "ar":
level_names = {"high": "عالية", "medium": "متوسطة", "low": "منخفضة"}
context_parts.append(f"\n=== أحكام ذات أهمية {level_names[level]} ===")
else:
context_parts.append(f"\n=== {level_display} RELEVANCE JURISPRUDENCE ===")
for i, source in enumerate(juris_by_level[level], 1):
context_parts.append(f"\n[JURIS_{level.upper()}_{i}]")
if source.juridiction:
context_parts.append(f"Juridiction: {source.juridiction}")
if source.date_decision:
context_parts.append(f"Date: {source.date_decision}")
if source.code_fr or source.code_ar:
code_display = source.code_ar if query_analysis.original_language == "ar" else source.code_fr
context_parts.append(f"Code: {code_display}")
if source.tags:
tags_display = ", ".join(source.tags[:8])
context_parts.append(f"Tags: {tags_display}")
if source.retrieval_method:
context_parts.append(f"Search Method: {source.retrieval_method}")
context_parts.append(f"Full Decision / القرار الكامل:")
context_parts.append(f"{source.full_text if source.full_text else source.content}")
context_parts.append(f"Relevance Score: {source.relevance.combined_score:.3f}\n")
source_registry["jurisprudence"][f"JURIS_{level.upper()}_{i}"] = {
"content": source.content,
"full_text": source.full_text if source.full_text else source.content,
"relevance": source.relevance.combined_score,
"source_id": source.source_id,
"level": level,
"method": source.retrieval_method,
"juridiction": source.juridiction,
"date": source.date_decision,
"tags": source.tags,
"code_fr": source.code_fr,
"code_ar": source.code_ar
}
context_parts.append("\n=== CRITICAL INSTRUCTIONS ===")
context_parts.append(f"Primary Legal Code: {Config.CODE_NAMES[primary_code]['fr']}")
context_parts.append(f"مجلة القانون الأساسي: {Config.CODE_NAMES[primary_code]['ar']}")
context_parts.append(f"Total statutes retrieved: {statute_count}")
context_parts.append(f"Total jurisprudence decisions: {len(juris_sources)}")
context_parts.append(f"Cited articles requested: {query_analysis.cited_articles}")
context_parts.append(f"Missing from retrieval: {retrieval_meta['missing_articles']}")
context_parts.append(f"Search methods used: {', '.join(retrieval_meta.get('search_methods', []))}")
context_parts.append("\n=== JURISPRUDENCE SPECIFIC RULES ===")
context_parts.append("1. You CAN and SHOULD reference relevant jurisprudence principles")
context_parts.append("2. When citing jurisprudence, mention the court and date if available")
context_parts.append("3. Focus on the legal principles established in the jurisprudence")
context_parts.append("4. Use jurisprudence to support statutory interpretation")
context_parts.append("5. Highlight how jurisprudence applies to the specific case")
context_parts.append("6. Consider jurisprudence from ALL relevant codes, not just the primary one")
context_parts.append("\n=== STRICT RULES TO PREVENT HALLUCINATIONS ===")
context_parts.append("1. ONLY cite articles that appear in the retrieved sources above")
context_parts.append(" استشهد فقط بالمواد التي تظهر في المصادر المسترجعة أعلاه")
context_parts.append("2. If NO statutes are retrieved, DO NOT cite any articles")
context_parts.append(" إذا لم يتم استرجاع أي مواد قانونية، لا تستشهد بأي مواد")
context_parts.append("3. You CAN reference principles from jurisprudence, but DO NOT cite article numbers from jurisprudence")
context_parts.append(" يمكنك الإشارة إلى المبادئ من الأحكام القضائية، لكن لا تستشهد بأرقام المواد من الأحكام")
context_parts.append("\nYOU MAY ONLY CITE ARTICLES THAT APPEAR ABOVE.")
context_parts.append("يُسمح لك بالاستشهاد فقط بالمواد التي تظهر أعلاه.")
context_parts.append("When citing, ALWAYS specify which code the article comes from.")
context_parts.append("عند الاستشهاد، حدد دائمًا المجلة التي ينتمي إليها النص.")
context_parts.append("DO NOT cite, quote, or reference any article not explicitly retrieved.")
context_parts.append("لا تستشهد أو تنقل أو تشير إلى أي مادة لم يتم استرجاعها صراحة.")
context_parts.append("USE JURISPRUDENCE FROM ALL RELEVANT CODES TO SUPPORT AND ILLUSTRATE LEGAL PRINCIPLES.")
context_parts.append("استخدم الأحكام القضائية من جميع المجلات ذات الصلة لدعم وتوضيح المبادئ القانونية.")
context = "\n".join(context_parts)
return context, source_registry
# ============================================================
# ANSWER GENERATOR PROFESSIONNEL
# ============================================================
class EnterpriseAnswerGenerator:
@staticmethod
def generate_answer(query: str, context: str, source_registry: Dict[str, Any],
query_analysis: QueryAnalysis) -> Tuple[str, List[CitedSource]]:
"""Génère une réponse juridique améliorée"""
language = query_analysis.original_language
system_prompt = EnterpriseAnswerGenerator._build_system_prompt(language, source_registry, query_analysis)
user_prompt = f"""CONTEXTE ET SOURCES RÉCUPÉRÉES / السياق والمصادر المسترجعة:
{context}
QUESTION ORIGINALE / السؤال الأصلي: {query}
QUESTION TRADUITE (pour référence) / السؤال المترجم (للإشارة): {query_analysis.translated_query}
RÈGLES STRICTES POUR LA GÉNÉRATION / قواعد صارمة للتوليد:
1. Basez-vous uniquement sur les sources récupérées ci-dessus
اعتمد فقط على المصادر المسترجعة أعلاه
2. Citez exactement les articles comme ils apparaissent dans les sources
استشهد بالضبط بالمواد كما تظهر في المصادر
3. Si une source n'est pas disponible, expliquez clairement cette limite
إذا لم يكن المصدر متوفراً، اشرح هذا القيد بوضوح
4. Fournissez une analyse juridique complète et pratique
قدم تحليلاً قانونياً شاملاً وعملياً
5. Adaptez la réponse à la langue de l'utilisateur ({language})
قم بتكييف الإجابة مع لغة المستخدم ({language})
6. Utilisez la jurisprudence pour illustrer et soutenir les principes légaux
استخدم الأحكام القضائية لتوضيح ودعم المبادئ القانونية
7. Considérez la jurisprudence de TOUS les codes pertinents
ضع في الاعتبار الأحكام القضائية من جميع المجلات ذات الصلة
Générez une réponse juridique complète, précise et pratique.
قم بتوليد إجابة قانونية شاملة ودقيقة وعملية."""
try:
response = chat_client.chat.completions.create(
model=Config.CHAT_MODEL,
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt}
],
temperature=Config.TEMP_GENERATION,
max_tokens=4000
)
answer = response.choices[0].message.content.strip()
cited_sources = EnterpriseAnswerGenerator._extract_citations(answer, source_registry)
logger.info(f"✅ Réponse générée ({len(answer)} caractères)")
logger.info(f"📌 Sources citées: {len(cited_sources)}")
return answer, cited_sources
except Exception as e:
logger.error(f"Erreur génération réponse: {e}")
return "Une erreur est survenue lors de la génération de la réponse. Veuillez réessayer.", []
@staticmethod
def _build_system_prompt(language: str, source_registry: Dict[str, Any], query_analysis: QueryAnalysis) -> str:
"""Construit le prompt système amélioré"""
primary_code = query_analysis.primary_legal_code.value
primary_code_name_fr = Config.CODE_NAMES[primary_code]["fr"]
primary_code_name_ar = Config.CODE_NAMES[primary_code]["ar"]
articles_by_code = {}
for article, info in source_registry["statutes"].items():
code = info["code"]
if code not in articles_by_code:
articles_by_code[code] = []
code_name = info["code_name_ar"] if language == "ar" else info["code_name_fr"]
articles_by_code[code].append(f"{article} ({code_name})")
available_text = ""
for code, articles in articles_by_code.items():
code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"]
available_text += f"\n{code_name}: {', '.join(articles)}"
juris_count = len(source_registry.get("jurisprudence", {}))
missing_articles = source_registry.get("missing_articles", [])
juris_by_code = {}
for juris_id, juris_info in source_registry.get("jurisprudence", {}).items():
code_fr = juris_info.get("code_fr", "")
code_ar = juris_info.get("code_ar", "")
for code, code_info in Config.CODE_NAMES.items():
if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar):
if code not in juris_by_code:
juris_by_code[code] = 0
juris_by_code[code] += 1
break
juris_by_code_text = ""
for code, count in juris_by_code.items():
code_name = Config.CODE_NAMES[code]["ar"] if language == "ar" else Config.CODE_NAMES[code]["fr"]
juris_by_code_text += f"\n{code_name}: {count} decisions"
if language == "ar":
return f"""أنت خبير قانوني تونسي محترف متخصص في جميع المجلات العشرة التونسية.
المجلة الأساسية المعنية: {primary_code_name_ar}
المواد القانونية المتاحة:{available_text if articles_by_code else " لا توجد مواد قانونية مسترجعة"}
عدد الأحكام القضائية المتاحة: {juris_count} حكم
توزيع الأحكام القضائية حسب المجلة:{juris_by_code_text if juris_by_code_text else " لا توجد أحكام قضائية مصنفة"}
المواد المطلوبة وغير المتوفرة: {', '.join(missing_articles)}
أنت خبير في:
1. تفسير النصوص القانونية التونسية
2. تحليل الأحكام القضائية وتطبيقها على الحالات الواقعية
3. تقديم نصائح قانونية عملية ومفصلة
4. التمييز بين المصادر القانونية المختلفة
5. استخدام الاجتهاد القضائي من جميع المجلات ذات الصلة لتوضيح المبادئ القانونية
قواعد صارمة:
1. لا تخترع أي مواد أو أحكام غير موجودة في المصادر
2. إذا لم تجد مصدراً، اعترف بذلك واشرح البدائل
3. كن دقيقاً في الاستشهادات والإحالات
4. قدم إجابة متوازنة وعملية
5. استخدم الأحكام القضائية من جميع المجلات ذات الصلة لتوضيح كيفية تطبيق النصوص القانونية
استخدم الأحكام القضائية المتاحة لتوضيح:
- كيفية تفسير المحاكم للنصوص القانونية
- المبادئ القانونية المستقرة في الاجتهاد
- كيفية تطبيق القانون على حالات مشابهة
- الاتجاهات الحديثة في التفسير القضائي
- الاختلافات أو التشابهات بين تفسيرات المحاكم للمواد المختلفة
قم بتحليل السؤال بدقة وقدم إجابة شاملة تعتمد على المصادر المتاحة من جميع المجلات ذات الصلة."""
else:
return f"""You are a professional Tunisian legal expert specializing in all 10 Tunisian codes.
Primary Legal Code: {primary_code_name_fr}
Available Legal Articles:{available_text if articles_by_code else " NO statutes retrieved"}
Available Jurisprudence Decisions: {juris_count} decisions
Jurisprudence Distribution by Code:{juris_by_code_text if juris_by_code_text else " No jurisprudence classified by code"}
Requested but Unavailable Articles: {', '.join(missing_articles)}
You are expert in:
1. Interpreting Tunisian legal texts
2. Analyzing case law and applying it to real cases
3. Providing practical, detailed legal advice
4. Distinguishing between different legal sources
5. Using jurisprudence from ALL relevant codes to illustrate legal principles
Strict Rules:
1. DO NOT invent any articles or jurisprudence not in the sources
2. If a source is not found, acknowledge this and explain alternatives
3. Be precise in citations and references
4. Provide balanced, practical advice
5. Use available jurisprudence from ALL relevant codes to clarify how legal texts are applied
Use available jurisprudence to illustrate:
- How courts interpret legal texts
- Established legal principles in case law
- How the law is applied to similar cases
- Recent trends in judicial interpretation
- Differences or similarities between court interpretations of different articles
Analyze the question accurately and provide a comprehensive answer based on available sources from ALL relevant codes."""
@staticmethod
def _extract_citations(answer: str, source_registry: Dict[str, Any]) -> List[CitedSource]:
"""Extrait les citations de la réponse"""
citations = []
for article, info in source_registry["statutes"].items():
normalized_article = ArticleNumberNormalizer.normalize(article)
patterns = [
f"(?:الفصل|فصل|Article|article|المادة|مادة)\\s*{normalized_article}\\b",
f"\\b{normalized_article}\\b(?!\\s*bis|\\s*ter|\\s*quater)"
]
for pattern in patterns:
matches = re.finditer(pattern, answer, re.IGNORECASE)
for match in matches:
context = answer[max(0, match.start()-50):min(len(answer), match.end()+50)]
citations.append(CitedSource(
article_number=article,
code_type=LegalCode.from_string(info["code"]),
citation_context=context,
citation_position=match.start(),
retrieval_status=RetrievalStatus.SUCCESSFULLY_RETRIEVED
))
return citations
# ============================================================
# VALIDATOR PROFESSIONNEL
# ============================================================
class AnswerValidator:
@staticmethod
def validate(answer: str, cited_sources: List[CitedSource], source_registry: Dict[str, Any]) -> ValidationResult:
"""Valide la réponse générée"""
hallucinated = []
missing_retrievals = []
errors = []
warnings = []
retrieved_lookup = {}
for article, info in source_registry["statutes"].items():
normalized = ArticleNumberNormalizer.normalize(article)
retrieved_lookup[normalized] = {
"code": info["code"],
"original": article,
"info": info
}
for citation in cited_sources:
normalized_cite = ArticleNumberNormalizer.normalize(citation.article_number)
if normalized_cite in retrieved_lookup:
retrieved_info = retrieved_lookup[normalized_cite]
if retrieved_info["code"] == citation.code_type.value:
citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED
else:
warnings.append(f"Article {citation.article_number} cited with code {citation.code_type.value} but retrieved from {retrieved_info['code']}")
citation.retrieval_status = RetrievalStatus.SUCCESSFULLY_RETRIEVED
else:
hallucinated.append(f"{citation.article_number} ({citation.code_type.value})")
citation.retrieval_status = RetrievalStatus.CITED_NOT_RETRIEVED
errors.append(f"Article {citation.article_number} from {citation.code_type.value} cited but NOT retrieved")
all_article_numbers = ArticleNumberNormalizer.extract_all_numbers(answer)
for art_num in all_article_numbers:
normalized_art_num = ArticleNumberNormalizer.normalize(art_num)
found_in_sources = normalized_art_num in retrieved_lookup
already_cited = any(
ArticleNumberNormalizer.normalize(citation.article_number) == normalized_art_num
for citation in cited_sources
)
if not found_in_sources and not already_cited:
warnings.append(f"Potential uncited reference to article {art_num}")
for article in source_registry.get("cited_articles", []):
normalized = ArticleNumberNormalizer.normalize(article)
if normalized not in retrieved_lookup:
missing_retrievals.append(article)
warnings.append(f"Requested article {article} was NOT retrieved")
if hallucinated:
confidence = 0.0
elif missing_retrievals:
confidence = 0.7
elif warnings:
confidence = 0.85
else:
confidence = 0.95
is_valid = len(hallucinated) == 0
if not is_valid:
errors.append(f"CRITICAL: {len(hallucinated)} hallucinated citations detected")
return ValidationResult(
is_valid=is_valid,
confidence_score=confidence,
hallucinated_citations=hallucinated,
missing_retrievals=missing_retrievals,
validation_errors=errors,
validation_warnings=warnings
)
# ============================================================
# CHATBOT FINAL AMÉLIORÉ
# ============================================================
class UltimateLegalChatbot:
"""Chatbot juridique ultime avec toutes les améliorations"""
def __init__(self):
self.query_analyzer = EnterpriseQueryAnalyzer()
self.search_engine = EnhancedEnterpriseSearchEngine()
self.context_builder = EnterpriseContextBuilder()
self.answer_generator = EnterpriseAnswerGenerator()
self.validator = AnswerValidator()
def process_query(self, query: str) -> LegalAnswer:
"""Traite une requête avec toutes les améliorations"""
logger.info(f"\n{'='*120}")
logger.info("🚀 ULTIMATE LEGAL CHATBOT - 10 TUNISIAN CODES (HYBRID SEARCH)")
logger.info(f"{'='*120}\n")
start_time = datetime.now()
logger.info("📊 ÉTAPE 1: Analyse de la requête")
query_analysis = self.query_analyzer.analyze_query(query)
logger.info("\n🔍 ÉTAPE 2: Récupération des sources améliorée avec recherche hybride")
search_results = self.search_engine.search(query_analysis)
logger.info("\n🧱 ÉTAPE 3: Construction du contexte améliorée")
context, source_registry = self.context_builder.build_context(search_results)
logger.info("\n✍️ ÉTAPE 4: Génération de la réponse améliorée")
answer_text, cited_sources = self.answer_generator.generate_answer(
query, context, source_registry, query_analysis
)
logger.info("\n✅ ÉTAPE 5: Validation de la réponse")
validation_result = self.validator.validate(answer_text, cited_sources, source_registry)
all_sources = []
for level in ["high", "medium", "low"]:
all_sources.extend(search_results["statute_sources"].get(level, []))
all_sources.extend(search_results["jurisprudence_sources"])
end_time = datetime.now()
processing_time = (end_time - start_time).total_seconds()
juris_code_distribution = {}
for source in search_results["jurisprudence_sources"]:
code_fr = source.code_fr
code_ar = source.code_ar
for code_key, code_info in Config.CODE_NAMES.items():
if (code_info["fr"] and code_info["fr"] in code_fr) or (code_info["ar"] and code_info["ar"] in code_ar):
if code_key not in juris_code_distribution:
juris_code_distribution[code_key] = 0
juris_code_distribution[code_key] += 1
legal_answer = LegalAnswer(
answer_text=answer_text,
retrieved_sources=all_sources,
cited_sources=cited_sources,
validation_result=validation_result,
query_analysis=query_analysis,
processing_metadata={
"processing_time": processing_time,
"retrieval_stats": search_results["retrieval_metadata"],
"translation_applied": query_analysis.original_language == "ar",
"jurisprudence_found": len(search_results["jurisprudence_sources"]),
"primary_code": query_analysis.primary_legal_code.value,
"secondary_codes": [c.value for c in query_analysis.secondary_codes],
"total_codes_available": 10,
"search_methods": search_results["retrieval_metadata"]["search_methods"],
"hybrid_search_used": True,
"jurisprudence_code_distribution": juris_code_distribution,
"timestamp": datetime.now().isoformat()
},
confidence_score=validation_result.confidence_score * query_analysis.code_confidence
)
logger.info(f"\n✅ Traitement terminé en {processing_time:.2f} secondes")
logger.info(f"📊 Résumé: {len(all_sources)} sources totales")
logger.info(f"🎯 Système 10 codes tunisiens avec recherche hybride opérationnel")
return legal_answer
# ============================================================
# CHAINLIT APPLICATION (CHAT INTERFACE)
# ============================================================
# ... (all previous imports and classes remain exactly the same until the Chainlit part) ...
# ============================================================
# CHAINLIT APPLICATION (CHAT INTERFACE) - MODIFIED
# ============================================================
chatbot = UltimateLegalChatbot()
@cl.on_chat_start
async def start():
"""Initialise la session de chat avec un message d'accueil minimal"""
await cl.Message(
content="""Legal Assistant | مساعد قانوني
Ask your legal question | اطرح سؤالك القانوني""" ).send()
@cl.on_message
async def main(message: cl.Message):
"""Traite le message de l'utilisateur et renvoie la réponse + sources complètes sans termes techniques"""
user_query = message.content.strip()
if not user_query:
await cl.Message(content="Veuillez poser une question valide.").send()
return
async with cl.Step(name="Analyse et recherche en cours", type="loading"):
legal_answer = await asyncio.to_thread(chatbot.process_query, user_query)
await cl.Message(content=legal_answer.answer_text).send()
sources_message = ""
statutes = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.STATUTE]
if statutes:
sources_message += "### 📜 Articles de loi (texte intégral)\n\n"
for idx, stat in enumerate(statutes, 1):
if stat.article_metadata:
art_num = stat.article_metadata.article_number
if legal_answer.query_analysis.original_language == "ar":
code_name = stat.article_metadata.code_name_ar
else:
code_name = stat.article_metadata.code_name_fr
sources_message += f"**{idx}. {code_name} – Article {art_num}**\n"
sources_message += f"```\n{stat.full_text}\n```\n\n"
juris = [s for s in legal_answer.retrieved_sources if s.source_type == SourceType.JURISPRUDENCE]
if juris:
sources_message += "### ⚖️ Décisions de jurisprudence (texte intégral)\n\n"
for idx, dec in enumerate(juris, 1):
title = f"**{idx}. {dec.juridiction or 'Décision'}**"
if dec.date_decision:
title += f" – {dec.date_decision}"
sources_message += title + "\n"
if legal_answer.query_analysis.original_language == "ar":
code_display = dec.code_ar
else:
code_display = dec.code_fr
if code_display:
sources_message += f"*Code : {code_display}*\n"
sources_message += f"```\n{dec.full_text}\n```\n\n"
if sources_message:
await cl.Message(content=sources_message).send()
else:
await cl.Message(content="*Aucune source textuelle n'a été trouvée pour cette question.*").send()
if __name__ == "__main__":
from chainlit.cli import run
run()