"""Arabic-only presentation layer shared by the Gradio app and the static (in-browser) page.
Colour rules: green = verified, red = a real mismatch or a missing reference, amber = needs a human. Verified items
never use red. Every user-facing string is Arabic.
"""
from __future__ import annotations
import html
from typing import List, Optional
from benchmark_format import official_payload
_e = html.escape
APP_TITLE = "التحقق من هلوسة القرآن والحديث وتصحيحها"
APP_TAGLINE = "تحقّق من آيات القرآن والأحاديث النبوية بالدليل"
COPIED_MESSAGE = "تم نسخ النص بنجاح"
GROUP_LABEL = {"verified": "موثّق", "mismatch": "غير مطابق", "review": "تحتاج مراجعة بشرية"}
TYPE_LABEL = {"Ayah": "آية قرآنية", "Hadith": "حديث نبوي"}
INDICATORS = [
("composite", "الدرجة المركبة"),
("coverage", "تغطية الكلمات"),
("lcs_ratio", "التسلسل النصي"),
("token_overlap", "تداخل الكلمات"),
("edit_sim", "تشابه الحروف"),
("diacritic_sim", "التشكيل"),
]
METHODS = {
"substring_match": "الاقتباس واردٌ كاملًا في المصدر",
"threshold_pass": "تجاوز عتبة التطابق",
"threshold_fail": "دون عتبة التطابق",
"borderline_multi_cov": "حالة حدّية: عدة مصادر متقاربة",
"borderline_default": "حالة حدّية",
"borderline_low_retrieval": "حالة حدّية: استرجاع ضعيف",
"no_candidates": "لا توجد مصادر مرشحة",
"empty_span": "نص فارغ",
"error": "تعذّرت المعالجة",
}
def _pct(value: float) -> str:
return f"{round(value * 100)}%"
def source_label(source: dict) -> str:
if source["type"] == "Quran":
start, end = source["ayah_start"], source["ayah_end"]
verses = f"{start}" if start == end else f"{start}–{end}"
return f"سورة {source['surah_name']} — الآية {verses}"
return f"حديث رقم {source['hadithID']} — {source['title']}"
def reason_text(reason: dict) -> str:
"""Arabic explanation of a decision code produced by the pipeline."""
code = reason["code"]
src = reason.get("source", "")
if code == "exact_match":
return f"تطابق تام مع {src} بعد تجاهل اختلافات الرسم والتشكيل."
if code == "close_match":
return f"تطابق شبه كامل مع {src}."
if code == "altered_passage":
n = reason.get("n", 0)
how = "تغيّر ترتيب الكلمات" if reason.get("reordered") else f"{n} موضع مختلف"
return f"النص يخالف {src} ({how}). التصحيح المقترح هو نص المصدر حرفيًا ولم يُولَّد."
if code == "weak_match":
return "وُجد مصدر قريب لكن التطابق غير كافٍ للحكم الآلي."
if code == "too_short":
return "الاقتباس قصير جدًا (كلمتان فقط) فلا يمكن تحديد مصدره بيقين؛ يلزم الرجوع إلى مختص."
if code == "no_source":
return "لا يوجد في المراجع المضمّنة نصٌّ يشبه هذا الاقتباس، وقد يكون مختلَقًا."
if code == "insufficient_evidence":
return "الأدلة غير كافية: لا مصدر واضح، ودرجة اليقين منخفضة."
if code == "candidate_not_strong":
return f"وُجد مرشح ({src}، قوة المطابقة {round(reason.get('strength', 0) * 100)}%) لكن الدليل لا يكفي للتصحيح الآلي."
if code == "hadith_candidate":
return f"وُجدت رواية مشابهة ({src}) لكن لا يُصحَّح الحديث آليًا لاختلاف الروايات؛ يلزم الرجوع إلى مختص."
return "تعذّرت معالجة هذا الاقتباس آليًا."
def note_text(note: dict) -> str:
n = note.get("n", 0)
code = note["code"]
if code == "diacritic_conflict":
return f"تنبيه: تشكيل {n} كلمة يخالف المصحف (الكلمات صحيحة لكن الحركات مختلفة)."
if code == "orthographic_variant":
return f"ملاحظة: {n} اختلاف إملائي (رسم أو مسافات) لا يغيّر الكلمة."
if code == "hadith_minor_diffs":
return f"ملاحظة: {n} اختلاف طفيف عن أقرب رواية في المراجع."
return ""
def _diff_html(comparison: dict) -> str:
"""Red = words only in the quotation; green = the source words that should be there."""
parts = []
for op in comparison["word_diff"]:
if op["op"] == "equal":
parts.append(f'{_e(op["span"])}')
continue
if op["span"]:
parts.append(f'{_e(op["span"])}')
if op["source"]:
parts.append(f'{_e(op["source"])}')
return " ".join(parts)
def _legend(group: str, comparison: Optional[dict]) -> str:
if group == "verified" or not comparison or all(op["op"] == "equal" for op in comparison["word_diff"]):
return ""
return '
كلمات في الاقتباس تخالف المصدرالصواب من المصدر
'
def _indicator_table(signals: Optional[dict]) -> str:
if not signals:
return ""
rows = []
for key, label in INDICATORS:
value = float(signals.get(key, 0.0))
rows.append(
f'
{label}
'
f'
{_pct(value)}
'
)
rows.append(f'
وروده كاملًا في المصدر
{"نعم" if signals.get("is_substring") else "لا"}
')
return '
' + "".join(rows) + "
"
def _match_percent(span: dict) -> int:
evidence = span["evidence"]
if evidence and evidence.get("comparison"):
return round(evidence["comparison"]["word_similarity"] * 100)
if evidence and evidence.get("signals"):
return round(evidence["signals"]["composite"] * 100)
return 0
def _evidence_panel(span: dict) -> str:
evidence, verification = span["evidence"], span["verification"]
rows = [
f'
الاقتباس المكتشف
{_e(span["text"])}
',
f'
النوع
{TYPE_LABEL[span["type"]]}
',
]
if evidence:
comparison = evidence["comparison"]
rows += [
f'
',
]
if comparison["diacritic_notes"]:
items = "، ".join(f"{_e(n['word'])} ← {_e(n['source_word'])}" for n in comparison["diacritic_notes"])
rows.append(f'
"
def final_text(result: dict) -> str:
"""Corrected text with an inline Arabic flag after every quotation that still needs attention."""
text, reports = result["input_text"], result["spans"]
for report in sorted(reports, key=lambda r: r["start"], reverse=True):
if report["status"] == "CORRECTED" and report["correction"]:
text = text[: report["start"]] + report["correction"]["display_text"] + text[report["end"]:]
elif report["status"] == "UNSUPPORTED":
text = text[: report["end"]] + " [⚠ لا يوجد مصدر مطابق]" + text[report["end"]:]
elif report["status"] == "HUMAN_REVIEW":
text = text[: report["end"]] + " [⚠ يحتاج مراجعة بشرية]" + text[report["end"]:]
return text
def _final_block(result: dict) -> str:
summary = result["summary"]
fixed = summary["CORRECTED"]
open_items = summary["UNSUPPORTED"] + summary["HUMAN_REVIEW"]
if fixed == 0 and open_items == 0:
headline = "كل الاقتباسات المكتشفة مطابقة للمصادر."
else:
headline = f"صُحِّح {fixed} اقتباس من نص المصدر، وبقي {open_items} اقتباس يحتاج مراجعة بشرية ومعلَّم بعلامة تنبيه."
text = final_text(result)
return (
'
النسخة المصحّحة'
f'
'
f'
{headline}
{_e(text)}
'
)
_EXPORT_LABELS = (("task1A_predictions.tsv", "ملف الكشف (1A)", "أماكن الاقتباسات ونوعها"),
("task1B_predictions.tsv", "ملف التحقق (1B)", "صحيح أو خطأ لكل اقتباس"),
("task1C_predictions.tsv", "ملف التصحيح (1C)", "النص الصحيح من المصدر"))
def _export_block(result: dict) -> str:
"""Download buttons for the three official submission files (client-side Blob download, nothing is uploaded)."""
payload = official_payload(result)
buttons = "".join(
f''
for name, label, sub in _EXPORT_LABELS
)
return ('للمقيِّمين: تنزيل النتائج بصيغة ملفات المسابقة الرسمية'
'
هذه الملفات الثلاثة هي صيغة تسليم المهمة الأولى في مسابقة الذكاء الاصطناعي لخدمة المحتوى الإسلامي: '
'الكشف عن مواضع الآيات والأحاديث، ثم الحكم على كل اقتباس، ثم تصحيح الخاطئ منها. '
'لا تحتاجها لقراءة النتيجة أعلاه؛ تفيد لجان التحكيم ومن يقيس الأداء على بيانات المسابقة.
'
f'
{buttons}
')
def render_results(result: dict, generated_answer: Optional[str] = None) -> str:
"""HTML report for a pipeline result. Pass the model's answer (mode B) to show it first and label the text as generated."""
header = render_generated_header(generated_answer) if generated_answer is not None else ""
if not result["spans"]:
body = (
'
لم يُعثر على اقتباسات قرآنية أو حديثية في هذا النص. يتعرّف النظام على الاقتباسات بمطابقتها مع المراجع، '
'سواء وُضعت بين علامات تنصيص أو أقواس أو وردت داخل الكلام دون أي عبارة تمهيدية.
'
)
return f'
{header}{body}
'
generated = generated_answer is not None
title = "إجابة النموذج مع الاقتباسات المكتشفة" if generated else "النص مع الاقتباسات المكتشفة"
parts = [_summary(result["summary"]), _highlighted_text(result, title), "".join(_card(s) for s in result["spans"])]
summary = result["summary"]
if generated or summary["CORRECTED"] or summary["UNSUPPORTED"] or summary["HUMAN_REVIEW"]:
parts.append(_final_block(result))
parts.append(_export_block(result))
return f'
{header}{"".join(parts)}
'
def render_generated_header(answer: str) -> str:
return f'
'
# --------------------------------------------------------------------------------------------------------------
# Static markup and CSS
# --------------------------------------------------------------------------------------------------------------
STAR = (
''
)
HERO = f"""
{STAR}
{APP_TITLE}
{APP_TAGLINE}
يكتشف النظام الاقتباسات داخل أي نص، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل.
وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.
كشف‹استرجاع‹محاذاة‹دليل‹قرار
"""
DISCLAIMER = """
أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة.
النتائج مبنية على مراجع القرآن الكريم والكتب الستة المضمّنة فقط.