"""Islamic Content Verifier: Gradio interface around the verification pipeline in ``verifier.py``. Run locally with ``python app.py``. The rendering helpers are plain functions that return HTML, so they can be tested without Gradio; Gradio is imported only when the interface is built. """ from __future__ import annotations import html import json import logging import os import threading from pathlib import Path from typing import List, Optional from verifier import MAX_INPUT_CHARS, IslamicContentVerifier logger = logging.getLogger(__name__) EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json" DEFAULT_MODEL_DIR = Path(__file__).resolve().parent / "models" / "span_detector" # -------------------------------------------------------------------------------------------------------------- # Pipeline access # -------------------------------------------------------------------------------------------------------------- _pipeline: Optional[IslamicContentVerifier] = None _pipeline_lock = threading.Lock() def get_pipeline() -> IslamicContentVerifier: """Create the pipeline once (the corpus index builds in a few seconds) and reuse it for every request.""" global _pipeline with _pipeline_lock: if _pipeline is None: model_dir = os.environ.get("ICV_DETECTOR_MODEL") or (str(DEFAULT_MODEL_DIR) if DEFAULT_MODEL_DIR.is_dir() else None) # With a trained model folder the BERT detector is used; if it cannot be loaded the rules detector takes over. _pipeline = IslamicContentVerifier(detector="auto" if model_dir else "rules", model_dir=model_dir) return _pipeline def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]: try: with open(path, encoding="utf-8") as handle: return json.load(handle) except (OSError, json.JSONDecodeError): logger.exception("Could not load demo examples from %s", path) return [] # -------------------------------------------------------------------------------------------------------------- # HTML rendering # -------------------------------------------------------------------------------------------------------------- _e = html.escape GROUP_LABEL = { "verified": ("موثّق", "Verified"), "mismatch": ("غير مطابق", "Mismatch"), "review": ("يحتاج مراجعة بشرية", "Needs Human Review"), } TYPE_LABEL = {"Ayah": ("آية قرآنية", "Quran"), "Hadith": ("حديث نبوي", "Hadith")} INDICATORS = [ ("composite", "الدرجة المركبة", "Composite score"), ("coverage", "تغطية الكلمات", "Word coverage"), ("lcs_ratio", "التسلسل النصي", "Sequence (LCS)"), ("token_overlap", "تداخل الكلمات", "Token overlap"), ("edit_sim", "تشابه الأحرف", "Character similarity"), ("diacritic_sim", "مع التشكيل", "With diacritics"), ] def _pct(value: float) -> str: return f"{round(value * 100)}%" def _source_label(source: dict) -> str: if source["type"] == "Quran": start = source.get("ayah", source.get("ayah_start")) end = source.get("ayah_end", start) verses = f"{start}" if start == end else f"{start}–{end}" return f"سورة {source['surah_name']} — الآية {verses}" return f"حديث رقم {source['hadithID']} — {source['title']}" def _truncate(text: str, limit: int) -> str: return text if len(text) <= limit else text[:limit].rstrip() + " …" def _diff_html(comparison: dict) -> str: """Word-level diff: green = words only in the source, red = words only in the quotation.""" parts = [] for op in comparison["word_diff"]: if op["op"] == "equal": parts.append(f'{_e(op["span"])}') else: if op["span"]: parts.append(f'{_e(op["span"])}') if op["source"]: parts.append(f'{_e(op["source"])}') return " ".join(parts) def _indicator_table(signals: Optional[dict]) -> str: if not signals: return "" rows = [] for key, ar, en in INDICATORS: value = float(signals.get(key, 0.0)) rows.append( f'
{ar}{en}
' f'
{_pct(value)}
' ) substring = "نعم · Yes" if signals.get("is_substring") else "لا · No" rows.append(f'
احتواء كاملContained in source
{substring}
') return '
' + "".join(rows) + "
" def _evidence_panel(span: dict) -> str: evidence, verification = span["evidence"], span["verification"] blocks = [ f'
الاقتباس المكتشف · Detected quotation

{_e(span["text"])}

', f'
النوع · Type

{TYPE_LABEL[span["type"]][0]} · {TYPE_LABEL[span["type"]][1]}

', ] if evidence: comparison = evidence["comparison"] blocks += [ f'
المصدر المرشح · Candidate source

{_e(_source_label(evidence["source"]))}

', f'
نص المصدر الأصلي · Original source text

{_e(comparison["source_excerpt"])}

', f'
المقارنة · Comparison

{_diff_html(comparison)}

' f'

في الاقتباس فقط في المصدر فقط ' f'· تشابه الكلمات {_pct(comparison["word_similarity"])}

', f'
مؤشرات التحقق · Similarity indicators{_indicator_table(evidence.get("signals"))}
', ] else: blocks.append('
المصدر · Source

لم يُسترجع أي مصدر مرشح · No candidate source was retrieved.

') blocks += [ f'
الثقة · Confidence

{_pct(verification["confidence"])}

' f'

verdict: {_e(verification["verdict"])} · method: {_e(verification["method"])} · ' f'candidates checked: {verification["n_candidates"]}

', f'
القرار · Decision

{_e(span["status_ar"])}
{_e(span["status_en"])}

', f'
سبب القرار · Decision reason

{_e(span["reason"])}

', ] return "".join(blocks) def _card(span: dict) -> str: group = span["group"] ar_label, en_label = GROUP_LABEL[group] type_ar, type_en = TYPE_LABEL[span["type"]] evidence = span["evidence"] confidence = span["verification"]["confidence"] source_html = comparison_html = "" if evidence: comparison = evidence["comparison"] source_html = ( f'

{_e(_source_label(evidence["source"]))}

' f'
' f'

{_e(_truncate(comparison["source_excerpt"], 360))}

' ) comparison_html = ( f'

{_diff_html(comparison)}

' ) else: source_html = '

لا يوجد · None found

' action = "" correction, suggestion = span["correction"], span["suggestion"] if correction: action = ( '
تصحيح مدعوم بالمصدر · Source-backed correction' f'

{_e(correction["display_text"])}

' f'

{_e(_source_label({"type": "Quran", **correction["source"]}) if correction["source"]["type"] == "Quran" else _source_label(correction["source"]))}' f' · قوة المطابقة {_pct(correction["match_strength"])}

' ) elif group == "review": closest = "" if suggestion: src = suggestion["source"] label = _source_label({"type": "Quran", **src}) if src["type"] == "Quran" else _source_label(src) closest = (f'

أقرب مصدر وُجد للمراجِع (ليس تصحيحًا آليًا) · Closest source for the reviewer (not an automatic correction): ' f'{_e(label)}

{_e(_truncate(suggestion["display_text"], 360))}

') action = ( '
الأدلة غير كافية للتصحيح الآلي. يُوصى بالمراجعة البشرية.' '

Insufficient evidence for automatic correction. Human review is recommended.

' + closest + "
" ) elif span["status"] == "UNSUPPORTED": action = ( '
لا يوجد مصدر مطابق في المراجع المتاحة للنظام. لم يُقترح أي نص بديل.' '

No matching source in the corpus. No replacement text is proposed.

' ) return f"""
{span["id"]} {type_ar} · {type_en} {ar_label} · {en_label} الثقة · Confidence {_pct(confidence)}

{_e(span["text"])}

{source_html} {comparison_html}

{_e(span["status_ar"])}
{_e(span["status_en"])}

{action}
عرض الدليل · View Evidence
{_evidence_panel(span)}
""" def _summary(summary: dict) -> str: tiles = [ (summary["n_spans"], "إجمالي الاقتباسات", "Total quotations", ""), (summary["n_ayah"], "آيات قرآنية", "Quranic", ""), (summary["n_hadith"], "أحاديث", "Hadith", ""), (summary["VERIFIED"], "موثّق", "Verified", "verified"), (summary["CORRECTED"] + summary["UNSUPPORTED"], "غير مطابق", "Mismatch", "mismatch"), (summary["HUMAN_REVIEW"], "يحتاج مراجعة بشرية", "Needs Human Review", "review"), ] return '
' + "".join( f'
{value}{ar}{en}
' for value, ar, en, cls in tiles ) + "
" def _highlighted_text(result: dict) -> str: text, pieces, cursor = result["input_text"], [], 0 for span in result["spans"]: pieces.append(_e(text[cursor:span["start"]])) pieces.append(f'{_e(text[span["start"]:span["end"]])}') cursor = span["end"] pieces.append(_e(text[cursor:])) legend = ( '
موثّق · Verifiedغير مطابق · Mismatch' 'مراجعة بشرية · Review
' ) return ( '

' + "".join(pieces) + "

" + legend + "
" ) def render_results(result: dict) -> str: """HTML report for a pipeline result.""" if not result["spans"]: return ( '
لم يُعثر على اقتباسات قرآنية أو حديثية في هذا النص. ' 'No Quranic or Hadith quotations were detected. Quotations are currently detected when they are ' 'enclosed in quotation marks or brackets and introduced by a typical phrase (see Limitations).
' ) names = {"RuleDetector": ("القواعد", "Rule-based"), "BertDetector": ("نموذج BERT", "BERT model")} ar, en = names.get(result["detector"], (result["detector"], result["detector"])) chip = f'
الكاشف · Detector: {ar} · {en}
' return chip + _summary(result["summary"]) + _highlighted_text(result) + "".join(_card(s) for s in result["spans"]) def render_message(message_ar: str, message_en: str, kind: str = "info") -> str: return f'
{_e(message_ar)}{_e(message_en)}
' def verify_text(text: str) -> str: """Gradio handler: validate the input, run the pipeline and return HTML (never raises).""" if not text or not text.strip(): return render_message("الرجاء إدخال نص للتحقق منه.", "Please enter a text to verify.", "warn") try: return render_results(get_pipeline().analyze(text)) except ValueError: return render_message( f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", f"The text is too long (maximum {MAX_INPUT_CHARS} characters).", "warn" ) except Exception: logger.exception("Verification failed") return render_message("حدث خطأ غير متوقع أثناء التحقق.", "An unexpected error occurred during verification.", "bad") # -------------------------------------------------------------------------------------------------------------- # Interface # -------------------------------------------------------------------------------------------------------------- _PATTERN = ( "url(\"data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='80' height='80' viewBox='0 0 80 80'%3E" "%3Cg fill='none' stroke='%2322d3ee' stroke-opacity='0.07' stroke-width='1'%3E" "%3Crect x='20' y='20' width='40' height='40'/%3E%3Crect x='20' y='20' width='40' height='40' transform='rotate(45 40 40)'/%3E" "%3C/g%3E%3C/svg%3E\")" ) CSS = """ @import url('https://fonts.googleapis.com/css2?family=Cairo:wght@400;600;700&family=Amiri:wght@400;700&display=swap'); .gradio-container { --body-background-fill:#0a1128; --block-background-fill:#101a38; --block-border-color:#1e2d57; --input-background-fill:#0d1631; --body-text-color:#e8eefc; --block-label-text-color:#9fb3da; --button-primary-background-fill:#14b8a6; --button-primary-background-fill-hover:#22d3ee; --button-primary-text-color:#04202a; --button-secondary-background-fill:#16244a; --button-secondary-text-color:#e8eefc; --button-secondary-border-color:#2a3d73; background:#0a1128 PATTERN; font-family:'Cairo','Segoe UI',Tahoma,sans-serif; max-width:1080px !important; color:#e8eefc; } .hero { text-align:center; padding:34px 12px 18px; direction:rtl; } .hero .eyebrow { color:#22d3ee; letter-spacing:.14em; font-size:.8rem; text-transform:uppercase; } .hero h1 { font-size:2.3rem; margin:.3rem 0 .1rem; color:#fff; } .hero h2 { font-size:1.05rem; font-weight:400; color:#9fb3da; margin:0 0 .9rem; direction:ltr; } .hero p { max-width:760px; margin:.3rem auto; color:#c7d4f0; line-height:1.9; } .hero .en { direction:ltr; color:#8ea3cf; font-size:.92rem; } .input-area textarea { direction:rtl; text-align:right; font-family:'Amiri','Cairo',serif !important; font-size:1.2rem !important; line-height:2 !important; } .btn-row button { font-weight:700 !important; border-radius:12px !important; } .disclaimer { font-size:.82rem; color:#8ea3cf; text-align:center; padding:12px 8px; direction:rtl; line-height:1.8; } .disclaimer .en { display:block; direction:ltr; } .results { direction:rtl; text-align:right; } .results .en, .notice .en { display:block; direction:ltr; text-align:left; color:#8ea3cf; font-size:.88rem; font-weight:400; } .summary { display:grid; grid-template-columns:repeat(auto-fit,minmax(130px,1fr)); gap:10px; margin:6px 0 14px; } .tile { background:#101a38; border:1px solid #1e2d57; border-radius:14px; padding:12px; text-align:center; } .tile b { display:block; font-size:1.8rem; color:#fff; } .tile span { display:block; font-size:.9rem; } .tile small { color:#8ea3cf; } .tile.verified { border-color:#14b8a6; } .tile.verified b { color:#2dd4bf; } .tile.mismatch { border-color:#f43f5e; } .tile.mismatch b { color:#fb7185; } .tile.review { border-color:#f59e0b; } .tile.review b { color:#fbbf24; } .highlight { background:#0d1631; border:1px solid #1e2d57; border-radius:14px; padding:14px 16px; margin-bottom:14px; } .highlight label, .field label { display:block; color:#7e93c0; font-size:.78rem; margin-bottom:4px; } .highlight p { font-family:'Amiri','Cairo',serif; font-size:1.15rem; line-height:2.1; margin:0; white-space:pre-wrap; } mark { color:#fff; border-radius:6px; padding:1px 4px; } mark.verified { background:rgba(20,184,166,.35); } mark.mismatch { background:rgba(244,63,94,.35); } mark.review { background:rgba(245,158,11,.35); } .qcard { background:#101a38; border:1px solid #1e2d57; border-inline-start:5px solid #3b4b7a; border-radius:16px; padding:16px 18px; margin:12px 0; } .qcard.verified { border-inline-start-color:#14b8a6; } .qcard.mismatch { border-inline-start-color:#f43f5e; } .qcard.review { border-inline-start-color:#f59e0b; } .qhead { display:flex; flex-wrap:wrap; gap:8px; align-items:center; margin-bottom:10px; } .idx { background:#1b2a55; border-radius:50%; width:28px; height:28px; display:inline-flex; align-items:center; justify-content:center; font-weight:700; } .badge { padding:3px 12px; border-radius:999px; font-size:.82rem; background:#16244a; border:1px solid #2a3d73; } .badge.st.verified { background:rgba(20,184,166,.18); border-color:#14b8a6; color:#5eead4; } .badge.st.mismatch { background:rgba(244,63,94,.16); border-color:#f43f5e; color:#fda4af; } .badge.st.review { background:rgba(245,158,11,.16); border-color:#f59e0b; color:#fcd34d; } .conf { margin-inline-start:auto; color:#9fb3da; font-size:.88rem; } .conf b { color:#22d3ee; } .field { margin:8px 0; } .field p { margin:0; line-height:1.8; } .quote { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.1; color:#fff; } .quote.small { font-size:1.05rem; color:#d6e0f7; } .diff { font-family:'Amiri','Cairo',serif; font-size:1.1rem; line-height:2.1; } .w-extra { background:rgba(244,63,94,.25); border-radius:4px; padding:0 3px; text-decoration:line-through; } .w-missing { background:rgba(20,184,166,.28); border-radius:4px; padding:0 3px; } .legend { color:#8ea3cf; font-size:.82rem; margin:4px 0 0; } .legend .w-extra { text-decoration:none; } .tile.zero { opacity:.5; } .det-chip { text-align:center; color:#8ea3cf; font-size:.82rem; margin:2px 0 8px; } .det-chip b { color:#22d3ee; font-weight:600; } .hl-legend { display:flex; flex-wrap:wrap; gap:8px; margin-top:10px; font-size:.78rem; } .hl-legend mark { padding:2px 10px; } .flow { display:flex; flex-wrap:wrap; justify-content:center; align-items:center; gap:8px; margin:16px 0 4px; direction:rtl; } .flow span { background:#101a38; border:1px solid #1e2d57; border-radius:999px; padding:5px 14px; font-size:.85rem; color:#cfe0ff; } .flow span small { color:#7e93c0; margin-inline-start:6px; direction:ltr; display:inline-block; } .flow i { color:#22d3ee; font-style:normal; } .hero .mark { width:52px; height:52px; margin:0 auto 6px; display:block; } .hero h1 { background:linear-gradient(90deg,#fff,#a5f3fc); -webkit-background-clip:text; background-clip:text; color:transparent; } .action { border-radius:12px; padding:10px 14px; margin-top:10px; line-height:1.8; } .action.ok { background:rgba(20,184,166,.12); border:1px solid #14b8a6; } .action.warn { background:rgba(245,158,11,.12); border:1px solid #f59e0b; } .action.bad { background:rgba(244,63,94,.12); border:1px solid #f43f5e; } .evidence { margin-top:12px; border-top:1px dashed #2a3d73; padding-top:8px; } .evidence summary { cursor:pointer; color:#22d3ee; font-weight:700; } .ev-row { margin:10px 0; } .ev-row b { color:#9fb3da; font-size:.85rem; } .ev-row p { margin:2px 0; } .indicators { display:grid; gap:6px; margin-top:6px; } .ind { display:grid; grid-template-columns:150px 1fr 48px; gap:10px; align-items:center; font-size:.85rem; } .ind-name small { display:block; color:#7e93c0; font-size:.72rem; direction:ltr; text-align:right; } .bar { background:#0d1631; border-radius:999px; height:8px; overflow:hidden; } .bar span { display:block; height:100%; background:linear-gradient(90deg,#14b8a6,#22d3ee); } .ind-val { text-align:left; direction:ltr; color:#cfe0ff; } .ind-val.wide { grid-column:2 / span 2; text-align:right; } .notice { background:#101a38; border:1px solid #1e2d57; border-radius:14px; padding:16px; direction:rtl; line-height:1.9; } .notice.warn { border-color:#f59e0b; } .notice.bad { border-color:#f43f5e; } @media (max-width:640px){ .hero h1{font-size:1.7rem;} .ind{grid-template-columns:110px 1fr 40px;} } """.replace("PATTERN", _PATTERN) HERO = """
Islamic Content Verifier

مُدقِّق المحتوى الإسلامي

Verify Quranic and Hadith quotations with evidence.

الصق نصًا ولّده نموذج لغوي، وسيكتشف النظام الآيات والأحاديث الواردة فيه، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل. وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.

Paste text generated by a language model. The system detects Quran and Hadith quotations, retrieves the source texts, compares them word by word and shows the evidence. When evidence is insufficient it never invents a correction: it recommends human review.

كشفDetect←استرجاعRetrieve← تحققVerify←دليلEvidence←قرارDecide
""" DISCLAIMER = """
أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة. النتائج مبنية على مراجع القرآن والكتب الستة المضمّنة فقط. A text-verification aid, not a religious ruling and not a substitute for expert review. Results rely only on the bundled Quran and Six Books corpora.
""" PLACEHOLDER = "الصق هنا الرد الذي ولّده النموذج اللغوي، ويفضَّل أن تكون الاقتباسات بين علامات تنصيص بعد عبارة مثل «قال الله تعالى» أو «قال رسول الله ﷺ»…" def build_interface(): import gradio as gr examples = load_examples() def next_example(index: int): if not examples: return "", 0 return examples[index % len(examples)]["text"], (index + 1) % len(examples) with gr.Blocks(title="Islamic Content Verifier", css=CSS, theme=gr.themes.Base(primary_hue="teal", neutral_hue="slate")) as demo: gr.HTML(HERO) example_index = gr.State(0) text_input = gr.Textbox( label="النص المراد التحقق منه · Text to verify", lines=9, max_lines=24, placeholder=PLACEHOLDER, rtl=True, elem_classes="input-area", ) with gr.Row(elem_classes="btn-row"): verify_button = gr.Button("تحقق من النص · Verify Text", variant="primary", scale=3) example_button = gr.Button("جرّب مثالًا · Try an Example", variant="secondary", scale=2) results = gr.HTML(elem_classes="results") gr.HTML(DISCLAIMER) verify_button.click(verify_text, inputs=text_input, outputs=results) text_input.submit(verify_text, inputs=text_input, outputs=results) example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then( verify_text, inputs=text_input, outputs=results ) return demo def main() -> None: logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") get_pipeline() # build the index before serving the first request build_interface().queue().launch(share=os.environ.get("ICV_SHARE") == "1") # ICV_SHARE=1 -> temporary public link (e.g. on Colab) if __name__ == "__main__": main()