"""Islamic Content Verifier: Gradio interface around the verification pipeline in ``verifier.py``.
Run locally with ``python app.py``. The rendering helpers are plain functions that return HTML, so they can be
tested without Gradio; Gradio is imported only when the interface is built.
"""
from __future__ import annotations
import html
import json
import logging
import os
import threading
from pathlib import Path
from typing import List, Optional
from verifier import MAX_INPUT_CHARS, IslamicContentVerifier
logger = logging.getLogger(__name__)
EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json"
DEFAULT_MODEL_DIR = Path(__file__).resolve().parent / "models" / "span_detector"
# --------------------------------------------------------------------------------------------------------------
# Pipeline access
# --------------------------------------------------------------------------------------------------------------
_pipeline: Optional[IslamicContentVerifier] = None
_pipeline_lock = threading.Lock()
def get_pipeline() -> IslamicContentVerifier:
"""Create the pipeline once (the corpus index builds in a few seconds) and reuse it for every request."""
global _pipeline
with _pipeline_lock:
if _pipeline is None:
model_dir = os.environ.get("ICV_DETECTOR_MODEL") or (str(DEFAULT_MODEL_DIR) if DEFAULT_MODEL_DIR.is_dir() else None)
# With a trained model folder the BERT detector is used; if it cannot be loaded the rules detector takes over.
_pipeline = IslamicContentVerifier(detector="auto" if model_dir else "rules", model_dir=model_dir)
return _pipeline
def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]:
try:
with open(path, encoding="utf-8") as handle:
return json.load(handle)
except (OSError, json.JSONDecodeError):
logger.exception("Could not load demo examples from %s", path)
return []
# --------------------------------------------------------------------------------------------------------------
# HTML rendering
# --------------------------------------------------------------------------------------------------------------
_e = html.escape
GROUP_LABEL = {
"verified": ("موثّق", "Verified"),
"mismatch": ("غير مطابق", "Mismatch"),
"review": ("يحتاج مراجعة بشرية", "Needs Human Review"),
}
TYPE_LABEL = {"Ayah": ("آية قرآنية", "Quran"), "Hadith": ("حديث نبوي", "Hadith")}
INDICATORS = [
("composite", "الدرجة المركبة", "Composite score"),
("coverage", "تغطية الكلمات", "Word coverage"),
("lcs_ratio", "التسلسل النصي", "Sequence (LCS)"),
("token_overlap", "تداخل الكلمات", "Token overlap"),
("edit_sim", "تشابه الأحرف", "Character similarity"),
("diacritic_sim", "مع التشكيل", "With diacritics"),
]
def _pct(value: float) -> str:
return f"{round(value * 100)}%"
def _source_label(source: dict) -> str:
if source["type"] == "Quran":
start = source.get("ayah", source.get("ayah_start"))
end = source.get("ayah_end", start)
verses = f"{start}" if start == end else f"{start}–{end}"
return f"سورة {source['surah_name']} — الآية {verses}"
return f"حديث رقم {source['hadithID']} — {source['title']}"
def _truncate(text: str, limit: int) -> str:
return text if len(text) <= limit else text[:limit].rstrip() + " …"
def _diff_html(comparison: dict) -> str:
"""Word-level diff: green = words only in the source, red = words only in the quotation."""
parts = []
for op in comparison["word_diff"]:
if op["op"] == "equal":
parts.append(f'{_e(op["span"])}')
else:
if op["span"]:
parts.append(f'{_e(op["span"])}')
if op["source"]:
parts.append(f'{_e(op["source"])}')
return " ".join(parts)
def _indicator_table(signals: Optional[dict]) -> str:
if not signals:
return ""
rows = []
for key, ar, en in INDICATORS:
value = float(signals.get(key, 0.0))
rows.append(
f'
تصحيح مدعوم بالمصدر · Source-backed correction'
f'
{_e(correction["display_text"])}
'
f'
{_e(_source_label({"type": "Quran", **correction["source"]}) if correction["source"]["type"] == "Quran" else _source_label(correction["source"]))}'
f' · قوة المطابقة {_pct(correction["match_strength"])}
'
)
elif group == "review":
closest = ""
if suggestion:
src = suggestion["source"]
label = _source_label({"type": "Quran", **src}) if src["type"] == "Quran" else _source_label(src)
closest = (f'
أقرب مصدر وُجد للمراجِع (ليس تصحيحًا آليًا) · Closest source for the reviewer (not an automatic correction): '
f'{_e(label)}
{_e(_truncate(suggestion["display_text"], 360))}
')
action = (
'
الأدلة غير كافية للتصحيح الآلي. يُوصى بالمراجعة البشرية.'
'
Insufficient evidence for automatic correction. Human review is recommended.
موثّق · Verifiedغير مطابق · Mismatch'
'مراجعة بشرية · Review
'
)
return (
'
'
+ "".join(pieces) + "
" + legend + "
"
)
def render_results(result: dict) -> str:
"""HTML report for a pipeline result."""
if not result["spans"]:
return (
'
لم يُعثر على اقتباسات قرآنية أو حديثية في هذا النص. '
'No Quranic or Hadith quotations were detected. Quotations are currently detected when they are '
'enclosed in quotation marks or brackets and introduced by a typical phrase (see Limitations).
Verify Quranic and Hadith quotations with evidence.
الصق نصًا ولّده نموذج لغوي، وسيكتشف النظام الآيات والأحاديث الواردة فيه، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة،
ثم يعرض الدليل. وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.
Paste text generated by a language model. The system detects Quran and Hadith quotations, retrieves the source texts,
compares them word by word and shows the evidence. When evidence is insufficient it never invents a correction: it recommends human review.
أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة. النتائج مبنية على مراجع القرآن والكتب الستة المضمّنة فقط.
A text-verification aid, not a religious ruling and not a substitute for expert review. Results rely only on the bundled Quran and Six Books corpora.
"""
PLACEHOLDER = "الصق هنا الرد الذي ولّده النموذج اللغوي، ويفضَّل أن تكون الاقتباسات بين علامات تنصيص بعد عبارة مثل «قال الله تعالى» أو «قال رسول الله ﷺ»…"
def build_interface():
import gradio as gr
examples = load_examples()
def next_example(index: int):
if not examples:
return "", 0
return examples[index % len(examples)]["text"], (index + 1) % len(examples)
with gr.Blocks(title="Islamic Content Verifier", css=CSS, theme=gr.themes.Base(primary_hue="teal", neutral_hue="slate")) as demo:
gr.HTML(HERO)
example_index = gr.State(0)
text_input = gr.Textbox(
label="النص المراد التحقق منه · Text to verify", lines=9, max_lines=24, placeholder=PLACEHOLDER,
rtl=True, elem_classes="input-area",
)
with gr.Row(elem_classes="btn-row"):
verify_button = gr.Button("تحقق من النص · Verify Text", variant="primary", scale=3)
example_button = gr.Button("جرّب مثالًا · Try an Example", variant="secondary", scale=2)
results = gr.HTML(elem_classes="results")
gr.HTML(DISCLAIMER)
verify_button.click(verify_text, inputs=text_input, outputs=results)
text_input.submit(verify_text, inputs=text_input, outputs=results)
example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then(
verify_text, inputs=text_input, outputs=results
)
return demo
def main() -> None:
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
get_pipeline() # build the index before serving the first request
build_interface().queue().launch(share=os.environ.get("ICV_SHARE") == "1") # ICV_SHARE=1 -> temporary public link (e.g. on Colab)
if __name__ == "__main__":
main()