"""التحقق من هلوسة القرآن والحديث وتصحيحها: Gradio interface. python app.py # http://127.0.0.1:7860 Mode A verifies pasted text with the chosen detection engine. Mode B ("ask then verify") sends the question to a single pre-configured OpenAI client and verifies the answer; the key comes from the OPENAI_API_KEY environment variable only (never from the form, never from the repository). The in-browser page (build_static_space.py) shares these handlers. """ from __future__ import annotations import json import logging import os import threading from pathlib import Path from typing import List, Optional import ui from camelbert_adapter import analyze_with_spans, entities_to_spans, query_hosted_model, simulate_spans from llm_client import LLMError, generate from verifier import MAX_INPUT_CHARS, IslamicContentVerifier logger = logging.getLogger(__name__) EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json" # Detection engines offered in the UI (English labels by design; the first is the default). ENGINES = [ ("Standard · Rules + Corpus Scan", "standard"), ("CAMeLBERT-MSA · Fine-tuned (Hugging Face)", "camelbert"), ] _pipeline: Optional[IslamicContentVerifier] = None _lock = threading.Lock() def get_pipeline() -> IslamicContentVerifier: """Created once. The Quran index loads immediately; the Hadith index loads lazily (see ``warm_in_background``).""" global _pipeline with _lock: if _pipeline is None: _pipeline = IslamicContentVerifier() return _pipeline def warm_in_background() -> None: threading.Thread(target=lambda: get_pipeline().retriever.warm(), daemon=True).start() def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]: try: with open(path, encoding="utf-8") as handle: return json.load(handle) except (OSError, json.JSONDecodeError): logger.exception("Could not load demo examples from %s", path) return [] def _analyze(text: str, engine: str = "standard") -> dict: """Run the pipeline with the selected detection engine. The CAMeLBERT engine uses a hosted model when ``ICV_HF_MODEL`` is set and reachable; otherwise it runs as a simulation on top of the bundled detector (and says so).""" pipeline = get_pipeline() if engine != "camelbert": return pipeline.analyze(text) model = os.environ.get("ICV_HF_MODEL", "").strip() if model: try: spans = query_hosted_model(text, model, os.environ.get("HF_TOKEN", "")) return analyze_with_spans(pipeline, text, spans, engine="camelbert") except RuntimeError: logger.warning("Hosted CAMeLBERT model unavailable; using the simulation") return analyze_with_spans(pipeline, text, simulate_spans(pipeline, text), engine="camelbert-simulated") def verify_with_entities(text: str, entities_json: str, engine: str = "camelbert", generated: bool = False) -> str: """Browser path: the page called the hosted token-classification model and passes its raw entities here.""" try: spans = entities_to_spans(text, json.loads(entities_json or "[]")) result = analyze_with_spans(get_pipeline(), text, spans, engine=engine) return ui.render_results(result, generated_answer=text if generated else None) except ValueError: return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn") except Exception: logger.exception("Verification with external spans failed") return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad") def verify_text(text: str, engine: str = "standard") -> str: """Mode A. Never raises: problems become Arabic notices.""" if not text or not text.strip(): return ui.render_message("الرجاء إدخال نص للتحقق منه.", "warn") try: return ui.render_results(_analyze(text, engine)) except ValueError: return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn") except Exception: logger.exception("Verification failed") return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad") def verify_generated_answer(answer: str, engine: str = "standard") -> str: """Verify a model answer and show it above the report (also used by the in-browser page).""" try: return ui.render_results(_analyze(answer, engine), generated_answer=answer) except Exception: logger.exception("Verification of the generated answer failed") return ui.render_message("تعذّر التحقق من إجابة النموذج.", "bad") def ask_then_verify(prompt: str, engine: str = "standard") -> str: """Mode B: ask the pre-configured OpenAI client, then verify every quotation in its answer.""" try: answer = generate(None, prompt) except LLMError as exc: return ui.render_message(str(exc), "warn") return verify_generated_answer(answer, engine) def build_interface(): import gradio as gr examples = load_examples() def next_example(index: int): if not examples: return "", 0 return examples[index % len(examples)]["text"], (index + 1) % len(examples) theme = gr.themes.Base(primary_hue="emerald", neutral_hue="stone") with gr.Blocks(title=ui.APP_TITLE, css=ui.CSS, theme=theme, head=f"") as demo: gr.HTML(ui.HERO) with gr.Tabs(): with gr.Tab("تحقّق مباشر"): example_index = gr.State(0) engine = gr.Dropdown(choices=ENGINES, value="standard", label="Detection engine", elem_classes="engine-select") text_input = gr.Textbox(label="النص المراد التحقق منه", lines=9, max_lines=24, placeholder=ui.PLACEHOLDER, rtl=True, elem_classes="input-area") with gr.Row(): verify_button = gr.Button("تحقّق من النص", variant="primary", scale=3) example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2) results = gr.HTML(elem_classes="results") verify_button.click(verify_text, inputs=[text_input, engine], outputs=results) example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then( verify_text, inputs=[text_input, engine], outputs=results) with gr.Tab("اسأل ثم تحقّق"): gr.HTML('