File size: 8,277 Bytes
597dbb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
"""التحقق من هلوسة القرآن والحديث وتصحيحها: Gradio interface.

    python app.py                      # http://127.0.0.1:7860

Mode A verifies pasted text with the chosen detection engine. Mode B ("ask then verify") sends the question to a single
pre-configured OpenAI client and verifies the answer; the key comes from the OPENAI_API_KEY environment variable only
(never from the form, never from the repository). The in-browser page (build_static_space.py) shares these handlers.
"""
from __future__ import annotations

import json
import logging
import os
import threading
from pathlib import Path
from typing import List, Optional

import ui
from camelbert_adapter import analyze_with_spans, entities_to_spans, query_hosted_model, simulate_spans
from llm_client import LLMError, generate
from verifier import MAX_INPUT_CHARS, IslamicContentVerifier

logger = logging.getLogger(__name__)

EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json"

# Detection engines offered in the UI (English labels by design; the first is the default).
ENGINES = [
    ("Standard · Rules + Corpus Scan", "standard"),
    ("CAMeLBERT-MSA · Fine-tuned (Hugging Face)", "camelbert"),
]

_pipeline: Optional[IslamicContentVerifier] = None
_lock = threading.Lock()


def get_pipeline() -> IslamicContentVerifier:
    """Created once. The Quran index loads immediately; the Hadith index loads lazily (see ``warm_in_background``)."""
    global _pipeline
    with _lock:
        if _pipeline is None:
            _pipeline = IslamicContentVerifier()
        return _pipeline


def warm_in_background() -> None:
    threading.Thread(target=lambda: get_pipeline().retriever.warm(), daemon=True).start()


def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]:
    try:
        with open(path, encoding="utf-8") as handle:
            return json.load(handle)
    except (OSError, json.JSONDecodeError):
        logger.exception("Could not load demo examples from %s", path)
        return []


def _analyze(text: str, engine: str = "standard") -> dict:
    """Run the pipeline with the selected detection engine. The CAMeLBERT engine uses a hosted model when ``ICV_HF_MODEL``
    is set and reachable; otherwise it runs as a simulation on top of the bundled detector (and says so)."""
    pipeline = get_pipeline()
    if engine != "camelbert":
        return pipeline.analyze(text)
    model = os.environ.get("ICV_HF_MODEL", "").strip()
    if model:
        try:
            spans = query_hosted_model(text, model, os.environ.get("HF_TOKEN", ""))
            return analyze_with_spans(pipeline, text, spans, engine="camelbert")
        except RuntimeError:
            logger.warning("Hosted CAMeLBERT model unavailable; using the simulation")
    return analyze_with_spans(pipeline, text, simulate_spans(pipeline, text), engine="camelbert-simulated")


def verify_with_entities(text: str, entities_json: str, engine: str = "camelbert", generated: bool = False) -> str:
    """Browser path: the page called the hosted token-classification model and passes its raw entities here."""
    try:
        spans = entities_to_spans(text, json.loads(entities_json or "[]"))
        result = analyze_with_spans(get_pipeline(), text, spans, engine=engine)
        return ui.render_results(result, generated_answer=text if generated else None)
    except ValueError:
        return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
    except Exception:
        logger.exception("Verification with external spans failed")
        return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad")


def verify_text(text: str, engine: str = "standard") -> str:
    """Mode A. Never raises: problems become Arabic notices."""
    if not text or not text.strip():
        return ui.render_message("الرجاء إدخال نص للتحقق منه.", "warn")
    try:
        return ui.render_results(_analyze(text, engine))
    except ValueError:
        return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
    except Exception:
        logger.exception("Verification failed")
        return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad")


def verify_generated_answer(answer: str, engine: str = "standard") -> str:
    """Verify a model answer and show it above the report (also used by the in-browser page)."""
    try:
        return ui.render_results(_analyze(answer, engine), generated_answer=answer)
    except Exception:
        logger.exception("Verification of the generated answer failed")
        return ui.render_message("تعذّر التحقق من إجابة النموذج.", "bad")


def ask_then_verify(prompt: str, engine: str = "standard") -> str:
    """Mode B: ask the pre-configured OpenAI client, then verify every quotation in its answer."""
    try:
        answer = generate(None, prompt)
    except LLMError as exc:
        return ui.render_message(str(exc), "warn")
    return verify_generated_answer(answer, engine)


def build_interface():
    import gradio as gr

    examples = load_examples()

    def next_example(index: int):
        if not examples:
            return "", 0
        return examples[index % len(examples)]["text"], (index + 1) % len(examples)

    theme = gr.themes.Base(primary_hue="emerald", neutral_hue="stone")
    with gr.Blocks(title=ui.APP_TITLE, css=ui.CSS, theme=theme, head=f"<script>{ui.COPY_JS}</script>") as demo:
        gr.HTML(ui.HERO)
        with gr.Tabs():
            with gr.Tab("تحقّق مباشر"):
                example_index = gr.State(0)
                engine = gr.Dropdown(choices=ENGINES, value="standard", label="Detection engine", elem_classes="engine-select")
                text_input = gr.Textbox(label="النص المراد التحقق منه", lines=9, max_lines=24, placeholder=ui.PLACEHOLDER,
                                        rtl=True, elem_classes="input-area")
                with gr.Row():
                    verify_button = gr.Button("تحقّق من النص", variant="primary", scale=3)
                    example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2)
                results = gr.HTML(elem_classes="results")
                verify_button.click(verify_text, inputs=[text_input, engine], outputs=results)
                example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then(
                    verify_text, inputs=[text_input, engine], outputs=results)
            with gr.Tab("اسأل ثم تحقّق"):
                gr.HTML('<div class="icv"><div class="notice">اكتب سؤالًا، وسيجيب عنه النظام مباشرةً، ثم يفحص كل آية '
                        'وحديث ورد في الإجابة ويعرض الأخطاء والتصحيحات.</div></div>')
                prompt = gr.Textbox(label="سؤالك", lines=3, placeholder=ui.PROMPT_PLACEHOLDER, rtl=True, elem_classes="input-area")
                ask_button = gr.Button("اسأل ثم تحقّق", variant="primary")
                answer_results = gr.HTML(elem_classes="results")
                ask_button.click(ask_then_verify, inputs=[prompt, engine], outputs=answer_results)
        gr.HTML(ui.DISCLAIMER)
    return demo


def _load_dotenv(path: Path = Path(__file__).resolve().parent / ".env") -> None:
    """Minimal ``.env`` reader for local runs (no extra dependency); existing environment variables win."""
    try:
        lines = path.read_text(encoding="utf-8").splitlines()
    except OSError:
        return
    for line in lines:
        name, sep, value = line.strip().partition("=")
        if sep and name and not name.startswith("#") and value.strip():
            os.environ.setdefault(name.strip(), value.strip().strip('"').strip("'"))


def main() -> None:
    logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
    _load_dotenv()
    get_pipeline()
    warm_in_background()
    build_interface().queue().launch(share=os.environ.get("ICV_SHARE") == "1")


if __name__ == "__main__":
    main()