fin09
Use stable Gradio demo.launch() with named api_name='summarize'
b4255e9
Raw History Blame Contribute Delete
5.39 kB
"""
Bilingual Arabic-English Text Summarization Suite
Production Hugging Face Space (Gradio 5.x + API Endpoint).
"""
import os
import sys
import time
from typing import Optional
PROJECT_ROOT = os.path.abspath(os.path.dirname(__file__))
if PROJECT_ROOT not in sys.path:
sys.path.insert(0, PROJECT_ROOT)
# ZeroGPU decorator support
try:
import spaces
has_spaces = True
except ImportError:
has_spaces = False
def gpu_decorator(func):
if has_spaces and hasattr(spaces, "GPU"):
return spaces.GPU(func)
return func
import gradio as gr
from nlp_core.language_detector import LanguageDetector
from nlp_core.tokenizer import BilingualTokenizer
from models.extractive.textrank import TextRankSummarizer
from models.extractive.lsa import LSASummarizer
from models.extractive.hybrid_scorer import HybridSummarizer
from models.abstractive.seq2seq_model import Seq2SeqSummarizer
from evaluation.metrics_manager import MetricsManager
lang_detector = LanguageDetector()
tokenizer = BilingualTokenizer()
metrics_mgr = MetricsManager()
CKPT_AR = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_arabic.pt")
CKPT_EN = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_english.pt")
seq_ar = Seq2SeqSummarizer.load_checkpoint(CKPT_AR, device="cpu") if os.path.exists(CKPT_AR) else None
seq_en = Seq2SeqSummarizer.load_checkpoint(CKPT_EN, device="cpu") if os.path.exists(CKPT_EN) else None
@gpu_decorator
def summarize_text(text, algorithm="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)", sentences=3, beam_width=3):
"""Core summarization engine exposed to both Gradio UI and REST API."""
if not text or not str(text).strip():
return "Please enter text to summarize.", "N/A", "0%", "0 ms"
text = str(text).strip()
t0 = time.time()
detected_lang = lang_detector.detect_language(text)
if "lsa" in str(algorithm).lower():
res = LSASummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang)
summary = res["summary"]
elif "hybrid" in str(algorithm).lower():
res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang)
summary = res["summary"]
elif "textrank" in str(algorithm).lower():
res = TextRankSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang)
summary = res["summary"]
else: # Seq2Seq Abstractive
model = seq_ar if detected_lang == "ar" else seq_en
if model:
toks = tokenizer.tokenize_words(text, lang=detected_lang)
gen_toks = model.summarize_beam(toks, beam_width=int(beam_width), max_len=60)
summary = " ".join(gen_toks)
else:
res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang)
summary = res["summary"] + " (Used Hybrid fallback)"
latency = f"{(time.time() - t0)*1000.0:.1f} ms"
orig_w = len(text.split())
sum_w = len(summary.split())
reduction = f"{(1 - (sum_w/orig_w))*100:.1f}%" if orig_w > 0 else "0%"
return summary, detected_lang.upper(), reduction, latency
# Build Gradio UI with named API endpoints
with gr.Blocks(title="Bilingual Arabic-English Text Summarization AI", theme=gr.themes.Soft()) as demo:
gr.Markdown("# 📖 Bilingual Arabic-English Text Summarization AI System")
gr.Markdown("""
**Production NLP Platform** supporting Extractive (TextRank, LSA, Hybrid) and Abstractive (Seq2Seq + Bahdanau Attention) summarization.
""")
with gr.Row():
with gr.Column():
input_text = gr.Textbox(
label="Source Document Text | النص الأصلي",
placeholder="Paste Arabic or English article text here...",
lines=10
)
algorithm = gr.Dropdown(
label="Summarization Algorithm | خوارزمية التلخيص",
choices=[
"Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)",
"Hybrid Scorer (Extractive)",
"TextRank (Extractive)",
"LSA (Extractive)"
],
value="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)"
)
with gr.Row():
sentences = gr.Slider(minimum=1, maximum=10, value=3, step=1, label="Sentences (Extractive)")
beam_width = gr.Slider(minimum=1, maximum=6, value=3, step=1, label="Beam Width (Seq2Seq)")
btn = gr.Button("✨ Generate Summary | توليد الملخص", variant="primary")
with gr.Column():
output_summary = gr.Textbox(label="Generated Summary | الملخص الناتج", lines=8)
with gr.Row():
out_lang = gr.Textbox(label="Detected Language", lines=1)
out_reduc = gr.Textbox(label="Word Reduction", lines=1)
out_lat = gr.Textbox(label="Latency (ms)", lines=1)
btn.click(
fn=summarize_text,
inputs=[input_text, algorithm, sentences, beam_width],
outputs=[output_summary, out_lang, out_reduc, out_lat],
api_name="summarize"
)
# Launch Gradio app directly (required for Hugging Face Space supervisor)
demo.queue()
demo.launch(server_name="0.0.0.0", server_port=7860, ssr_mode=False)