""" Bilingual Arabic-English Text Summarization Suite Production Hugging Face Space (Gradio 5.x + API Endpoint). """ import os import sys import time from typing import Optional PROJECT_ROOT = os.path.abspath(os.path.dirname(__file__)) if PROJECT_ROOT not in sys.path: sys.path.insert(0, PROJECT_ROOT) # ZeroGPU decorator support try: import spaces has_spaces = True except ImportError: has_spaces = False def gpu_decorator(func): if has_spaces and hasattr(spaces, "GPU"): return spaces.GPU(func) return func import gradio as gr from nlp_core.language_detector import LanguageDetector from nlp_core.tokenizer import BilingualTokenizer from models.extractive.textrank import TextRankSummarizer from models.extractive.lsa import LSASummarizer from models.extractive.hybrid_scorer import HybridSummarizer from models.abstractive.seq2seq_model import Seq2SeqSummarizer from evaluation.metrics_manager import MetricsManager lang_detector = LanguageDetector() tokenizer = BilingualTokenizer() metrics_mgr = MetricsManager() CKPT_AR = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_arabic.pt") CKPT_EN = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_english.pt") seq_ar = Seq2SeqSummarizer.load_checkpoint(CKPT_AR, device="cpu") if os.path.exists(CKPT_AR) else None seq_en = Seq2SeqSummarizer.load_checkpoint(CKPT_EN, device="cpu") if os.path.exists(CKPT_EN) else None @gpu_decorator def summarize_text(text, algorithm="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)", sentences=3, beam_width=3): """Core summarization engine exposed to both Gradio UI and REST API.""" if not text or not str(text).strip(): return "Please enter text to summarize.", "N/A", "0%", "0 ms" text = str(text).strip() t0 = time.time() detected_lang = lang_detector.detect_language(text) if "lsa" in str(algorithm).lower(): res = LSASummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) summary = res["summary"] elif "hybrid" in str(algorithm).lower(): res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) summary = res["summary"] elif "textrank" in str(algorithm).lower(): res = TextRankSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) summary = res["summary"] else: # Seq2Seq Abstractive model = seq_ar if detected_lang == "ar" else seq_en if model: toks = tokenizer.tokenize_words(text, lang=detected_lang) gen_toks = model.summarize_beam(toks, beam_width=int(beam_width), max_len=60) summary = " ".join(gen_toks) else: res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) summary = res["summary"] + " (Used Hybrid fallback)" latency = f"{(time.time() - t0)*1000.0:.1f} ms" orig_w = len(text.split()) sum_w = len(summary.split()) reduction = f"{(1 - (sum_w/orig_w))*100:.1f}%" if orig_w > 0 else "0%" return summary, detected_lang.upper(), reduction, latency # Build Gradio UI with named API endpoints with gr.Blocks(title="Bilingual Arabic-English Text Summarization AI", theme=gr.themes.Soft()) as demo: gr.Markdown("# 📖 Bilingual Arabic-English Text Summarization AI System") gr.Markdown(""" **Production NLP Platform** supporting Extractive (TextRank, LSA, Hybrid) and Abstractive (Seq2Seq + Bahdanau Attention) summarization. """) with gr.Row(): with gr.Column(): input_text = gr.Textbox( label="Source Document Text | النص الأصلي", placeholder="Paste Arabic or English article text here...", lines=10 ) algorithm = gr.Dropdown( label="Summarization Algorithm | خوارزمية التلخيص", choices=[ "Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)", "Hybrid Scorer (Extractive)", "TextRank (Extractive)", "LSA (Extractive)" ], value="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)" ) with gr.Row(): sentences = gr.Slider(minimum=1, maximum=10, value=3, step=1, label="Sentences (Extractive)") beam_width = gr.Slider(minimum=1, maximum=6, value=3, step=1, label="Beam Width (Seq2Seq)") btn = gr.Button("✨ Generate Summary | توليد الملخص", variant="primary") with gr.Column(): output_summary = gr.Textbox(label="Generated Summary | الملخص الناتج", lines=8) with gr.Row(): out_lang = gr.Textbox(label="Detected Language", lines=1) out_reduc = gr.Textbox(label="Word Reduction", lines=1) out_lat = gr.Textbox(label="Latency (ms)", lines=1) btn.click( fn=summarize_text, inputs=[input_text, algorithm, sentences, beam_width], outputs=[output_summary, out_lang, out_reduc, out_lat], api_name="summarize" ) # Launch Gradio app directly (required for Hugging Face Space supervisor) demo.queue() demo.launch(server_name="0.0.0.0", server_port=7860, ssr_mode=False)