Spaces:
Running on Zero
Running on Zero
Download app.py from fady21/bilingual-summarizer-api: direct link, hf CLI and curl.
- Browser
- Download file 5.39 kB
-
https://huggingface.co/spaces/fady21/bilingual-summarizer-api/resolve/main/app.py
- Command line
-
hf download hf://spaces/fady21/bilingual-summarizer-api/app.py
-
curl -L -o app.py https://huggingface.co/spaces/fady21/bilingual-summarizer-api/resolve/main/app.py
5.39 kB
| """ | |
| Bilingual Arabic-English Text Summarization Suite | |
| Production Hugging Face Space (Gradio 5.x + API Endpoint). | |
| """ | |
| import os | |
| import sys | |
| import time | |
| from typing import Optional | |
| PROJECT_ROOT = os.path.abspath(os.path.dirname(__file__)) | |
| if PROJECT_ROOT not in sys.path: | |
| sys.path.insert(0, PROJECT_ROOT) | |
| # ZeroGPU decorator support | |
| try: | |
| import spaces | |
| has_spaces = True | |
| except ImportError: | |
| has_spaces = False | |
| def gpu_decorator(func): | |
| if has_spaces and hasattr(spaces, "GPU"): | |
| return spaces.GPU(func) | |
| return func | |
| import gradio as gr | |
| from nlp_core.language_detector import LanguageDetector | |
| from nlp_core.tokenizer import BilingualTokenizer | |
| from models.extractive.textrank import TextRankSummarizer | |
| from models.extractive.lsa import LSASummarizer | |
| from models.extractive.hybrid_scorer import HybridSummarizer | |
| from models.abstractive.seq2seq_model import Seq2SeqSummarizer | |
| from evaluation.metrics_manager import MetricsManager | |
| lang_detector = LanguageDetector() | |
| tokenizer = BilingualTokenizer() | |
| metrics_mgr = MetricsManager() | |
| CKPT_AR = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_arabic.pt") | |
| CKPT_EN = os.path.join(PROJECT_ROOT, "checkpoints", "seq2seq_english.pt") | |
| seq_ar = Seq2SeqSummarizer.load_checkpoint(CKPT_AR, device="cpu") if os.path.exists(CKPT_AR) else None | |
| seq_en = Seq2SeqSummarizer.load_checkpoint(CKPT_EN, device="cpu") if os.path.exists(CKPT_EN) else None | |
| def summarize_text(text, algorithm="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)", sentences=3, beam_width=3): | |
| """Core summarization engine exposed to both Gradio UI and REST API.""" | |
| if not text or not str(text).strip(): | |
| return "Please enter text to summarize.", "N/A", "0%", "0 ms" | |
| text = str(text).strip() | |
| t0 = time.time() | |
| detected_lang = lang_detector.detect_language(text) | |
| if "lsa" in str(algorithm).lower(): | |
| res = LSASummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) | |
| summary = res["summary"] | |
| elif "hybrid" in str(algorithm).lower(): | |
| res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) | |
| summary = res["summary"] | |
| elif "textrank" in str(algorithm).lower(): | |
| res = TextRankSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) | |
| summary = res["summary"] | |
| else: # Seq2Seq Abstractive | |
| model = seq_ar if detected_lang == "ar" else seq_en | |
| if model: | |
| toks = tokenizer.tokenize_words(text, lang=detected_lang) | |
| gen_toks = model.summarize_beam(toks, beam_width=int(beam_width), max_len=60) | |
| summary = " ".join(gen_toks) | |
| else: | |
| res = HybridSummarizer().summarize(text, num_sentences=int(sentences), lang=detected_lang) | |
| summary = res["summary"] + " (Used Hybrid fallback)" | |
| latency = f"{(time.time() - t0)*1000.0:.1f} ms" | |
| orig_w = len(text.split()) | |
| sum_w = len(summary.split()) | |
| reduction = f"{(1 - (sum_w/orig_w))*100:.1f}%" if orig_w > 0 else "0%" | |
| return summary, detected_lang.upper(), reduction, latency | |
| # Build Gradio UI with named API endpoints | |
| with gr.Blocks(title="Bilingual Arabic-English Text Summarization AI", theme=gr.themes.Soft()) as demo: | |
| gr.Markdown("# 📖 Bilingual Arabic-English Text Summarization AI System") | |
| gr.Markdown(""" | |
| **Production NLP Platform** supporting Extractive (TextRank, LSA, Hybrid) and Abstractive (Seq2Seq + Bahdanau Attention) summarization. | |
| """) | |
| with gr.Row(): | |
| with gr.Column(): | |
| input_text = gr.Textbox( | |
| label="Source Document Text | النص الأصلي", | |
| placeholder="Paste Arabic or English article text here...", | |
| lines=10 | |
| ) | |
| algorithm = gr.Dropdown( | |
| label="Summarization Algorithm | خوارزمية التلخيص", | |
| choices=[ | |
| "Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)", | |
| "Hybrid Scorer (Extractive)", | |
| "TextRank (Extractive)", | |
| "LSA (Extractive)" | |
| ], | |
| value="Seq2Seq with Bahdanau Attention (Abstractive 20-Ep)" | |
| ) | |
| with gr.Row(): | |
| sentences = gr.Slider(minimum=1, maximum=10, value=3, step=1, label="Sentences (Extractive)") | |
| beam_width = gr.Slider(minimum=1, maximum=6, value=3, step=1, label="Beam Width (Seq2Seq)") | |
| btn = gr.Button("✨ Generate Summary | توليد الملخص", variant="primary") | |
| with gr.Column(): | |
| output_summary = gr.Textbox(label="Generated Summary | الملخص الناتج", lines=8) | |
| with gr.Row(): | |
| out_lang = gr.Textbox(label="Detected Language", lines=1) | |
| out_reduc = gr.Textbox(label="Word Reduction", lines=1) | |
| out_lat = gr.Textbox(label="Latency (ms)", lines=1) | |
| btn.click( | |
| fn=summarize_text, | |
| inputs=[input_text, algorithm, sentences, beam_width], | |
| outputs=[output_summary, out_lang, out_reduc, out_lat], | |
| api_name="summarize" | |
| ) | |
| # Launch Gradio app directly (required for Hugging Face Space supervisor) | |
| demo.queue() | |
| demo.launch(server_name="0.0.0.0", server_port=7860, ssr_mode=False) | |