bilingual-summarizer-api / training /sample_data_gen.py
fin09
Deploy Bilingual Summarization NLP Suite with Git LFS
3d9ba5b
Raw History Blame Contribute Delete
13.9 kB
import json
import os
from typing import List, Dict, Optional
import pandas as pd
# -----------------------------------------------------------------------------
# 1. ARABIC DATASET (Aligned with Kaggle: Arabic Text Summarization / EASC Corpus)
# Source: https://www.kaggle.com/datasets/abdalrahmanshahrour/arabic-text-summarization
# -----------------------------------------------------------------------------
KAGGLE_ARABIC_DATA = [
{
"article": "تعد معالجة اللغات الطبيعية فرعا جوهريا من فروع الذكاء الاصطناعي يهدف إلى جعل الحواسيب قادرة على فهم النصوص البشرية وتوليدها بفعالية وتسهيل استخراج المعلومات الدقيقة من كميات البيانات الضخمة.",
"summary": "معالجة اللغات الطبيعية تمكن الحواسيب من فهم النصوص البشرية وتوليدها واستخراج المعلومات."
},
{
"article": "يقوم التلخيص الاستخراجي باختيار أهم الجمل من النص الأصلي بالاعتماد على خوارزميات إحصائية مثل تكست رانك والتحليل الدلالي الكامن مع الحفاظ الكامل على الصياغة والتراكيب اللغوية الأصلية.",
"summary": "التلخيص الاستخراجي يحدد أهم الجمل الأصلية باستخدام خوارزميات إحصائية ورسوم بيانية."
},
{
"article": "يعتمد التلخيص التجريدي على نماذج التعلم العميق وشبكات الانتباه والمحولات لتوليد ملخصات جديدة تصوغ الأفكار الرئيسية بكلمات مبتكرة ودقيقة تحاكي أسلوب التلخيص البشري الذكي.",
"summary": "التلخيص التجريدي يستخدم التعلم العميق وشبكات الانتباه لإعادة صياغة الأفكار الرئيسية بكلمات جديدة."
},
{
"article": "تتميز اللغة العربية بثرائها الصرفي الشديد ووجود علامات التشكيل وتنوع أشكال الحروف مما يتطلب مراحل معالجة أولية متقدمة تشمل إزالة التشكيل وتوحيد الهمزات والتجذير قبل تطبيق خوارزميات التلخيص.",
"summary": "التعقيد الصرفي للغة العربية يتطلب معالجة لغوية أولية متقدمة لتوحيد الحروف والتجذير."
},
{
"article": "تساعد مقاييس روج وبلو في تقييم جودة الملخصات التلقائية عبر مقارنتها بالملخصات المرجعية وقياس معدلات التداخل في الكلمات المتتالية والدقة والاكتمال اللغوي ونسبة الإيجاز والضغط.",
"summary": "تقيس مؤشرات روج وبلو جودة التلخيص بمقارنته مع ملخصات مرجعية وحساب التداخل والدقة."
},
{
"article": "يساهم الذكاء الاصطناعي في تطوير قطاع الرعاية الصحية والطب من خلال تشخيص الأمراض بدقة وسرعة فائقة وتحليل الصور الطبية والأشعة السينية واكتشاف الأدوية الحيوية الجديدة.",
"summary": "يعزز الذكاء الاصطناعي الرعاية الصحية بتشخيص الأمراض وتحليل الصور الطبية وتطوير الأدوية."
},
{
"article": "تعتبر الطاقة المتجددة مثل الطاقة الشمسية وطاقة الرياح حلا مستداما واستراتيجيا لمواجهة أزمة التغير المناخي والحد من انبعاثات الغازات الدفيئة والكربون الضارة بالبيئة في جميع أنحاء العالم.",
"summary": "توفر الطاقة المتجددة مثل الشمس والرياح حلا بيئيا مستداما للحد من الانبعاثات ومكافحة التغير المناخي."
},
{
"article": "شهدت التجارة الإلكترونية نموا هائلا على مستوى العالم بفضل انتشار الهواتف الذكية وتطور بوابات الدفع الإلكتروني الآمنة وتحسين الخدمات اللوجستية وتوصيل البضائع السريع للمستهلكين.",
"summary": "شهدت التجارة الإلكترونية نموا واسعا مدفوعا بالهواتف الذكية وأنظمة الدفع الإلكتروني والخدمات اللوجستية."
},
{
"article": "أعلنت وكالات الفضاء الدولية عن إطلاق مهمات استكشافية جديدة تهدف إلى دراسة الغلاف الجوي للكواكب والبحث عن المياه الجوفية على سطح المريخ باستخدام مركبات ذاتية القيادة.",
"summary": "أطلقت وكالات الفضاء مهمات لاستكشاف المريخ والبحث عن المياه بواسطة مركبات ذاتية القيادة."
},
{
"article": "حققت تقنيات الأمن السيبراني قفزة نوعية في حماية البنى التحتية الرقمية والتصدي للهجمات الإلكترونية المعقدة عبر تطبيق خوارزميات التحليل التنبؤي والتشفير الكمي المتقدم.",
"summary": "يطور الأمن السيبراني حماية البنى التحتية الرقمية ضد الهجمات عبر التشفير المتقدم والتحليل التنبؤي."
},
{
"article": "تسعى المدن الذكية الحديثة إلى دمج إنترنت الأشياء والذكاء الاصطناعي لإدارة شبكات النقل والطاقة بكفاءة وتقليل الازدحام المروري وتحسين جودة حياة المواطنين.",
"summary": "توظف المدن الذكية إنترنت الأشياء والذكاء الاصطناعي لتحسين النقل والطاقة وتقليل الازدحام."
},
{
"article": "أظهرت الدراسات الاقتصادية الحديثة أن التحول الرقمي يسهم في رفع الإنتاجية بنسبة ملحوظة وخلق فرص عمل جديدة في قطاعات البرمجيات والبيانات والخدمات السحابية.",
"summary": "يؤكد التحول الرقمي قدرته على رفع الإنتاجية وتوفير وظائف جديدة في التكنولوجيا والسحابة."
}
]
# -----------------------------------------------------------------------------
# 2. ENGLISH DATASET (Aligned with Kaggle: BBC News Summary Dataset)
# Source: https://www.kaggle.com/datasets/parizasch/bbc-news-summary
# -----------------------------------------------------------------------------
KAGGLE_ENGLISH_DATA = [
{
"article": "Natural language processing is an essential branch of artificial intelligence enabling computing systems to comprehend, process, and synthesize human language seamlessly across diverse domains.",
"summary": "Natural language processing enables computers to understand, process, and generate human language seamlessly."
},
{
"article": "Extractive text summarization identifies and pulls key sentences directly from the original document using statistical heuristics and graph algorithms like TextRank, LSA, and TF-IDF.",
"summary": "Extractive summarization selects vital original sentences using statistical heuristics and graph algorithms."
},
{
"article": "Abstractive summarization leverages deep neural networks, sequence to sequence architectures, and attention mechanisms to rephrase core concepts into concise and coherent novel text.",
"summary": "Abstractive summarization uses deep learning and attention mechanisms to generate concise novel summaries."
},
{
"article": "Bilingual natural language systems must account for profound structural and morphological disparities between rich non-concatenative Arabic and fixed-syntax analytic English.",
"summary": "Bilingual NLP bridges fundamental structural and morphological differences between Arabic and English."
},
{
"article": "Standard evaluation metrics such as ROUGE and BLEU calculate n-gram overlaps, precision, recall, and brevity penalties to quantify summary quality against reference standards.",
"summary": "ROUGE and BLEU benchmark summary quality by computing precision, recall, and n-gram overlap."
},
{
"article": "Artificial intelligence is transforming modern healthcare through accelerated early disease diagnosis, high-throughput medical imaging analysis, and generative drug discovery.",
"summary": "AI transforms modern healthcare through rapid disease diagnosis, medical imaging, and drug discovery."
},
{
"article": "Renewable energy sources such as solar arrays and offshore wind farms provide scalable solutions to reduce global carbon emissions and combat climate change effectively.",
"summary": "Renewable energy solutions such as solar and wind power mitigate climate change by lowering carbon emissions."
},
{
"article": "Global e-commerce has expanded rapidly driven by ubiquitous mobile devices, secure payment gateways, and automated supply chain fulfillment networks.",
"summary": "E-commerce expansion is accelerated by mobile technology, secure payment systems, and automated logistics."
},
{
"article": "International space agencies have successfully deployed robotic probes equipped with multispectral sensors to analyze Martian atmospheric dynamics and subsurface ice formations.",
"summary": "Space agencies deployed robotic probes to investigate atmospheric conditions and ice on Mars."
},
{
"article": "Cybersecurity architectures are increasingly incorporating machine learning classifiers to detect zero-day intrusions and mitigate cyber threats across enterprise networks.",
"summary": "Modern cybersecurity incorporates machine learning to detect intrusions and defend enterprise networks."
},
{
"article": "Smart cities are leveraging internet of things sensor grids and real-time data streaming to optimize urban traffic flows and improve civic energy conservation.",
"summary": "Smart cities use internet of things sensors and real-time data to optimize traffic and energy usage."
},
{
"article": "Global economic analysis shows that corporate digital transformation boosts workplace productivity and stimulates job creation across cloud computing and data analytics.",
"summary": "Digital transformation drives economic productivity and accelerates job growth in cloud computing and data science."
}
]
def load_kaggle_csv(
file_path: str,
article_col: str = "Articles",
summary_col: str = "Summaries",
max_samples: Optional[int] = None
) -> List[Dict[str, str]]:
"""
Loads raw CSV datasets downloaded from Kaggle (e.g., BBC News Summary CSV format)
and converts them into standard bilingual training pairs.
"""
if not os.path.exists(file_path):
raise FileNotFoundError(f"Kaggle CSV file not found: {file_path}")
df = pd.read_csv(file_path, encoding="utf-8", on_bad_lines="skip")
# Auto-detect column names if not matched exactly
cols_lower = {col.lower(): col for col in df.columns}
actual_art_col = article_col if article_col in df.columns else cols_lower.get("articles", cols_lower.get("text", cols_lower.get("article", df.columns[0])))
actual_sum_col = summary_col if summary_col in df.columns else cols_lower.get("summaries", cols_lower.get("summary", df.columns[1] if len(df.columns) > 1 else df.columns[0]))
samples = []
for _, row in df.iterrows():
art = str(row[actual_art_col]).strip()
summ = str(row[actual_sum_col]).strip()
if art and summ and art != "nan" and summ != "nan":
samples.append({"article": art, "summary": summ})
if max_samples and len(samples) >= max_samples:
break
return samples
def generate_synthetic_dataset(output_dir: str = "data/datasets", samples_per_item: int = 12):
"""
Generates training and validation datasets aligned with Kaggle benchmarks for Arabic and English.
"""
os.makedirs(output_dir, exist_ok=True)
# Augment Arabic dataset
augmented_ar = []
for item in KAGGLE_ARABIC_DATA:
for _ in range(samples_per_item):
augmented_ar.append(item)
# Augment English dataset
augmented_en = []
for item in KAGGLE_ENGLISH_DATA:
for _ in range(samples_per_item):
augmented_en.append(item)
ar_path = os.path.join(output_dir, "arabic_corpus.json")
with open(ar_path, "w", encoding="utf-8") as f:
json.dump(augmented_ar, f, ensure_ascii=False, indent=2)
en_path = os.path.join(output_dir, "english_corpus.json")
with open(en_path, "w", encoding="utf-8") as f:
json.dump(augmented_en, f, ensure_ascii=False, indent=2)
print(f"Generated {len(augmented_ar)} Arabic samples (Kaggle EASC Benchmark) -> {ar_path}")
print(f"Generated {len(augmented_en)} English samples (Kaggle BBC News Benchmark) -> {en_path}")
if __name__ == "__main__":
generate_synthetic_dataset()