Spaces:
Running on Zero
Running on Zero
Download tests/test_preprocessing.py from fady21/bilingual-summarizer-api: direct link, hf CLI and curl.
- Browser
- Download file 2.8 kB
-
https://huggingface.co/spaces/fady21/bilingual-summarizer-api/resolve/main/tests/test_preprocessing.py
- Command line
-
hf download hf://spaces/fady21/bilingual-summarizer-api/tests/test_preprocessing.py
-
curl -L -o test_preprocessing.py https://huggingface.co/spaces/fady21/bilingual-summarizer-api/resolve/main/tests/test_preprocessing.py
2.8 kB
| import unittest | |
| from nlp_core.language_detector import LanguageDetector | |
| from nlp_core.arabic_preprocessor import ArabicPreprocessor | |
| from nlp_core.english_preprocessor import EnglishPreprocessor | |
| from nlp_core.tokenizer import BilingualTokenizer | |
| from nlp_core.vocabulary import Vocabulary | |
| class TestPreprocessing(unittest.TestCase): | |
| def setUp(self): | |
| self.detector = LanguageDetector() | |
| self.ar_prep = ArabicPreprocessor() | |
| self.en_prep = EnglishPreprocessor() | |
| self.tokenizer = BilingualTokenizer() | |
| def test_language_detection(self): | |
| ar_text = "تعد معالجة اللغات الطبيعية فرعا مهما من فروع الذكاء الاصطناعي." | |
| en_text = "Natural language processing is an essential field of artificial intelligence." | |
| self.assertEqual(self.detector.detect_language(ar_text), "ar") | |
| self.assertEqual(self.detector.detect_language(en_text), "en") | |
| def test_arabic_tashkeel_and_normalization(self): | |
| text_with_tashkeel = "التَّلْخِيصُ التِّلْقَائِيُّ لِلنُّصُوصِ" | |
| no_tashkeel = self.ar_prep.remove_tashkeel(text_with_tashkeel) | |
| self.assertNotIn("َ", no_tashkeel) | |
| self.assertNotIn("ْ", no_tashkeel) | |
| normalized = self.ar_prep.normalize_letters("إستخراج أفكار رئيسية وأخرى") | |
| self.assertTrue(normalized.startswith("استخراج")) | |
| self.assertIn("اخري", normalized) | |
| def test_english_contractions(self): | |
| text = "We haven't seen the results, but they don't seem problematic." | |
| expanded = self.en_prep.expand_contractions(text) | |
| self.assertIn("have not", expanded) | |
| self.assertIn("do not", expanded) | |
| def test_sentence_tokenization(self): | |
| ar_doc = "الجملة الأولى هنا. الجملة الثانية هنا؟ وهل توجد جملة ثالثة!" | |
| sents = self.tokenizer.split_sentences(ar_doc, lang="ar") | |
| self.assertGreaterEqual(len(sents), 2) | |
| en_doc = "Dr. Smith presented the paper. It was very well received! What do you think?" | |
| sents_en = self.tokenizer.split_sentences(en_doc, lang="en") | |
| self.assertGreaterEqual(len(sents_en), 2) | |
| def test_vocabulary(self): | |
| vocab = Vocabulary("test_vocab") | |
| vocab.add_sentence(["natural", "language", "processing", "summarization"]) | |
| vocab.build_vocab(min_freq=1) | |
| encoded = vocab.encode(["natural", "language"]) | |
| self.assertEqual(encoded[0], vocab.SOS_IDX) | |
| self.assertEqual(encoded[-1], vocab.EOS_IDX) | |
| decoded = vocab.decode(encoded, remove_special_tokens=True) | |
| self.assertEqual(decoded, ["natural", "language"]) | |
| if __name__ == "__main__": | |
| unittest.main() | |