import unittest from nlp_core.language_detector import LanguageDetector from nlp_core.arabic_preprocessor import ArabicPreprocessor from nlp_core.english_preprocessor import EnglishPreprocessor from nlp_core.tokenizer import BilingualTokenizer from nlp_core.vocabulary import Vocabulary class TestPreprocessing(unittest.TestCase): def setUp(self): self.detector = LanguageDetector() self.ar_prep = ArabicPreprocessor() self.en_prep = EnglishPreprocessor() self.tokenizer = BilingualTokenizer() def test_language_detection(self): ar_text = "تعد معالجة اللغات الطبيعية فرعا مهما من فروع الذكاء الاصطناعي." en_text = "Natural language processing is an essential field of artificial intelligence." self.assertEqual(self.detector.detect_language(ar_text), "ar") self.assertEqual(self.detector.detect_language(en_text), "en") def test_arabic_tashkeel_and_normalization(self): text_with_tashkeel = "التَّلْخِيصُ التِّلْقَائِيُّ لِلنُّصُوصِ" no_tashkeel = self.ar_prep.remove_tashkeel(text_with_tashkeel) self.assertNotIn("َ", no_tashkeel) self.assertNotIn("ْ", no_tashkeel) normalized = self.ar_prep.normalize_letters("إستخراج أفكار رئيسية وأخرى") self.assertTrue(normalized.startswith("استخراج")) self.assertIn("اخري", normalized) def test_english_contractions(self): text = "We haven't seen the results, but they don't seem problematic." expanded = self.en_prep.expand_contractions(text) self.assertIn("have not", expanded) self.assertIn("do not", expanded) def test_sentence_tokenization(self): ar_doc = "الجملة الأولى هنا. الجملة الثانية هنا؟ وهل توجد جملة ثالثة!" sents = self.tokenizer.split_sentences(ar_doc, lang="ar") self.assertGreaterEqual(len(sents), 2) en_doc = "Dr. Smith presented the paper. It was very well received! What do you think?" sents_en = self.tokenizer.split_sentences(en_doc, lang="en") self.assertGreaterEqual(len(sents_en), 2) def test_vocabulary(self): vocab = Vocabulary("test_vocab") vocab.add_sentence(["natural", "language", "processing", "summarization"]) vocab.build_vocab(min_freq=1) encoded = vocab.encode(["natural", "language"]) self.assertEqual(encoded[0], vocab.SOS_IDX) self.assertEqual(encoded[-1], vocab.EOS_IDX) decoded = vocab.decode(encoded, remove_special_tokens=True) self.assertEqual(decoded, ["natural", "language"]) if __name__ == "__main__": unittest.main()