File size: 2,801 Bytes
3d9ba5b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
import unittest
from nlp_core.language_detector import LanguageDetector
from nlp_core.arabic_preprocessor import ArabicPreprocessor
from nlp_core.english_preprocessor import EnglishPreprocessor
from nlp_core.tokenizer import BilingualTokenizer
from nlp_core.vocabulary import Vocabulary

class TestPreprocessing(unittest.TestCase):

    def setUp(self):
        self.detector = LanguageDetector()
        self.ar_prep = ArabicPreprocessor()
        self.en_prep = EnglishPreprocessor()
        self.tokenizer = BilingualTokenizer()

    def test_language_detection(self):
        ar_text = "تعد معالجة اللغات الطبيعية فرعا مهما من فروع الذكاء الاصطناعي."
        en_text = "Natural language processing is an essential field of artificial intelligence."
        
        self.assertEqual(self.detector.detect_language(ar_text), "ar")
        self.assertEqual(self.detector.detect_language(en_text), "en")

    def test_arabic_tashkeel_and_normalization(self):
        text_with_tashkeel = "التَّلْخِيصُ التِّلْقَائِيُّ لِلنُّصُوصِ"
        no_tashkeel = self.ar_prep.remove_tashkeel(text_with_tashkeel)
        self.assertNotIn("َ", no_tashkeel)
        self.assertNotIn("ْ", no_tashkeel)

        normalized = self.ar_prep.normalize_letters("إستخراج أفكار رئيسية وأخرى")
        self.assertTrue(normalized.startswith("استخراج"))
        self.assertIn("اخري", normalized)

    def test_english_contractions(self):
        text = "We haven't seen the results, but they don't seem problematic."
        expanded = self.en_prep.expand_contractions(text)
        self.assertIn("have not", expanded)
        self.assertIn("do not", expanded)

    def test_sentence_tokenization(self):
        ar_doc = "الجملة الأولى هنا. الجملة الثانية هنا؟ وهل توجد جملة ثالثة!"
        sents = self.tokenizer.split_sentences(ar_doc, lang="ar")
        self.assertGreaterEqual(len(sents), 2)

        en_doc = "Dr. Smith presented the paper. It was very well received! What do you think?"
        sents_en = self.tokenizer.split_sentences(en_doc, lang="en")
        self.assertGreaterEqual(len(sents_en), 2)

    def test_vocabulary(self):
        vocab = Vocabulary("test_vocab")
        vocab.add_sentence(["natural", "language", "processing", "summarization"])
        vocab.build_vocab(min_freq=1)
        
        encoded = vocab.encode(["natural", "language"])
        self.assertEqual(encoded[0], vocab.SOS_IDX)
        self.assertEqual(encoded[-1], vocab.EOS_IDX)

        decoded = vocab.decode(encoded, remove_special_tokens=True)
        self.assertEqual(decoded, ["natural", "language"])

if __name__ == "__main__":
    unittest.main()