Spaces:
Running on Zero
Running on Zero
File size: 2,801 Bytes
3d9ba5b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 | import unittest
from nlp_core.language_detector import LanguageDetector
from nlp_core.arabic_preprocessor import ArabicPreprocessor
from nlp_core.english_preprocessor import EnglishPreprocessor
from nlp_core.tokenizer import BilingualTokenizer
from nlp_core.vocabulary import Vocabulary
class TestPreprocessing(unittest.TestCase):
def setUp(self):
self.detector = LanguageDetector()
self.ar_prep = ArabicPreprocessor()
self.en_prep = EnglishPreprocessor()
self.tokenizer = BilingualTokenizer()
def test_language_detection(self):
ar_text = "تعد معالجة اللغات الطبيعية فرعا مهما من فروع الذكاء الاصطناعي."
en_text = "Natural language processing is an essential field of artificial intelligence."
self.assertEqual(self.detector.detect_language(ar_text), "ar")
self.assertEqual(self.detector.detect_language(en_text), "en")
def test_arabic_tashkeel_and_normalization(self):
text_with_tashkeel = "التَّلْخِيصُ التِّلْقَائِيُّ لِلنُّصُوصِ"
no_tashkeel = self.ar_prep.remove_tashkeel(text_with_tashkeel)
self.assertNotIn("َ", no_tashkeel)
self.assertNotIn("ْ", no_tashkeel)
normalized = self.ar_prep.normalize_letters("إستخراج أفكار رئيسية وأخرى")
self.assertTrue(normalized.startswith("استخراج"))
self.assertIn("اخري", normalized)
def test_english_contractions(self):
text = "We haven't seen the results, but they don't seem problematic."
expanded = self.en_prep.expand_contractions(text)
self.assertIn("have not", expanded)
self.assertIn("do not", expanded)
def test_sentence_tokenization(self):
ar_doc = "الجملة الأولى هنا. الجملة الثانية هنا؟ وهل توجد جملة ثالثة!"
sents = self.tokenizer.split_sentences(ar_doc, lang="ar")
self.assertGreaterEqual(len(sents), 2)
en_doc = "Dr. Smith presented the paper. It was very well received! What do you think?"
sents_en = self.tokenizer.split_sentences(en_doc, lang="en")
self.assertGreaterEqual(len(sents_en), 2)
def test_vocabulary(self):
vocab = Vocabulary("test_vocab")
vocab.add_sentence(["natural", "language", "processing", "summarization"])
vocab.build_vocab(min_freq=1)
encoded = vocab.encode(["natural", "language"])
self.assertEqual(encoded[0], vocab.SOS_IDX)
self.assertEqual(encoded[-1], vocab.EOS_IDX)
decoded = vocab.decode(encoded, remove_special_tokens=True)
self.assertEqual(decoded, ["natural", "language"])
if __name__ == "__main__":
unittest.main()
|