GhadaAlothman commited on
Commit
fb821d7
·
verified ·
1 Parent(s): 2cba177

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +9 -47
app.py CHANGED
@@ -4,10 +4,9 @@ import torch
4
  import re
5
  from typing import Dict
6
  import textstat
7
- from fastapi import FastAPI
8
 
9
  # =======================
10
- # إعداد النموذج والمكتبات
11
  # =======================
12
  MODEL_PATH = "GhadaAlothman/arabert_readability_3class"
13
  tokenizer = AutoTokenizer.from_pretrained(MODEL_PATH)
@@ -19,15 +18,14 @@ AR_LETTERS = r"[\u0600-\u06FF]"
19
  SENT_SEP = re.compile(r"[\.!\?؟؛…]+")
20
  WORD_RE = re.compile(fr"{AR_LETTERS}+")
21
 
22
-
23
  # =======================
24
- # دوال المعالجة المساعدة
25
  # =======================
26
  def strip_diacritics(s: str):
27
  return DIACRITICS.sub("", s)
28
 
29
  def normalize_arabic(s: str):
30
- return re.sub("[\u0622\u0623\u0625]", "ا", strip_diacritics(s)).replace("ى", "ي").replace("ة", "ه")
31
 
32
  def split_sentences(text: str):
33
  return [p.strip() for p in SENT_SEP.split(text) if p.strip()]
@@ -37,68 +35,32 @@ def tokenize_words(text: str):
37
 
38
  def difficult_word(w: str, min_len: int = 6):
39
  return len(w) >= min_len
40
-
41
  def compute_metrics(ar_text: str) -> Dict[str, float]:
42
  text_norm = normalize_arabic(ar_text)
43
  sents = split_sentences(text_norm)
44
  words = tokenize_words(text_norm)
45
- n_sents, n_words = max(len(sents), 1), max(len(words), 1)
46
  diff_count = sum(1 for w in words if difficult_word(w))
47
  try:
48
  osman = float(textstat.osman(ar_text))
49
  except Exception:
50
  osman = 0.0
51
-
52
- # 🔹 حساب عدد الأحرف (بدون مسافات)
53
  n_chars = sum(len(w) for w in words)
54
-
55
- # 🔹 حساب مؤشر ARI للعربية
56
  ari_ar_score = round((4.71 * (n_chars / n_words)) + (0.5 * (n_words / n_sents)) - 21.43, 3)
57
-
58
  return {
59
  "Word count": n_words,
60
  "Sentence count": n_sents,
61
  "Character count": n_chars,
62
- "OSMAN_Score": round(osman, 3),
63
  "ARI_ArScore": ari_ar_score,
64
  "Difficult_Words_Count": diff_count,
65
- "Average_Sentence_Length_in_Words": round(n_words / n_sents, 3)
66
  }
67
 
68
-
69
  # =======================
70
- # دالة التنبؤ بالنص
71
  # =======================
72
  def analyze_text(text):
73
  inputs = tokenizer(text, return_tensors="pt", truncation=True, padding=True, max_length=128)
74
- with torch.no_grad():
75
- outputs = model(**inputs)
76
- probs = torch.nn.functional.softmax(outputs.logits, dim=-1)
77
- label_id = torch.argmax(probs, dim=1).item()
78
- label_map = {0: "سهل", 1: "متوسط", 2: "صعب"}
79
- label = label_map[label_id]
80
- stats = compute_metrics(text)
81
- return {"Predicted_Label": label, **stats}
82
-
83
-
84
- # =======================
85
- # واجهة Gradio + FastAPI
86
- # =======================
87
- app = FastAPI()
88
-
89
- demo = gr.Interface(
90
- fn=analyze_text,
91
- inputs=gr.Textbox(label="أدخل النص العربي هنا", lines=6),
92
- outputs=gr.JSON(label="نتائج التحليل"),
93
- title="Arabic Readability Analyzer",
94
- description="أداة ذكية لتقييم مقروئية النصوص العربية باستخدام نموذج AraBERT.",
95
- )
96
-
97
- # واجهة المستخدم في المسار الرئيسي "/"
98
- app = gr.mount_gradio_app(app, demo, path="/")
99
-
100
- # ✅ واجهة REST API بسيطة في /predict
101
- @app.post("/predict")
102
- async def predict(request: dict):
103
- text = request.get("text", "")
104
- return analyze_text(text)
 
4
  import re
5
  from typing import Dict
6
  import textstat
 
7
 
8
  # =======================
9
+ # إعداد النموذج
10
  # =======================
11
  MODEL_PATH = "GhadaAlothman/arabert_readability_3class"
12
  tokenizer = AutoTokenizer.from_pretrained(MODEL_PATH)
 
18
  SENT_SEP = re.compile(r"[\.!\?؟؛…]+")
19
  WORD_RE = re.compile(fr"{AR_LETTERS}+")
20
 
 
21
  # =======================
22
+ # دوال المساعدة
23
  # =======================
24
  def strip_diacritics(s: str):
25
  return DIACRITICS.sub("", s)
26
 
27
  def normalize_arabic(s: str):
28
+ return re.sub("[\u0622\u0623\u0625]", "ا", strip_diacritics(s)).replace("ى","ي").replace("ة","ه")
29
 
30
  def split_sentences(text: str):
31
  return [p.strip() for p in SENT_SEP.split(text) if p.strip()]
 
35
 
36
  def difficult_word(w: str, min_len: int = 6):
37
  return len(w) >= min_len
38
+
39
  def compute_metrics(ar_text: str) -> Dict[str, float]:
40
  text_norm = normalize_arabic(ar_text)
41
  sents = split_sentences(text_norm)
42
  words = tokenize_words(text_norm)
43
+ n_sents, n_words = max(len(sents),1), max(len(words),1)
44
  diff_count = sum(1 for w in words if difficult_word(w))
45
  try:
46
  osman = float(textstat.osman(ar_text))
47
  except Exception:
48
  osman = 0.0
 
 
49
  n_chars = sum(len(w) for w in words)
 
 
50
  ari_ar_score = round((4.71 * (n_chars / n_words)) + (0.5 * (n_words / n_sents)) - 21.43, 3)
 
51
  return {
52
  "Word count": n_words,
53
  "Sentence count": n_sents,
54
  "Character count": n_chars,
55
+ "OSMAN_Score": round(osman,3),
56
  "ARI_ArScore": ari_ar_score,
57
  "Difficult_Words_Count": diff_count,
58
+ "Average_Sentence_Length_in_Words": round(n_words/n_sents,3)
59
  }
60
 
 
61
  # =======================
62
+ # دالة التنبؤ
63
  # =======================
64
  def analyze_text(text):
65
  inputs = tokenizer(text, return_tensors="pt", truncation=True, padding=True, max_length=128)
66
+ w