Spaces:
Sleeping
Sleeping
Update app.py
Browse files
app.py
CHANGED
|
@@ -4,10 +4,9 @@ import torch
|
|
| 4 |
import re
|
| 5 |
from typing import Dict
|
| 6 |
import textstat
|
| 7 |
-
from fastapi import FastAPI
|
| 8 |
|
| 9 |
# =======================
|
| 10 |
-
# إعداد النموذج
|
| 11 |
# =======================
|
| 12 |
MODEL_PATH = "GhadaAlothman/arabert_readability_3class"
|
| 13 |
tokenizer = AutoTokenizer.from_pretrained(MODEL_PATH)
|
|
@@ -19,15 +18,14 @@ AR_LETTERS = r"[\u0600-\u06FF]"
|
|
| 19 |
SENT_SEP = re.compile(r"[\.!\?؟؛…]+")
|
| 20 |
WORD_RE = re.compile(fr"{AR_LETTERS}+")
|
| 21 |
|
| 22 |
-
|
| 23 |
# =======================
|
| 24 |
-
# دوال الم
|
| 25 |
# =======================
|
| 26 |
def strip_diacritics(s: str):
|
| 27 |
return DIACRITICS.sub("", s)
|
| 28 |
|
| 29 |
def normalize_arabic(s: str):
|
| 30 |
-
return re.sub("[\u0622\u0623\u0625]", "ا", strip_diacritics(s)).replace("ى",
|
| 31 |
|
| 32 |
def split_sentences(text: str):
|
| 33 |
return [p.strip() for p in SENT_SEP.split(text) if p.strip()]
|
|
@@ -37,68 +35,32 @@ def tokenize_words(text: str):
|
|
| 37 |
|
| 38 |
def difficult_word(w: str, min_len: int = 6):
|
| 39 |
return len(w) >= min_len
|
| 40 |
-
|
| 41 |
def compute_metrics(ar_text: str) -> Dict[str, float]:
|
| 42 |
text_norm = normalize_arabic(ar_text)
|
| 43 |
sents = split_sentences(text_norm)
|
| 44 |
words = tokenize_words(text_norm)
|
| 45 |
-
n_sents, n_words = max(len(sents),
|
| 46 |
diff_count = sum(1 for w in words if difficult_word(w))
|
| 47 |
try:
|
| 48 |
osman = float(textstat.osman(ar_text))
|
| 49 |
except Exception:
|
| 50 |
osman = 0.0
|
| 51 |
-
|
| 52 |
-
# 🔹 حساب عدد الأحرف (بدون مسافات)
|
| 53 |
n_chars = sum(len(w) for w in words)
|
| 54 |
-
|
| 55 |
-
# 🔹 حساب مؤشر ARI للعربية
|
| 56 |
ari_ar_score = round((4.71 * (n_chars / n_words)) + (0.5 * (n_words / n_sents)) - 21.43, 3)
|
| 57 |
-
|
| 58 |
return {
|
| 59 |
"Word count": n_words,
|
| 60 |
"Sentence count": n_sents,
|
| 61 |
"Character count": n_chars,
|
| 62 |
-
"OSMAN_Score": round(osman,
|
| 63 |
"ARI_ArScore": ari_ar_score,
|
| 64 |
"Difficult_Words_Count": diff_count,
|
| 65 |
-
"Average_Sentence_Length_in_Words": round(n_words
|
| 66 |
}
|
| 67 |
|
| 68 |
-
|
| 69 |
# =======================
|
| 70 |
-
# دالة التنبؤ
|
| 71 |
# =======================
|
| 72 |
def analyze_text(text):
|
| 73 |
inputs = tokenizer(text, return_tensors="pt", truncation=True, padding=True, max_length=128)
|
| 74 |
-
|
| 75 |
-
outputs = model(**inputs)
|
| 76 |
-
probs = torch.nn.functional.softmax(outputs.logits, dim=-1)
|
| 77 |
-
label_id = torch.argmax(probs, dim=1).item()
|
| 78 |
-
label_map = {0: "سهل", 1: "متوسط", 2: "صعب"}
|
| 79 |
-
label = label_map[label_id]
|
| 80 |
-
stats = compute_metrics(text)
|
| 81 |
-
return {"Predicted_Label": label, **stats}
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
# =======================
|
| 85 |
-
# واجهة Gradio + FastAPI
|
| 86 |
-
# =======================
|
| 87 |
-
app = FastAPI()
|
| 88 |
-
|
| 89 |
-
demo = gr.Interface(
|
| 90 |
-
fn=analyze_text,
|
| 91 |
-
inputs=gr.Textbox(label="أدخل النص العربي هنا", lines=6),
|
| 92 |
-
outputs=gr.JSON(label="نتائج التحليل"),
|
| 93 |
-
title="Arabic Readability Analyzer",
|
| 94 |
-
description="أداة ذكية لتقييم مقروئية النصوص العربية باستخدام نموذج AraBERT.",
|
| 95 |
-
)
|
| 96 |
-
|
| 97 |
-
# واجهة المستخدم في المسار الرئيسي "/"
|
| 98 |
-
app = gr.mount_gradio_app(app, demo, path="/")
|
| 99 |
-
|
| 100 |
-
# ✅ واجهة REST API بسيطة في /predict
|
| 101 |
-
@app.post("/predict")
|
| 102 |
-
async def predict(request: dict):
|
| 103 |
-
text = request.get("text", "")
|
| 104 |
-
return analyze_text(text)
|
|
|
|
| 4 |
import re
|
| 5 |
from typing import Dict
|
| 6 |
import textstat
|
|
|
|
| 7 |
|
| 8 |
# =======================
|
| 9 |
+
# إعداد النموذج
|
| 10 |
# =======================
|
| 11 |
MODEL_PATH = "GhadaAlothman/arabert_readability_3class"
|
| 12 |
tokenizer = AutoTokenizer.from_pretrained(MODEL_PATH)
|
|
|
|
| 18 |
SENT_SEP = re.compile(r"[\.!\?؟؛…]+")
|
| 19 |
WORD_RE = re.compile(fr"{AR_LETTERS}+")
|
| 20 |
|
|
|
|
| 21 |
# =======================
|
| 22 |
+
# دوال المساعدة
|
| 23 |
# =======================
|
| 24 |
def strip_diacritics(s: str):
|
| 25 |
return DIACRITICS.sub("", s)
|
| 26 |
|
| 27 |
def normalize_arabic(s: str):
|
| 28 |
+
return re.sub("[\u0622\u0623\u0625]", "ا", strip_diacritics(s)).replace("ى","ي").replace("ة","ه")
|
| 29 |
|
| 30 |
def split_sentences(text: str):
|
| 31 |
return [p.strip() for p in SENT_SEP.split(text) if p.strip()]
|
|
|
|
| 35 |
|
| 36 |
def difficult_word(w: str, min_len: int = 6):
|
| 37 |
return len(w) >= min_len
|
| 38 |
+
|
| 39 |
def compute_metrics(ar_text: str) -> Dict[str, float]:
|
| 40 |
text_norm = normalize_arabic(ar_text)
|
| 41 |
sents = split_sentences(text_norm)
|
| 42 |
words = tokenize_words(text_norm)
|
| 43 |
+
n_sents, n_words = max(len(sents),1), max(len(words),1)
|
| 44 |
diff_count = sum(1 for w in words if difficult_word(w))
|
| 45 |
try:
|
| 46 |
osman = float(textstat.osman(ar_text))
|
| 47 |
except Exception:
|
| 48 |
osman = 0.0
|
|
|
|
|
|
|
| 49 |
n_chars = sum(len(w) for w in words)
|
|
|
|
|
|
|
| 50 |
ari_ar_score = round((4.71 * (n_chars / n_words)) + (0.5 * (n_words / n_sents)) - 21.43, 3)
|
|
|
|
| 51 |
return {
|
| 52 |
"Word count": n_words,
|
| 53 |
"Sentence count": n_sents,
|
| 54 |
"Character count": n_chars,
|
| 55 |
+
"OSMAN_Score": round(osman,3),
|
| 56 |
"ARI_ArScore": ari_ar_score,
|
| 57 |
"Difficult_Words_Count": diff_count,
|
| 58 |
+
"Average_Sentence_Length_in_Words": round(n_words/n_sents,3)
|
| 59 |
}
|
| 60 |
|
|
|
|
| 61 |
# =======================
|
| 62 |
+
# دالة التنبؤ
|
| 63 |
# =======================
|
| 64 |
def analyze_text(text):
|
| 65 |
inputs = tokenizer(text, return_tensors="pt", truncation=True, padding=True, max_length=128)
|
| 66 |
+
w
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|