Spaces:
Running
Running
Download idgham.py from Ghada-99-Ragab/Islamic2: direct link, hf CLI and curl.
- Browser
- Download file 2.9 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Islamic2/resolve/main/idgham.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Islamic2/idgham.py
-
curl -L -o idgham.py https://huggingface.co/spaces/Ghada-99-Ragab/Islamic2/resolve/main/idgham.py
2.9 kB
| """Idgham rendering of the Quran text in the published mushaf convention (used by the Subtask 1C gold corrections). | |
| The flat Quran text writes assimilated letters without a shadda; the mushaf marks the assimilation. Covers noon | |
| sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across ayah-number markers | |
| such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention is inconsistent there.""" | |
| from __future__ import annotations | |
| import re | |
| _MARKER = re.compile(r"^\(\d+\)$") | |
| _SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B" | |
| _TANWEEN = set("\u064B\u064C\u064D") | |
| _IDGHAM_AFTER_NOON = set("نمرل") | |
| _IDGHAM_AFTER_LAM = set("لر") | |
| _DIACRITIC_CHARS = set( | |
| "\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A" | |
| "\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670" | |
| "\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4" | |
| "\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED" | |
| ) | |
| def _base_letters(word: str) -> str: | |
| return "".join(c for c in word if c not in _DIACRITIC_CHARS) | |
| def _insert_shadda(word: str) -> str: | |
| return word if not word else word[0] + _SHADDA + word[1:] | |
| def _ends_with_tanween(word: str) -> bool: | |
| if not word: | |
| return False | |
| if word[-1] in _TANWEEN: | |
| return True | |
| return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN | |
| def apply_idgham(text: str, extended: bool = True) -> str: | |
| """Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda. | |
| Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across | |
| ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention | |
| is inconsistent there.""" | |
| words = text.split(" ") | |
| noon_set = _IDGHAM_AFTER_NOON if extended else set("نم") | |
| i = 0 | |
| while i < len(words): | |
| word = words[i] | |
| if _MARKER.match(word) or not word: | |
| i += 1 | |
| continue | |
| j = i + 1 | |
| while j < len(words) and _MARKER.match(words[j]): | |
| j += 1 | |
| if j < len(words): | |
| base = _base_letters(words[j]) | |
| first = base[0] if base else "" | |
| if word.endswith("\u0646" + _SUKUN) and first in noon_set: | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| elif _ends_with_tanween(word) and first in noon_set: | |
| words[j] = _insert_shadda(words[j]) | |
| elif word.endswith("\u0645" + _SUKUN) and first == "\u0645": | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM: | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| i += 1 | |
| return " ".join(words) | |