File size: 1,996 Bytes
51d50ef
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
"""Normalize the BPHS extraction for retrieval.

The translation uses Sanskrit terms (Śani, Sūrya, Karm Bhava...) while the apps
query with English ones (Saturn, 10th house). We annotate each Sanskrit term
with its English equivalent so the embeddings carry both vocabularies:

    "Śani in Karm Bhava"  ->  "Śani (Saturn) in Karm Bhava (10th house)"

Keeps the original as bphs_raw.txt; writes the normalized bphs.txt.
Run:  python scripts/normalize_bphs.py
"""
from __future__ import annotations

import re
import shutil
from pathlib import Path

BASE = Path(__file__).resolve().parent.parent
TXT = BASE / "corpus" / "vedic" / "bphs.txt"
RAW = BASE / "corpus" / "vedic" / "bphs_raw.txt"

if not RAW.exists():
    shutil.copy(TXT, RAW)
text = RAW.read_text(encoding="utf-8")

# Planets (word-boundary, allow plural 's')
PLANETS = {
    "Sūrya": "Sun",
    "Candr": "Moon",
    "Mangal": "Mars",
    "Budh": "Mercury",
    "Guru": "Jupiter",
    "Śukr": "Venus",
    "Śani": "Saturn",
}
# Houses: Sanskrit bhava names -> ordinal house
HOUSES = {
    "Tanu": "1st house",
    "Dhan": "2nd house",
    "Sahaj": "3rd house",
    "Bandhu": "4th house",
    "Putr": "5th house",
    "Ari": "6th house",
    "Yuvati": "7th house",
    "Randhr": "8th house",
    "Dharm": "9th house",
    "Karm": "10th house",
    "Labh": "11th house",
    "Vyaya": "12th house",
}

n = 0
for skt, eng in PLANETS.items():
    text, k = re.subn(rf"\b{skt}(s?)\b(?! \()", rf"{skt}\1 ({eng})", text)
    n += k
for skt, house in HOUSES.items():
    text, k = re.subn(rf"\b{skt}\s+Bhava\b(?! \()", rf"{skt} Bhava ({house})", text)
    n += k
# Generic terms
text, k1 = re.subn(r"\bRāśi(s?)\b(?! \()", r"Rāśi\1 (sign\1)", text)
text, k2 = re.subn(r"\bBhava(s?)\b(?! \()(?! \(\d)", r"Bhava\1 (house\1)", text)
n += k1 + k2

TXT.write_text(text, encoding="utf-8")
print(f"Applied {n} annotations. Wrote {TXT}")
for probe in ("Saturn", "10th house", "Jupiter", "sign"):
    print(f"  '{probe}':", text.count(probe))