Download scripts/normalize_bphs.py from Zakir101/Apps: direct link, hf CLI and curl.
- Browser
- Download file 2 kB
-
https://huggingface.co/spaces/Zakir101/Apps/resolve/main/scripts/normalize_bphs.py
- Command line
-
hf download hf://spaces/Zakir101/Apps/scripts/normalize_bphs.py
-
curl -L -o normalize_bphs.py https://huggingface.co/spaces/Zakir101/Apps/resolve/main/scripts/normalize_bphs.py
2 kB
| """Normalize the BPHS extraction for retrieval. | |
| The translation uses Sanskrit terms (Śani, Sūrya, Karm Bhava...) while the apps | |
| query with English ones (Saturn, 10th house). We annotate each Sanskrit term | |
| with its English equivalent so the embeddings carry both vocabularies: | |
| "Śani in Karm Bhava" -> "Śani (Saturn) in Karm Bhava (10th house)" | |
| Keeps the original as bphs_raw.txt; writes the normalized bphs.txt. | |
| Run: python scripts/normalize_bphs.py | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import shutil | |
| from pathlib import Path | |
| BASE = Path(__file__).resolve().parent.parent | |
| TXT = BASE / "corpus" / "vedic" / "bphs.txt" | |
| RAW = BASE / "corpus" / "vedic" / "bphs_raw.txt" | |
| if not RAW.exists(): | |
| shutil.copy(TXT, RAW) | |
| text = RAW.read_text(encoding="utf-8") | |
| # Planets (word-boundary, allow plural 's') | |
| PLANETS = { | |
| "Sūrya": "Sun", | |
| "Candr": "Moon", | |
| "Mangal": "Mars", | |
| "Budh": "Mercury", | |
| "Guru": "Jupiter", | |
| "Śukr": "Venus", | |
| "Śani": "Saturn", | |
| } | |
| # Houses: Sanskrit bhava names -> ordinal house | |
| HOUSES = { | |
| "Tanu": "1st house", | |
| "Dhan": "2nd house", | |
| "Sahaj": "3rd house", | |
| "Bandhu": "4th house", | |
| "Putr": "5th house", | |
| "Ari": "6th house", | |
| "Yuvati": "7th house", | |
| "Randhr": "8th house", | |
| "Dharm": "9th house", | |
| "Karm": "10th house", | |
| "Labh": "11th house", | |
| "Vyaya": "12th house", | |
| } | |
| n = 0 | |
| for skt, eng in PLANETS.items(): | |
| text, k = re.subn(rf"\b{skt}(s?)\b(?! \()", rf"{skt}\1 ({eng})", text) | |
| n += k | |
| for skt, house in HOUSES.items(): | |
| text, k = re.subn(rf"\b{skt}\s+Bhava\b(?! \()", rf"{skt} Bhava ({house})", text) | |
| n += k | |
| # Generic terms | |
| text, k1 = re.subn(r"\bRāśi(s?)\b(?! \()", r"Rāśi\1 (sign\1)", text) | |
| text, k2 = re.subn(r"\bBhava(s?)\b(?! \()(?! \(\d)", r"Bhava\1 (house\1)", text) | |
| n += k1 + k2 | |
| TXT.write_text(text, encoding="utf-8") | |
| print(f"Applied {n} annotations. Wrote {TXT}") | |
| for probe in ("Saturn", "10th house", "Jupiter", "sign"): | |
| print(f" '{probe}':", text.count(probe)) | |