Apps / scripts /parse_iching.py
Zakir101's picture
Deploy classical guidance service
51d50ef verified
Raw History Blame Contribute Delete
6.09 kB
"""Parse Legge's I Ching (1882, SBE vol 16, archive.org OCR) into iching_lines.json.
Extracts, for each of the 64 hexagrams, the King Wen judgment and the six (or
seven) line readings. Output shape (per Giles's spec):
{ "29": { "name": "...", "judgment": "...", "lines": ["...", ...] }, ... }
Strategy: true hexagram headers carry roman numerals ("XXIX. The Khan Hexagram.");
page running-heads are ALL CAPS without numerals. We anchor on the roman-numeral
sequence 1..64 so OCR-mangled names don't break numbering, and treat everything
between two true headers as one hexagram (dropping page junk).
Run: python scripts/parse_iching.py
"""
from __future__ import annotations
import json
import re
from pathlib import Path
BASE = Path(__file__).resolve().parent.parent
SRC = BASE / "corpus" / "daoist" / "iching_legge_ocr.txt"
OUT = BASE / "corpus" / "daoist" / "iching_lines.json"
raw = SRC.read_text(encoding="utf-8", errors="ignore")
# โ”€โ”€ 1. Slice to the Text section only โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
start = raw.index("TEXT. SECTION I.")
end = raw.index("THE APPENDIXES.", start)
text = raw[start:end]
# โ”€โ”€ 2. Global OCR cleanup โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
text = re.sub(r"ยฌ\s*\n\s*", "", text) # re-join hyphenated words
text = text.replace("ยฌ", "")
# โ”€โ”€ 3. Find true headers: roman numeral + "The <name> Hexagram" โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
ROMAN = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100}
def roman_to_int(s: str) -> int | None:
s = s.lower().replace("!", "i").replace("1", "i").replace("|", "i").replace(" ", "")
if not s or any(c not in ROMAN for c in s):
return None
total = 0
for a, b in zip(s, s[1:] + "\0"):
v = ROMAN[a]
total += -v if b != "\0" and ROMAN.get(b, 0) > v else v
return total
# Header line: roman numeral, dot, "The <name> Hexagram." on its own line.
# Case-sensitive "Hexagram" excludes footnote sentences ("...this hexagram...").
# "T\w{1,3}e" tolerates OCR like "Tiie"; name may contain spaces ("Thung ZXn").
header_re = re.compile(
r"^\s*([IVXLCivxlc!1| ]{1,12})[.,]?\s+T\w{1,3}e\s+(.{1,30}?)\s+Hexagram\.?\s*$",
re.MULTILINE,
)
candidates = []
for m in header_re.finditer(text):
n = roman_to_int(m.group(1))
if n is not None and 1 <= n <= 64:
candidates.append((n, m))
# Keep an increasing sequence (small gaps allowed; gaps are reported below)
headers: list[tuple[int, re.Match]] = []
last = 0
for n, m in candidates:
if last < n <= last + 3:
headers.append((n, m))
last = n
found_nums = [n for n, _ in headers]
missing = [n for n in range(1, 65) if n not in found_nums]
print(f"True headers found: {len(headers)}; missing hexagrams: {missing}")
# โ”€โ”€ 4. Parse each hexagram block โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
PAGE_JUNK = re.compile(
r"^\s*(\d{1,3}|[A-Z .,'โ€™?!\d]{2,45}|Digitized by.*)\s*$" # bare numbers / ALL-CAPS heads
)
# Line paragraphs are matched on the ordinal WORDS (which survive OCR far better
# than the digits): "1. In the first (or lowest) line, ..." / "The second line, ..."
ORDINALS = ["first", "second", "third", "fourth", "fifth", r"(?:sixth|topmost)"]
# The article is OCR-fragile: "The", "Tiie", "1 he", "the", "In the", "From the",
# and up to ~30 chars of leading junk ("4. (To the subject of) the fourth line...").
LINE_RES = [
re.compile(
rf"^\s*.{{0,30}}?(?<![a-z])((?:In\s+|From\s+)?[Tt1l]\s?\w{{0,3}}e\s+{o}\b"
rf"(?=[^.]{{0,40}}\b(?:line|place)).*)$",
re.DOTALL,
)
for o in ORDINALS
]
SEVENTH_RE = re.compile(r"^\s*[/(]?\s*7\s*[.,]\s+(.*)$", re.DOTALL)
def clean(s: str) -> str:
s = re.sub(r"\s+", " ", s).strip()
return s.replace("โ€™", "'").replace("โ€˜", "'").replace("โ€œ", '"').replace("โ€", '"')
result: dict[str, dict] = {}
for idx, (num, h) in enumerate(headers):
block_end = headers[idx + 1][1].start() if idx + 1 < len(headers) else len(text)
block = text[h.end() : block_end]
paras = [p for p in re.split(r"\n\s*\n", block) if p.strip()]
paras = [p for p in paras if not PAGE_JUNK.match(p.strip())]
judgment_parts: list[str] = []
lines: list[str] = []
expecting = 1
for p in paras:
if expecting <= 6:
m = LINE_RES[expecting - 1].match(p)
if m:
lines.append(clean(m.group(1)))
expecting += 1
continue
elif expecting == 7:
m = SEVENTH_RE.match(p)
if m:
lines.append(clean(m.group(1)))
expecting += 1
continue
if not lines:
s = clean(p)
if s and not s.lower().startswith("explanation"):
judgment_parts.append(s)
# after lines begin, non-matching paragraphs (footnotes, page junk)
# are skipped; we keep scanning for the next expected ordinal
result[str(num)] = {
"name": clean(h.group(2)).rstrip(".,"),
"judgment": " ".join(judgment_parts)[:600],
"lines": lines,
}
# โ”€โ”€ 5. Validate โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
bad = {k: len(v["lines"]) for k, v in result.items() if not (6 <= len(v["lines"]) <= 7)}
print(f"Hexagrams parsed: {len(result)}")
print(f"Unexpected line counts: {bad or 'none'}")
OUT.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
print(f"Wrote {OUT}")
for k in ("1", "29", "64"):
if k in result:
print(f"\n--- Hexagram {k} ({result[k]['name']}) : {len(result[k]['lines'])} lines")
for i, l in enumerate(result[k]["lines"][:2], 1):
print(f" {i}. {l[:110]}")