"""Parse Legge's I Ching (1882, SBE vol 16, archive.org OCR) into iching_lines.json. Extracts, for each of the 64 hexagrams, the King Wen judgment and the six (or seven) line readings. Output shape (per Giles's spec): { "29": { "name": "...", "judgment": "...", "lines": ["...", ...] }, ... } Strategy: true hexagram headers carry roman numerals ("XXIX. The Khan Hexagram."); page running-heads are ALL CAPS without numerals. We anchor on the roman-numeral sequence 1..64 so OCR-mangled names don't break numbering, and treat everything between two true headers as one hexagram (dropping page junk). Run: python scripts/parse_iching.py """ from __future__ import annotations import json import re from pathlib import Path BASE = Path(__file__).resolve().parent.parent SRC = BASE / "corpus" / "daoist" / "iching_legge_ocr.txt" OUT = BASE / "corpus" / "daoist" / "iching_lines.json" raw = SRC.read_text(encoding="utf-8", errors="ignore") # ── 1. Slice to the Text section only ──────────────────────────────────────── start = raw.index("TEXT. SECTION I.") end = raw.index("THE APPENDIXES.", start) text = raw[start:end] # ── 2. Global OCR cleanup ──────────────────────────────────────────────────── text = re.sub(r"¬\s*\n\s*", "", text) # re-join hyphenated words text = text.replace("¬", "") # ── 3. Find true headers: roman numeral + "The Hexagram" ───────────── ROMAN = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100} def roman_to_int(s: str) -> int | None: s = s.lower().replace("!", "i").replace("1", "i").replace("|", "i").replace(" ", "") if not s or any(c not in ROMAN for c in s): return None total = 0 for a, b in zip(s, s[1:] + "\0"): v = ROMAN[a] total += -v if b != "\0" and ROMAN.get(b, 0) > v else v return total # Header line: roman numeral, dot, "The Hexagram." on its own line. # Case-sensitive "Hexagram" excludes footnote sentences ("...this hexagram..."). # "T\w{1,3}e" tolerates OCR like "Tiie"; name may contain spaces ("Thung ZXn"). header_re = re.compile( r"^\s*([IVXLCivxlc!1| ]{1,12})[.,]?\s+T\w{1,3}e\s+(.{1,30}?)\s+Hexagram\.?\s*$", re.MULTILINE, ) candidates = [] for m in header_re.finditer(text): n = roman_to_int(m.group(1)) if n is not None and 1 <= n <= 64: candidates.append((n, m)) # Keep an increasing sequence (small gaps allowed; gaps are reported below) headers: list[tuple[int, re.Match]] = [] last = 0 for n, m in candidates: if last < n <= last + 3: headers.append((n, m)) last = n found_nums = [n for n, _ in headers] missing = [n for n in range(1, 65) if n not in found_nums] print(f"True headers found: {len(headers)}; missing hexagrams: {missing}") # ── 4. Parse each hexagram block ───────────────────────────────────────────── PAGE_JUNK = re.compile( r"^\s*(\d{1,3}|[A-Z .,'’?!\d]{2,45}|Digitized by.*)\s*$" # bare numbers / ALL-CAPS heads ) # Line paragraphs are matched on the ordinal WORDS (which survive OCR far better # than the digits): "1. In the first (or lowest) line, ..." / "The second line, ..." ORDINALS = ["first", "second", "third", "fourth", "fifth", r"(?:sixth|topmost)"] # The article is OCR-fragile: "The", "Tiie", "1 he", "the", "In the", "From the", # and up to ~30 chars of leading junk ("4. (To the subject of) the fourth line..."). LINE_RES = [ re.compile( rf"^\s*.{{0,30}}?(? str: s = re.sub(r"\s+", " ", s).strip() return s.replace("’", "'").replace("‘", "'").replace("“", '"').replace("”", '"') result: dict[str, dict] = {} for idx, (num, h) in enumerate(headers): block_end = headers[idx + 1][1].start() if idx + 1 < len(headers) else len(text) block = text[h.end() : block_end] paras = [p for p in re.split(r"\n\s*\n", block) if p.strip()] paras = [p for p in paras if not PAGE_JUNK.match(p.strip())] judgment_parts: list[str] = [] lines: list[str] = [] expecting = 1 for p in paras: if expecting <= 6: m = LINE_RES[expecting - 1].match(p) if m: lines.append(clean(m.group(1))) expecting += 1 continue elif expecting == 7: m = SEVENTH_RE.match(p) if m: lines.append(clean(m.group(1))) expecting += 1 continue if not lines: s = clean(p) if s and not s.lower().startswith("explanation"): judgment_parts.append(s) # after lines begin, non-matching paragraphs (footnotes, page junk) # are skipped; we keep scanning for the next expected ordinal result[str(num)] = { "name": clean(h.group(2)).rstrip(".,"), "judgment": " ".join(judgment_parts)[:600], "lines": lines, } # ── 5. Validate ─────────────────────────────────────────────────────────────── bad = {k: len(v["lines"]) for k, v in result.items() if not (6 <= len(v["lines"]) <= 7)} print(f"Hexagrams parsed: {len(result)}") print(f"Unexpected line counts: {bad or 'none'}") OUT.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8") print(f"Wrote {OUT}") for k in ("1", "29", "64"): if k in result: print(f"\n--- Hexagram {k} ({result[k]['name']}) : {len(result[k]['lines'])} lines") for i, l in enumerate(result[k]["lines"][:2], 1): print(f" {i}. {l[:110]}")