File size: 6,088 Bytes
51d50ef | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 | """Parse Legge's I Ching (1882, SBE vol 16, archive.org OCR) into iching_lines.json.
Extracts, for each of the 64 hexagrams, the King Wen judgment and the six (or
seven) line readings. Output shape (per Giles's spec):
{ "29": { "name": "...", "judgment": "...", "lines": ["...", ...] }, ... }
Strategy: true hexagram headers carry roman numerals ("XXIX. The Khan Hexagram.");
page running-heads are ALL CAPS without numerals. We anchor on the roman-numeral
sequence 1..64 so OCR-mangled names don't break numbering, and treat everything
between two true headers as one hexagram (dropping page junk).
Run: python scripts/parse_iching.py
"""
from __future__ import annotations
import json
import re
from pathlib import Path
BASE = Path(__file__).resolve().parent.parent
SRC = BASE / "corpus" / "daoist" / "iching_legge_ocr.txt"
OUT = BASE / "corpus" / "daoist" / "iching_lines.json"
raw = SRC.read_text(encoding="utf-8", errors="ignore")
# โโ 1. Slice to the Text section only โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
start = raw.index("TEXT. SECTION I.")
end = raw.index("THE APPENDIXES.", start)
text = raw[start:end]
# โโ 2. Global OCR cleanup โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
text = re.sub(r"ยฌ\s*\n\s*", "", text) # re-join hyphenated words
text = text.replace("ยฌ", "")
# โโ 3. Find true headers: roman numeral + "The <name> Hexagram" โโโโโโโโโโโโโ
ROMAN = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100}
def roman_to_int(s: str) -> int | None:
s = s.lower().replace("!", "i").replace("1", "i").replace("|", "i").replace(" ", "")
if not s or any(c not in ROMAN for c in s):
return None
total = 0
for a, b in zip(s, s[1:] + "\0"):
v = ROMAN[a]
total += -v if b != "\0" and ROMAN.get(b, 0) > v else v
return total
# Header line: roman numeral, dot, "The <name> Hexagram." on its own line.
# Case-sensitive "Hexagram" excludes footnote sentences ("...this hexagram...").
# "T\w{1,3}e" tolerates OCR like "Tiie"; name may contain spaces ("Thung ZXn").
header_re = re.compile(
r"^\s*([IVXLCivxlc!1| ]{1,12})[.,]?\s+T\w{1,3}e\s+(.{1,30}?)\s+Hexagram\.?\s*$",
re.MULTILINE,
)
candidates = []
for m in header_re.finditer(text):
n = roman_to_int(m.group(1))
if n is not None and 1 <= n <= 64:
candidates.append((n, m))
# Keep an increasing sequence (small gaps allowed; gaps are reported below)
headers: list[tuple[int, re.Match]] = []
last = 0
for n, m in candidates:
if last < n <= last + 3:
headers.append((n, m))
last = n
found_nums = [n for n, _ in headers]
missing = [n for n in range(1, 65) if n not in found_nums]
print(f"True headers found: {len(headers)}; missing hexagrams: {missing}")
# โโ 4. Parse each hexagram block โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
PAGE_JUNK = re.compile(
r"^\s*(\d{1,3}|[A-Z .,'โ?!\d]{2,45}|Digitized by.*)\s*$" # bare numbers / ALL-CAPS heads
)
# Line paragraphs are matched on the ordinal WORDS (which survive OCR far better
# than the digits): "1. In the first (or lowest) line, ..." / "The second line, ..."
ORDINALS = ["first", "second", "third", "fourth", "fifth", r"(?:sixth|topmost)"]
# The article is OCR-fragile: "The", "Tiie", "1 he", "the", "In the", "From the",
# and up to ~30 chars of leading junk ("4. (To the subject of) the fourth line...").
LINE_RES = [
re.compile(
rf"^\s*.{{0,30}}?(?<![a-z])((?:In\s+|From\s+)?[Tt1l]\s?\w{{0,3}}e\s+{o}\b"
rf"(?=[^.]{{0,40}}\b(?:line|place)).*)$",
re.DOTALL,
)
for o in ORDINALS
]
SEVENTH_RE = re.compile(r"^\s*[/(]?\s*7\s*[.,]\s+(.*)$", re.DOTALL)
def clean(s: str) -> str:
s = re.sub(r"\s+", " ", s).strip()
return s.replace("โ", "'").replace("โ", "'").replace("โ", '"').replace("โ", '"')
result: dict[str, dict] = {}
for idx, (num, h) in enumerate(headers):
block_end = headers[idx + 1][1].start() if idx + 1 < len(headers) else len(text)
block = text[h.end() : block_end]
paras = [p for p in re.split(r"\n\s*\n", block) if p.strip()]
paras = [p for p in paras if not PAGE_JUNK.match(p.strip())]
judgment_parts: list[str] = []
lines: list[str] = []
expecting = 1
for p in paras:
if expecting <= 6:
m = LINE_RES[expecting - 1].match(p)
if m:
lines.append(clean(m.group(1)))
expecting += 1
continue
elif expecting == 7:
m = SEVENTH_RE.match(p)
if m:
lines.append(clean(m.group(1)))
expecting += 1
continue
if not lines:
s = clean(p)
if s and not s.lower().startswith("explanation"):
judgment_parts.append(s)
# after lines begin, non-matching paragraphs (footnotes, page junk)
# are skipped; we keep scanning for the next expected ordinal
result[str(num)] = {
"name": clean(h.group(2)).rstrip(".,"),
"judgment": " ".join(judgment_parts)[:600],
"lines": lines,
}
# โโ 5. Validate โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
bad = {k: len(v["lines"]) for k, v in result.items() if not (6 <= len(v["lines"]) <= 7)}
print(f"Hexagrams parsed: {len(result)}")
print(f"Unexpected line counts: {bad or 'none'}")
OUT.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
print(f"Wrote {OUT}")
for k in ("1", "29", "64"):
if k in result:
print(f"\n--- Hexagram {k} ({result[k]['name']}) : {len(result[k]['lines'])} lines")
for i, l in enumerate(result[k]["lines"][:2], 1):
print(f" {i}. {l[:110]}")
|