File size: 6,088 Bytes
51d50ef
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
"""Parse Legge's I Ching (1882, SBE vol 16, archive.org OCR) into iching_lines.json.

Extracts, for each of the 64 hexagrams, the King Wen judgment and the six (or
seven) line readings. Output shape (per Giles's spec):

    { "29": { "name": "...", "judgment": "...", "lines": ["...", ...] }, ... }

Strategy: true hexagram headers carry roman numerals ("XXIX. The Khan Hexagram.");
page running-heads are ALL CAPS without numerals. We anchor on the roman-numeral
sequence 1..64 so OCR-mangled names don't break numbering, and treat everything
between two true headers as one hexagram (dropping page junk).

Run:  python scripts/parse_iching.py
"""
from __future__ import annotations

import json
import re
from pathlib import Path

BASE = Path(__file__).resolve().parent.parent
SRC = BASE / "corpus" / "daoist" / "iching_legge_ocr.txt"
OUT = BASE / "corpus" / "daoist" / "iching_lines.json"

raw = SRC.read_text(encoding="utf-8", errors="ignore")

# โ”€โ”€ 1. Slice to the Text section only โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
start = raw.index("TEXT. SECTION I.")
end = raw.index("THE APPENDIXES.", start)
text = raw[start:end]

# โ”€โ”€ 2. Global OCR cleanup โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
text = re.sub(r"ยฌ\s*\n\s*", "", text)  # re-join hyphenated words
text = text.replace("ยฌ", "")

# โ”€โ”€ 3. Find true headers: roman numeral + "The <name> Hexagram" โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
ROMAN = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100}

def roman_to_int(s: str) -> int | None:
    s = s.lower().replace("!", "i").replace("1", "i").replace("|", "i").replace(" ", "")
    if not s or any(c not in ROMAN for c in s):
        return None
    total = 0
    for a, b in zip(s, s[1:] + "\0"):
        v = ROMAN[a]
        total += -v if b != "\0" and ROMAN.get(b, 0) > v else v
    return total

# Header line: roman numeral, dot, "The <name> Hexagram." on its own line.
# Case-sensitive "Hexagram" excludes footnote sentences ("...this hexagram...").
# "T\w{1,3}e" tolerates OCR like "Tiie"; name may contain spaces ("Thung ZXn").
header_re = re.compile(
    r"^\s*([IVXLCivxlc!1| ]{1,12})[.,]?\s+T\w{1,3}e\s+(.{1,30}?)\s+Hexagram\.?\s*$",
    re.MULTILINE,
)
candidates = []
for m in header_re.finditer(text):
    n = roman_to_int(m.group(1))
    if n is not None and 1 <= n <= 64:
        candidates.append((n, m))

# Keep an increasing sequence (small gaps allowed; gaps are reported below)
headers: list[tuple[int, re.Match]] = []
last = 0
for n, m in candidates:
    if last < n <= last + 3:
        headers.append((n, m))
        last = n
found_nums = [n for n, _ in headers]
missing = [n for n in range(1, 65) if n not in found_nums]
print(f"True headers found: {len(headers)}; missing hexagrams: {missing}")

# โ”€โ”€ 4. Parse each hexagram block โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
PAGE_JUNK = re.compile(
    r"^\s*(\d{1,3}|[A-Z .,'โ€™?!\d]{2,45}|Digitized by.*)\s*$"  # bare numbers / ALL-CAPS heads
)

# Line paragraphs are matched on the ordinal WORDS (which survive OCR far better
# than the digits): "1. In the first (or lowest) line, ..." / "The second line, ..."
ORDINALS = ["first", "second", "third", "fourth", "fifth", r"(?:sixth|topmost)"]
# The article is OCR-fragile: "The", "Tiie", "1 he", "the", "In the", "From the",
# and up to ~30 chars of leading junk ("4. (To the subject of) the fourth line...").
LINE_RES = [
    re.compile(
        rf"^\s*.{{0,30}}?(?<![a-z])((?:In\s+|From\s+)?[Tt1l]\s?\w{{0,3}}e\s+{o}\b"
        rf"(?=[^.]{{0,40}}\b(?:line|place)).*)$",
        re.DOTALL,
    )
    for o in ORDINALS
]
SEVENTH_RE = re.compile(r"^\s*[/(]?\s*7\s*[.,]\s+(.*)$", re.DOTALL)

def clean(s: str) -> str:
    s = re.sub(r"\s+", " ", s).strip()
    return s.replace("โ€™", "'").replace("โ€˜", "'").replace("โ€œ", '"').replace("โ€", '"')

result: dict[str, dict] = {}
for idx, (num, h) in enumerate(headers):
    block_end = headers[idx + 1][1].start() if idx + 1 < len(headers) else len(text)
    block = text[h.end() : block_end]
    paras = [p for p in re.split(r"\n\s*\n", block) if p.strip()]
    paras = [p for p in paras if not PAGE_JUNK.match(p.strip())]

    judgment_parts: list[str] = []
    lines: list[str] = []
    expecting = 1
    for p in paras:
        if expecting <= 6:
            m = LINE_RES[expecting - 1].match(p)
            if m:
                lines.append(clean(m.group(1)))
                expecting += 1
                continue
        elif expecting == 7:
            m = SEVENTH_RE.match(p)
            if m:
                lines.append(clean(m.group(1)))
                expecting += 1
                continue
        if not lines:
            s = clean(p)
            if s and not s.lower().startswith("explanation"):
                judgment_parts.append(s)
        # after lines begin, non-matching paragraphs (footnotes, page junk)
        # are skipped; we keep scanning for the next expected ordinal

    result[str(num)] = {
        "name": clean(h.group(2)).rstrip(".,"),
        "judgment": " ".join(judgment_parts)[:600],
        "lines": lines,
    }

# โ”€โ”€ 5. Validate โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
bad = {k: len(v["lines"]) for k, v in result.items() if not (6 <= len(v["lines"]) <= 7)}
print(f"Hexagrams parsed: {len(result)}")
print(f"Unexpected line counts: {bad or 'none'}")

OUT.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
print(f"Wrote {OUT}")

for k in ("1", "29", "64"):
    if k in result:
        print(f"\n--- Hexagram {k} ({result[k]['name']}) : {len(result[k]['lines'])} lines")
        for i, l in enumerate(result[k]["lines"][:2], 1):
            print(f"  {i}. {l[:110]}")