Download scripts/parse_iching.py from Zakir101/Apps: direct link, hf CLI and curl.
- Browser
- Download file 6.09 kB
-
https://huggingface.co/spaces/Zakir101/Apps/resolve/main/scripts/parse_iching.py
- Command line
-
hf download hf://spaces/Zakir101/Apps/scripts/parse_iching.py
-
curl -L -o parse_iching.py https://huggingface.co/spaces/Zakir101/Apps/resolve/main/scripts/parse_iching.py
6.09 kB
| """Parse Legge's I Ching (1882, SBE vol 16, archive.org OCR) into iching_lines.json. | |
| Extracts, for each of the 64 hexagrams, the King Wen judgment and the six (or | |
| seven) line readings. Output shape (per Giles's spec): | |
| { "29": { "name": "...", "judgment": "...", "lines": ["...", ...] }, ... } | |
| Strategy: true hexagram headers carry roman numerals ("XXIX. The Khan Hexagram."); | |
| page running-heads are ALL CAPS without numerals. We anchor on the roman-numeral | |
| sequence 1..64 so OCR-mangled names don't break numbering, and treat everything | |
| between two true headers as one hexagram (dropping page junk). | |
| Run: python scripts/parse_iching.py | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import re | |
| from pathlib import Path | |
| BASE = Path(__file__).resolve().parent.parent | |
| SRC = BASE / "corpus" / "daoist" / "iching_legge_ocr.txt" | |
| OUT = BASE / "corpus" / "daoist" / "iching_lines.json" | |
| raw = SRC.read_text(encoding="utf-8", errors="ignore") | |
| # โโ 1. Slice to the Text section only โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| start = raw.index("TEXT. SECTION I.") | |
| end = raw.index("THE APPENDIXES.", start) | |
| text = raw[start:end] | |
| # โโ 2. Global OCR cleanup โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| text = re.sub(r"ยฌ\s*\n\s*", "", text) # re-join hyphenated words | |
| text = text.replace("ยฌ", "") | |
| # โโ 3. Find true headers: roman numeral + "The <name> Hexagram" โโโโโโโโโโโโโ | |
| ROMAN = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100} | |
| def roman_to_int(s: str) -> int | None: | |
| s = s.lower().replace("!", "i").replace("1", "i").replace("|", "i").replace(" ", "") | |
| if not s or any(c not in ROMAN for c in s): | |
| return None | |
| total = 0 | |
| for a, b in zip(s, s[1:] + "\0"): | |
| v = ROMAN[a] | |
| total += -v if b != "\0" and ROMAN.get(b, 0) > v else v | |
| return total | |
| # Header line: roman numeral, dot, "The <name> Hexagram." on its own line. | |
| # Case-sensitive "Hexagram" excludes footnote sentences ("...this hexagram..."). | |
| # "T\w{1,3}e" tolerates OCR like "Tiie"; name may contain spaces ("Thung ZXn"). | |
| header_re = re.compile( | |
| r"^\s*([IVXLCivxlc!1| ]{1,12})[.,]?\s+T\w{1,3}e\s+(.{1,30}?)\s+Hexagram\.?\s*$", | |
| re.MULTILINE, | |
| ) | |
| candidates = [] | |
| for m in header_re.finditer(text): | |
| n = roman_to_int(m.group(1)) | |
| if n is not None and 1 <= n <= 64: | |
| candidates.append((n, m)) | |
| # Keep an increasing sequence (small gaps allowed; gaps are reported below) | |
| headers: list[tuple[int, re.Match]] = [] | |
| last = 0 | |
| for n, m in candidates: | |
| if last < n <= last + 3: | |
| headers.append((n, m)) | |
| last = n | |
| found_nums = [n for n, _ in headers] | |
| missing = [n for n in range(1, 65) if n not in found_nums] | |
| print(f"True headers found: {len(headers)}; missing hexagrams: {missing}") | |
| # โโ 4. Parse each hexagram block โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| PAGE_JUNK = re.compile( | |
| r"^\s*(\d{1,3}|[A-Z .,'โ?!\d]{2,45}|Digitized by.*)\s*$" # bare numbers / ALL-CAPS heads | |
| ) | |
| # Line paragraphs are matched on the ordinal WORDS (which survive OCR far better | |
| # than the digits): "1. In the first (or lowest) line, ..." / "The second line, ..." | |
| ORDINALS = ["first", "second", "third", "fourth", "fifth", r"(?:sixth|topmost)"] | |
| # The article is OCR-fragile: "The", "Tiie", "1 he", "the", "In the", "From the", | |
| # and up to ~30 chars of leading junk ("4. (To the subject of) the fourth line..."). | |
| LINE_RES = [ | |
| re.compile( | |
| rf"^\s*.{{0,30}}?(?<![a-z])((?:In\s+|From\s+)?[Tt1l]\s?\w{{0,3}}e\s+{o}\b" | |
| rf"(?=[^.]{{0,40}}\b(?:line|place)).*)$", | |
| re.DOTALL, | |
| ) | |
| for o in ORDINALS | |
| ] | |
| SEVENTH_RE = re.compile(r"^\s*[/(]?\s*7\s*[.,]\s+(.*)$", re.DOTALL) | |
| def clean(s: str) -> str: | |
| s = re.sub(r"\s+", " ", s).strip() | |
| return s.replace("โ", "'").replace("โ", "'").replace("โ", '"').replace("โ", '"') | |
| result: dict[str, dict] = {} | |
| for idx, (num, h) in enumerate(headers): | |
| block_end = headers[idx + 1][1].start() if idx + 1 < len(headers) else len(text) | |
| block = text[h.end() : block_end] | |
| paras = [p for p in re.split(r"\n\s*\n", block) if p.strip()] | |
| paras = [p for p in paras if not PAGE_JUNK.match(p.strip())] | |
| judgment_parts: list[str] = [] | |
| lines: list[str] = [] | |
| expecting = 1 | |
| for p in paras: | |
| if expecting <= 6: | |
| m = LINE_RES[expecting - 1].match(p) | |
| if m: | |
| lines.append(clean(m.group(1))) | |
| expecting += 1 | |
| continue | |
| elif expecting == 7: | |
| m = SEVENTH_RE.match(p) | |
| if m: | |
| lines.append(clean(m.group(1))) | |
| expecting += 1 | |
| continue | |
| if not lines: | |
| s = clean(p) | |
| if s and not s.lower().startswith("explanation"): | |
| judgment_parts.append(s) | |
| # after lines begin, non-matching paragraphs (footnotes, page junk) | |
| # are skipped; we keep scanning for the next expected ordinal | |
| result[str(num)] = { | |
| "name": clean(h.group(2)).rstrip(".,"), | |
| "judgment": " ".join(judgment_parts)[:600], | |
| "lines": lines, | |
| } | |
| # โโ 5. Validate โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| bad = {k: len(v["lines"]) for k, v in result.items() if not (6 <= len(v["lines"]) <= 7)} | |
| print(f"Hexagrams parsed: {len(result)}") | |
| print(f"Unexpected line counts: {bad or 'none'}") | |
| OUT.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8") | |
| print(f"Wrote {OUT}") | |
| for k in ("1", "29", "64"): | |
| if k in result: | |
| print(f"\n--- Hexagram {k} ({result[k]['name']}) : {len(result[k]['lines'])} lines") | |
| for i, l in enumerate(result[k]["lines"][:2], 1): | |
| print(f" {i}. {l[:110]}") | |