Download src/knowledge_parsing/normalize_mistral.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 27.1 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize_mistral.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_parsing/normalize_mistral.py
-
curl -L -o normalize_mistral.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize_mistral.py
27.1 kB
| """Stage 3 (Mistral) — turn Mistral OCR's block stream into per-section chunks. | |
| The counterpart of `normalize.py`, for the Mistral OCR backend. With | |
| `include_blocks=True` Mistral returns a per-page list of typed blocks | |
| (`title` / `text` / `equation` / `table` / `caption` / `list` / `image` / | |
| `header` / `footer`), each with a bbox and its content — a structure ~1:1 with | |
| MinerU's `content_list.json`. So this is a close port of `normalize.py`, and it | |
| REUSES that module's `_heading` logic verbatim so section boundaries are decided | |
| identically on both backends. | |
| Two things it does that `normalize.py` does not have to: | |
| - **Strip markdown markup from text.** Mistral renders content as markdown, so | |
| a text line arrives as `*Production* yang **digunakan** ...`. Those `*`/`#` | |
| are markup Mistral added, NOT characters in the source PDF — and the | |
| extraction span-check locates LLM-quoted spans LITERALLY inside `text`, so | |
| leaving them in makes a quote of "Production" fail to match. Stripping markup | |
| is the same move `render.py` makes for LaTeX/HTML; it is NOT reflowing. | |
| Whitespace and line breaks are still left exactly as they are. | |
| - **Correlate image descriptions across the response.** A block of type | |
| `image` carries only `` + an `image_id`; the model-written | |
| description lives at the page level in `pages[].images[].image_annotation` | |
| (produced by `bbox_annotation`). They are matched by id here. | |
| ⚠️ Text is NEVER tidied beyond markup stripping — no rejoining lines, no | |
| whitespace normalisation. See contracts.py for why. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| import re | |
| from collections.abc import Iterator | |
| from functools import lru_cache | |
| from pathlib import Path | |
| from typing import Any | |
| from .assets import AssetSink, store_asset, synthetic_asset_id | |
| from .contracts import Asset, Chunk | |
| from .normalize import MAX_CHUNK_CHARS, _heading # identical section-boundary logic | |
| from .render import render_latex, render_table | |
| # Fingerprint of the code that decides chunk CONTENT on this backend — same | |
| # mechanism and reasoning as normalize.normaliser_version (see there). Covers | |
| # this module and render.py; contracts.py is deliberately excluded (its shape is | |
| # versioned by schema_version). | |
| _CONTENT_MODULES = ("normalize_mistral.py", "render.py") | |
| def normaliser_version() -> str: | |
| here = Path(__file__).parent | |
| h = hashlib.sha256() | |
| for name in _CONTENT_MODULES: | |
| h.update((here / name).read_bytes()) | |
| return h.hexdigest()[:12] | |
| # Page furniture, dropped on purpose — matches normalize._DISCARDED. `header` | |
| # is handled separately (it becomes chapter context). `signature` is Mistral's | |
| # label for the cover-page "Disusun/Disetujui" stamps, whose content is empty in | |
| # the source anyway. | |
| _DISCARDED = {"page_number", "footer"} | |
| # Recorded on every Asset so a later swap of the description model is traceable | |
| # rather than silently changing what the field means. | |
| _DESCRIPTION_SOURCE = "mistral-ocr/bbox_annotation" | |
| _LEADING_HASHES = re.compile(r"^\s*(#{1,6})\s*") | |
| _MD_IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)") | |
| # Same shape, but capturing the target so a block's own markdown can be read | |
| # for the image id it refers to (see _block_image_id). | |
| _MD_IMAGE_TARGET = re.compile(r"!\[[^\]]*\]\(([^)]*)\)") | |
| _MD_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)") | |
| _MD_BOLD = re.compile(r"\*\*(.+?)\*\*", re.S) | |
| _MD_ITALIC = re.compile(r"\*(.+?)\*", re.S) | |
| # Inline math inside prose: "... nilai $x_i$ ...". Rendered to readable text for | |
| # the same reason render.render_table renders math inside table cells. | |
| _INLINE_MATH = re.compile(r"\$([^$]{1,400})\$") | |
| def _strip_markdown(text: str, vocabulary: frozenset[str] | None = None) -> str: | |
| """Remove markup Mistral ADDED, keep the document's own characters. | |
| Deliberately conservative: only paired `*`/`**` emphasis, heading `#`, | |
| backticks, image and link syntax, and inline `$…$`. Underscores are left | |
| untouched — they appear in real variable names (`INPR_Hours`) far more often | |
| than as emphasis, and stripping them would damage content. Whitespace and | |
| newlines are preserved: this is markup removal, not reflow. | |
| """ | |
| text = _MD_IMAGE.sub("", text) # images: drop | |
| text = _MD_LINK.sub(r"\1", text) # links: keep label | |
| text = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", text) | |
| text = _LEADING_HASHES.sub("", text) | |
| text = _MD_BOLD.sub(r"\1", text) | |
| text = _MD_ITALIC.sub(r"\1", text) | |
| return text.replace("`", "") | |
| def _clean_heading(raw: str) -> tuple[str, int | None]: | |
| """A `title` block's markdown → (plain heading text, level from '#' depth).""" | |
| m = _LEADING_HASHES.match(raw) | |
| level = len(m.group(1)) if m else None | |
| text = _LEADING_HASHES.sub("", raw) | |
| text = _MD_BOLD.sub(r"\1", text) | |
| text = _MD_ITALIC.sub(r"\1", text) | |
| return text.strip(), level | |
| def _bbox(block: dict[str, Any]) -> list[int] | None: | |
| keys = ("top_left_x", "top_left_y", "bottom_right_x", "bottom_right_y") | |
| if all(k in block and block[k] is not None for k in keys): | |
| return [int(block[k]) for k in keys] | |
| return None | |
| def _blocks_with_pages(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]: | |
| """Flatten pages[].blocks into (page_idx, block), page order preserved.""" | |
| for page in response.get("pages", []): | |
| pidx = int(page.get("index", 0)) | |
| for block in page.get("blocks") or []: | |
| yield pidx, block | |
| def _image_descriptions(response: dict[str, Any]) -> dict[str, str]: | |
| """image_id -> description, from the page-level annotations (bbox_annotation). | |
| Empty when the OCR call did not request annotations; description then stays | |
| None, exactly as it does on MinerU's `pipeline` backend. | |
| """ | |
| out: dict[str, str] = {} | |
| for page in response.get("pages", []): | |
| for im in page.get("images") or []: | |
| ann = im.get("image_annotation") | |
| iid = im.get("id") | |
| if iid and ann: | |
| out[iid] = _unwrap_description(ann) | |
| return out | |
| def _unwrap_description(annotation: Any) -> str: | |
| """Pull the prose out of an annotation payload. | |
| Mistral returns bbox annotations JSON-WRAPPED — literally | |
| `{"description": "The diagram illustrates…"}` — and 0.3.x stored that string | |
| raw, braces and all, so every consumer got JSON where it expected a sentence. | |
| Unwrapped here, at the one place that reads the field. | |
| Never raises: anything unparseable falls back to the string form, because a | |
| slightly ugly description is worth more than a failed parse. | |
| """ | |
| if isinstance(annotation, dict): | |
| value = annotation.get("description") or annotation.get("text") | |
| return str(value) if value else json.dumps(annotation, ensure_ascii=False) | |
| if isinstance(annotation, str): | |
| text = annotation.strip() | |
| if text.startswith("{"): | |
| try: | |
| parsed = json.loads(text) | |
| except (ValueError, TypeError): | |
| return text | |
| if isinstance(parsed, dict): | |
| value = parsed.get("description") or parsed.get("text") | |
| if value: | |
| return str(value) | |
| return text | |
| return text | |
| return str(annotation) | |
| def _block_image_id(block: dict[str, Any]) -> str | None: | |
| """The image this block refers to, preferring the id in its own markdown. | |
| ⚠️ **`block["image_id"]` is not trustworthy.** Measured on the Open Pit | |
| textbook: page 6 carries two image blocks, and the FIRST one's `content` reads | |
| `` while its `image_id` field says **`img-2.jpeg`** — | |
| so both blocks claim the same image. Keying on `image_id` therefore resolved | |
| both to img-2's bytes, gave two different figures one content hash, and lost | |
| img-1 entirely (7 files written for 8 figures). | |
| The `content` markdown is the document's own reference and was correct in | |
| every case, so it wins; `image_id` is the fallback for a block whose content | |
| carries no reference. This also corrects an earlier diagnosis: the ids are | |
| globally unique across the response (`img-0` … `img-7`), so the duplication | |
| was never page-local naming — it was this mislabelled field. | |
| """ | |
| match = _MD_IMAGE_TARGET.search(block.get("content") or "") | |
| if match and match.group(1).strip(): | |
| return match.group(1).strip() | |
| return block.get("image_id") | |
| def _image_payloads(response: dict[str, Any]) -> dict[str, str]: | |
| """image_id -> base64 payload. Empty when `include_image_base64` was off. | |
| 0.3.x never read these: the bytes were returned by the API, recorded nowhere, | |
| and `Chunk.images` pointed at filenames that did not exist on disk. 0.4.0 | |
| content-addresses and persists them. | |
| """ | |
| out: dict[str, str] = {} | |
| for page in response.get("pages", []): | |
| for im in page.get("images") or []: | |
| iid, payload = im.get("id"), im.get("image_base64") | |
| if iid and payload: | |
| out[iid] = payload | |
| return out | |
| def _chunk_by_page( | |
| blocks: list[tuple[int, dict[str, Any]]], | |
| descriptions: dict[str, str], | |
| chapter_per_page: dict[int, str], | |
| doc_id: str, | |
| ) -> list[Chunk]: | |
| """Page granularity: all prose on a page → one chunk; tables/images their own. | |
| Recall-only. A page chunk has no single heading, so `heading_path` / | |
| `section_no` stay empty and the extraction outline degrades — section | |
| granularity (the default) is the real path. | |
| """ | |
| def chapters(page: int) -> list[str]: | |
| return [chapter_per_page[page]] if page in chapter_per_page else [] | |
| chunks: list[Chunk] = [] | |
| by_page: dict[int, list[tuple[int, dict[str, Any]]]] = {} | |
| for i, (pidx, block) in enumerate(blocks): | |
| by_page.setdefault(pidx, []).append((i, block)) | |
| for pidx in sorted(by_page): | |
| pieces: list[str] = [] | |
| latex: list[str] = [] | |
| src: list[int] = [] | |
| has_formula = False | |
| prose_bbox: list[int] | None = None | |
| for i, block in by_page[pidx]: | |
| kind = block.get("type") | |
| if kind in _DISCARDED or kind == "header": | |
| continue | |
| if kind == "table": | |
| c = Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, | |
| kind="table", text=render_table(block.get("content") or ""), | |
| page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx), | |
| is_tabular=True, table_html=block.get("content") or None, | |
| source_items=[i], bbox=_bbox(block), | |
| ) | |
| if c.text or c.images: | |
| chunks.append(c) | |
| continue | |
| if kind == "image": | |
| iid = block.get("image_id") | |
| chunks.append(Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, | |
| kind="image", text="", page_idx=pidx, page_idxs=[pidx], | |
| chapters=chapters(pidx), | |
| generated_description=descriptions.get(iid) if iid else None, | |
| source_items=[i], bbox=_bbox(block), | |
| images=[iid] if iid else [], | |
| )) | |
| continue | |
| if kind == "equation": | |
| raw = (block.get("content") or "").strip().strip("$").strip() | |
| has_formula = True | |
| src.append(i) | |
| if raw: | |
| latex.append(raw) | |
| rendered = render_latex(raw) | |
| if rendered: | |
| pieces.append(rendered) | |
| continue | |
| # text / title / list / caption / signature -> prose. A numbered | |
| # heading is NOT a boundary in page mode; it stays as prose text. | |
| text = _strip_markdown(block.get("content") or "").rstrip() | |
| if not text.strip(): | |
| continue | |
| if prose_bbox is None: | |
| prose_bbox = _bbox(block) | |
| src.append(i) | |
| pieces.append(text) | |
| if pieces: | |
| chunks.append(Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, | |
| kind="equation" if has_formula else "text", | |
| text="\n\n".join(pieces).strip(), | |
| page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx), | |
| has_formula=has_formula, latex=latex, source_items=src, | |
| bbox=prose_bbox, | |
| )) | |
| return chunks | |
| def normalise_mistral( | |
| response: dict[str, Any], | |
| doc_id: str, | |
| granularity: str = "section", | |
| asset_sink: AssetSink | None = None, | |
| ) -> list[Chunk]: | |
| """Mistral OCR response (include_blocks=True) → chunks. | |
| `granularity` decides how prose is grouped, and only that: | |
| - "section" (default): a numbered heading opens a new chunk, mirroring | |
| normalize.py. This is the default because the extraction half derives | |
| the document `outline` from `heading_path` and drives `source_wording` | |
| from `heading` — page grouping leaves both EMPTY (a page chunk has no | |
| single heading), which silently degrades a locked product behaviour. | |
| - "page": all prose on a page becomes ONE chunk. Higher raw term recall | |
| (bigger context per NER window) but no per-chunk heading, so only fit | |
| for a recall-only measurement, never for the real pipeline. | |
| The section/page recall gap (0.878 vs 0.951 on the BUMA fixture) is the NER | |
| filter's, not content: every gold term is present in the text either way, and | |
| closing it is the extraction side's lever (span-filter overlap resolution). | |
| In BOTH modes tables and images are their OWN chunks, so `table_html` stays | |
| one-table-per-chunk regardless. Switching modes is a one-word change and | |
| never touches the extraction half — the artifact shape is identical. | |
| No `page_vocabulary` parameter: Mistral does not spell formulas out one | |
| character at a time, so the word-boundary recovery that MinerU needs has | |
| nothing to do here. | |
| """ | |
| if granularity not in {"page", "section"}: | |
| raise ValueError(f"granularity must be 'page' or 'section', got {granularity!r}") | |
| blocks = list(_blocks_with_pages(response)) | |
| if not blocks: | |
| raise ValueError( | |
| "no blocks in Mistral response — call ocr.process with include_blocks=True" | |
| ) | |
| descriptions = _image_descriptions(response) | |
| # Chapter context per page, from the running header (type 'header'). | |
| chapter_per_page: dict[int, str] = {} | |
| for pidx, b in blocks: | |
| if b.get("type") == "header": | |
| txt = _strip_markdown(b.get("content") or "").strip() | |
| if txt: | |
| chapter_per_page.setdefault(pidx, txt) | |
| # --- page mode: recall-only; a page chunk has no single heading --- | |
| if granularity == "page": | |
| # ⚠️ FENCED OFF AT 0.4.0, deliberately, rather than given the asset | |
| # treatment. Page mode emits 0.3.x-shaped chunks — images standalone with | |
| # empty text, no `assets` — so letting it run would quietly produce an | |
| # artifact that CLAIMS schema_version 0.4.0 while having the old shape. | |
| # Two normalisers silently diverging is exactly how the `type: "list"` | |
| # content drop survived unnoticed and cost five gold terms. | |
| # | |
| # Page mode exists only for a recall-only measurement (it nulls | |
| # `heading_path`, which the outline and `source_wording` both need), so it | |
| # is not worth duplicating the asset logic into. Restore by porting the | |
| # image/table branches from the section path below. | |
| raise NotImplementedError( | |
| "granularity='page' is not supported at schema 0.4.0: it would emit " | |
| "0.3.x-shaped chunks (standalone image chunks, no assets) under a " | |
| "0.4.0 version stamp. Use granularity='section', or port the asset " | |
| "handling into _chunk_by_page first." | |
| ) | |
| # --- section mode (default): a numbered heading opens a new chunk (mirrors normalize.py) --- | |
| chunks: list[Chunk] = [] | |
| running: Chunk | None = None | |
| pieces: list[str] = [] | |
| stack: list[tuple[int, str]] = [] | |
| payloads = _image_payloads(response) | |
| # A table's asset has to be added to the PROSE chunk that precedes it, but | |
| # that chunk is already closed and appended by then — so the back-reference is | |
| # queued and resolved once, after the loop. | |
| _pending_back_refs: list[tuple[str, Asset]] = [] | |
| # A printed caption arrives as its own `caption` block AFTER its image block. | |
| # The figure waiting for one is held here; the caption is attached to the asset | |
| # AND left in the prose text, because it is verbatim document content. | |
| _pending_caption_for: list[Asset] = [] | |
| def push_heading(level: int, text: str) -> None: | |
| while stack and stack[-1][0] >= level: | |
| stack.pop() | |
| stack.append((level, text)) | |
| def close() -> None: | |
| nonlocal running, pieces | |
| if running is not None: | |
| running.text = "\n\n".join(pieces).strip() | |
| if running.text or running.assets or running.images: | |
| running.page_idxs = sorted(set(running.page_idxs)) | |
| chunks.append(running) | |
| running, pieces = None, [] | |
| def open_chunk( | |
| kind: str, page: int, section_no: str | None = None, heading: str | None = None | |
| ) -> Chunk: | |
| return Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", | |
| doc_id=doc_id, kind=kind, text="", | |
| page_idx=page, page_idxs=[page], | |
| section_no=section_no, heading=heading, | |
| chapters=[chapter_per_page[page]] if page in chapter_per_page else [], | |
| heading_path=[t for _, t in stack], | |
| ) | |
| for i, (page, block) in enumerate(blocks): | |
| kind = block.get("type") | |
| if kind in _DISCARDED or kind == "header": | |
| continue | |
| if kind == "table": | |
| # Tables stay their OWN chunk (0.4.0 kept this deliberately): unlike a | |
| # figure, a table's linearised text carries real terms the filter | |
| # finds — 74.6% of the Komatsu manual's top-3 evidence is tabular. What | |
| # 0.4.0 adds is reachability: an `Asset` plus a two-way pointer, so the | |
| # prose that introduces a table can be followed to it and back. | |
| prose_before = running.chunk_id if running is not None else None | |
| close() | |
| c = open_chunk("table", page) | |
| c.text = render_table(block.get("content") or "") | |
| c.is_tabular = True | |
| c.table_html = block.get("content") or None | |
| c.source_items = [i] | |
| c.bbox = _bbox(block) | |
| table_asset = Asset( | |
| # No bytes to hash, so the id is derived from stable document | |
| # coordinates instead — deterministic across re-parses, which is | |
| # what a durable pointer needs. | |
| asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)), | |
| kind="table", | |
| page_idx=page, | |
| bbox=_bbox(block), | |
| ) | |
| c.assets = [table_asset] | |
| if prose_before is not None: | |
| c.referenced_by = [prose_before] | |
| _pending_back_refs.append((prose_before, table_asset)) | |
| if c.text or c.assets: | |
| chunks.append(c) | |
| continue | |
| if kind == "image": | |
| # ── 0.4.0: a figure is NO LONGER ITS OWN CHUNK. ──────────────────── | |
| # It used to be, with an EMPTY `text`, which made it unreachable by | |
| # construction: no text -> no term mentions -> joins no cluster -> | |
| # never selected as evidence. Measured across three documents: 13 | |
| # figures, every one described by a vision model we pay for, and ZERO | |
| # appearing in any cluster's evidence. Meanwhile the prose saying | |
| # "Figure 1.1" had its `` reference stripped, so nothing | |
| # connected the sentence to the picture. | |
| # | |
| # Now the reference is folded into the RUNNING prose chunk at its | |
| # reading-order position — exactly where the document put it — and the | |
| # figure travels as an `Asset` on that chunk. | |
| iid = _block_image_id(block) | |
| identifier, storage_key = store_asset( | |
| asset_sink, iid, payloads.get(iid) if iid else None | |
| ) | |
| # Fall back to document coordinates when the parser returned no bytes, | |
| # so a figure is still referenced rather than silently vanishing. | |
| if identifier is None: | |
| identifier = synthetic_asset_id(doc_id, str(page), str(i)) | |
| asset = Asset( | |
| asset_id=identifier, | |
| kind="figure", | |
| storage_key=storage_key, | |
| page_idx=page, | |
| bbox=_bbox(block), | |
| description=descriptions.get(iid) if iid else None, | |
| description_source=_DESCRIPTION_SOURCE if iid else None, | |
| ) | |
| # P5 edge case: a figure can precede any prose on its page (Open Pit | |
| # p.3 is close to this shape). Open a chunk rather than drop the asset. | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| running.assets.append(asset) | |
| # Deprecated alias, one version only. | |
| if iid: | |
| running.images.append(iid) | |
| # The reference itself, inline in the prose. `asset://` rather than the | |
| # parser's filename because the filename is page-local and collides. | |
| pieces.append(f"") | |
| _pending_caption_for.append(asset) | |
| continue | |
| if kind == "equation": | |
| latex = (block.get("content") or "").strip().strip("$").strip() | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.has_formula = True | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| if latex: | |
| running.latex.append(latex) # raw, for the formula branch | |
| rendered = render_latex(latex) | |
| if rendered: | |
| pieces.append(rendered) # prose, so NER can find it | |
| continue | |
| raw = block.get("content") or "" | |
| # `list` and `caption` are ALWAYS content, never a section boundary: a | |
| # bullet or a caption can look like a numbered heading, and letting one | |
| # open a section would split a list from the paragraph introducing it | |
| # (see normalize.py's list note). | |
| if kind in {"list", "caption"}: | |
| text = _strip_markdown(raw).rstrip() | |
| if text.strip(): | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.bbox = _bbox(block) | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| pieces.append(text) | |
| # A caption directly after a figure names that figure. Attach it to | |
| # the asset so a consumer can render "Figure 1.1 …" beside the | |
| # image — and LEAVE it in `pieces` too, because it is printed in | |
| # the document: verbatim, quotable as evidence, span-checkable. | |
| # Copying it is the point; moving it would silently delete | |
| # evidence that survived 17/17 times on the Open Pit textbook. | |
| if kind == "caption" and _pending_caption_for: | |
| _pending_caption_for.pop(0).caption = text.strip() | |
| continue | |
| # `text` / `title` / `signature`: a NUMBERED line may open a section. | |
| # normalize._heading is applied to BOTH text and title items on MinerU — | |
| # here too, because Mistral labels many numbered headings as `text`, not | |
| # `title`. Checking only `title` left whole sections unopened and every | |
| # equation piled into one chunk. A title's level comes from its `#` | |
| # depth; a text line's from numbering depth (fallback inside _heading). | |
| if kind == "title": | |
| htext, level = _clean_heading(raw) | |
| else: | |
| htext, level = _strip_markdown(raw).strip(), None | |
| number, heading, hlevel = _heading( | |
| {"type": "title", "text": htext, "text_level": level} | |
| ) | |
| if heading is not None: | |
| if any(htext == t for _, t in stack): # reprinted breadcrumb | |
| if running is not None: | |
| running.page_idxs.append(page) | |
| continue | |
| close() | |
| push_heading(hlevel or 1, heading) | |
| running = open_chunk("text", page, section_no=number, heading=heading) | |
| running.source_items = [i] | |
| running.bbox = _bbox(block) | |
| # heading line NOT copied into text — lives in `heading`. See | |
| # normalize.py for the measured reason (mention_count inflation). | |
| continue | |
| text = _strip_markdown(raw).rstrip() | |
| if not text.strip(): | |
| continue | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.bbox = _bbox(block) | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| chapter = chapter_per_page.get(page) | |
| if chapter and chapter not in running.chapters: | |
| running.chapters.append(chapter) | |
| pieces.append(text) # verbatim (markup only removed) | |
| if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: | |
| continued_from = running | |
| close() | |
| running = open_chunk("text", page, | |
| section_no=continued_from.section_no, | |
| heading=continued_from.heading) | |
| close() | |
| # Two-way table pointers, resolved once the loop is done. The table chunk | |
| # already names the prose chunk in `referenced_by`; this adds the table's asset | |
| # to that prose chunk so the link is navigable from the prose side too — the | |
| # direction a reader, and a model reading evidence, actually travels. Deferred | |
| # rather than done inline because the prose chunk is already closed and | |
| # appended by the time its table is reached. | |
| by_id = {c.chunk_id: c for c in chunks} | |
| for prose_id, table_asset in _pending_back_refs: | |
| host = by_id.get(prose_id) | |
| if host is not None and not any( | |
| a.asset_id == table_asset.asset_id for a in host.assets | |
| ): | |
| host.assets.append(table_asset) | |
| return chunks | |