Download src/knowledge_extraction/extract/base.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 4.39 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/extract/base.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/extract/base.py
-
curl -L -o base.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/extract/base.py
4.39 kB
| """Shared plumbing for the four extraction branches. | |
| Prompts live in `prompts/*.txt`, never in code, for two reasons: a prompt change | |
| is not a code change, and **the fixed prefix must stay byte-identical across | |
| calls** or prompt caching silently stops engaging at roughly 10x the input cost. | |
| """ | |
| from __future__ import annotations | |
| from functools import lru_cache | |
| from pathlib import Path | |
| from ..models import Chunk | |
| from ..settings import CACHE_MIN_TOKENS | |
| PROMPT_DIR = Path(__file__).parent / "prompts" | |
| def est_tokens(text: str) -> int: | |
| """Cheap estimate, for dry-run budgeting only. Real counts come from the | |
| API's usage object — never report a cached price from an estimate.""" | |
| return max(1, int(len(text) / 3.6)) | |
| def load_prompt(branch: str) -> str: | |
| return (PROMPT_DIR / f"{branch}.txt").read_text(encoding="utf-8") | |
| def prefix_tokens(branch: str) -> int: | |
| return est_tokens(load_prompt(branch)) | |
| def cacheable(branch: str) -> bool: | |
| """Whether the fixed prefix is long enough to cache at all. | |
| Reported, never assumed: caching does not engage below the floor, so a | |
| shorter prefix caches nothing. Only the API's `cached_tokens` proves a hit. | |
| """ | |
| return prefix_tokens(branch) >= CACHE_MIN_TOKENS | |
| def evidence_block(chunks: list[Chunk], scores: list[float] | None = None) -> str: | |
| """Evidence labelled with chunk_id, section and page so the model can cite | |
| provenance and we can trace which evidence produced which field. | |
| **The heading is included, and must stay included.** Two reasons: | |
| 1. Indonesian standards name the term in the heading and open the body with | |
| the definition — "2.1.3 Physical of Availability (PA)" / "Adalah | |
| ketersediaan fisik…" — so the body often never repeats the term. Without | |
| the heading the model is asked to define a term the evidence never names. | |
| 2. It keeps one invariant true: **what the model reads is exactly what the | |
| span check searches.** `validate.evidence_text` composes heading + text; | |
| if this block showed only the text, the model could never quote a | |
| section title, and any field that did quote one would be rejected as | |
| unlocatable. | |
| """ | |
| parts = [] | |
| for i, chunk in enumerate(chunks): | |
| score = f" score={scores[i]:.1f}" if scores and i < len(scores) else "" | |
| head = f"{chunk.heading}\n" if chunk.heading else "" | |
| parts.append( | |
| f"[chunk_id={chunk.chunk_id} section={chunk.section_no or '-'} " | |
| f"page={chunk.page_start}{score}]\n{head}{chunk.text}" | |
| + _asset_block(chunk) | |
| ) | |
| return "EVIDENCE\n" + "\n\n---\n\n".join(parts) | |
| def _asset_block(chunk: Chunk) -> str: | |
| """Figures and tables this chunk references, appended to its evidence. 0.4.0. | |
| The chunk's own text already carries each figure's `` reference | |
| inline, at the position the document put it. This resolves those references so | |
| the model can actually use them — before 0.4.0 a figure was an unreachable | |
| chunk with empty text, and its vision-model description never reached a prompt | |
| at all. | |
| **The verbatim/generated split is labelled, not merely recorded.** A caption is | |
| printed in the document, so it is quotable as evidence and the span check can | |
| locate it. A description is a model's claim about a picture: it cannot be | |
| span-checked against anything, so quoting it would produce a field that the | |
| validator then nulls — wasting the call and looking like a model failure. The | |
| label is what stops that. | |
| """ | |
| if not chunk.assets: | |
| return "" | |
| lines = ["", "ASSETS REFERENCED BY THIS CHUNK"] | |
| for asset in chunk.assets: | |
| where = f" · page {asset.page_no}" if asset.page_no else "" | |
| lines.append(f" [asset://{asset.asset_id} · {asset.kind}{where}]") | |
| if asset.caption: | |
| lines.append(f" caption (VERBATIM, quotable as evidence): {asset.caption}") | |
| if asset.description: | |
| lines.append( | |
| " description (MODEL-GENERATED, not from the document — " | |
| "do NOT quote as evidence):" | |
| ) | |
| lines.append(f" {asset.description}") | |
| if not asset.caption and not asset.description: | |
| lines.append(" (no caption or description available)") | |
| return "\n".join(lines) | |