Download hv_reader.py from zeechimp/hv-reader: direct link, hf CLI and curl.
- Browser
- Download file 48.5 kB
-
https://huggingface.co/zeechimp/hv-reader/resolve/main/hv_reader.py
- Command line
-
hf download hf://zeechimp/hv-reader/hv_reader.py
-
curl -L -o hv_reader.py https://huggingface.co/zeechimp/hv-reader/resolve/main/hv_reader.py
48.5 kB
| """ | |
| hv-reader | |
| ========= | |
| The reading experience, in one call. | |
| Given a text, produce a ReadingProfile: pace, memory, passes, slip, and | |
| wall — the five axes that describe what it is like to read a text. | |
| This unifies seven component models: | |
| hv-tempo → pace | |
| hv-forget → memory | |
| hv-ttu → ttu_s (total time) | |
| hv-fold → passes | |
| hv-slip → slip | |
| hv-hunger → hunger (internal, feeds slip and wall) | |
| hv-wall → wall | |
| The profile is the artifact. The axes are the readings. The signature is | |
| what predicts whether a text gets finished. | |
| Pure stdlib. No dependencies. | |
| Author: zeechimp | |
| License: Apache-2.0 | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import math | |
| import os | |
| import re | |
| import sys | |
| from collections import Counter, defaultdict | |
| from dataclasses import dataclass, asdict, field | |
| from typing import Any, Dict, List, Optional, Set, Tuple | |
| # ============================================================================ | |
| # Lexicons | |
| # ============================================================================ | |
| COMMON_WORDS = frozenset(""" | |
| the be to of and a in that have i it for not on with he as you do at this | |
| but his by from they we say her she or an will my one all would there their | |
| what so up out if about who get which go me when make can like time no just | |
| him know take people into year your good some could them see other than then | |
| now look only come its over think also back after use two how our work first | |
| well way even new want because any these give day most us is are was were | |
| been being has had having does did doing will would shall should can could | |
| may might must man woman child water fire earth air sun moon star light dark | |
| hand head eye ear mouth nose foot leg arm body face heart mind life death | |
| food bread milk meat fish tree flower grass leaf root seed farm field hill | |
| mountain river sea lake boat ship road street city town house room door | |
| window wall floor roof bed chair table book page word line letter number | |
| name place thing part side end start middle top bottom front back left | |
| right high low long short big small old new young hot cold wet dry clean | |
| dirty light heavy soft hard fast slow easy true false good bad happy sad | |
| love hate fear hope help hurt win lose give take send bring buy sell pay | |
| cost money price work play run walk jump sit stand sleep wake eat drink | |
| cook wash read write speak hear see feel know think learn teach ask | |
| answer tell show hide open close push pull carry hold drop throw catch | |
| break fix build make do try use move turn stop start keep leave stay wait | |
| meet join save spend show thank want wish walk stop continue morning | |
| climb row push pull begin finish start end remain rest return arrive | |
| depart leave enter exit follow lead sit stand lie rise fall drop | |
| still quiet calm slow fast soft loud bright dark warm cool fresh clean | |
| never always often sometimes rarely usually speak spoke spoken | |
| take took taken give gave given see saw seen know knew known | |
| think thought thought come came come go went gone say said said | |
| tell told told find found found hold held held bring brought brought | |
| buy bought bought teach taught taught catch caught caught build built built | |
| send sent sent spend spent lose lost lost lead led led meet met met | |
| read read read write wrote written run ran run swim swam swum | |
| thing things word words time times year years day days | |
| man men woman women child children person people | |
| place places work works way ways life lives hand hands | |
| eye eyes part parts end ends line lines side sides | |
| name names head heads house houses friend friends | |
| family families group groups country countries | |
| world worlds city cities school schools | |
| """.split()) | |
| ABSTRACT_SUFFIXES = ( | |
| "tion", "sion", "ism", "ity", "ness", "ance", "ence", | |
| "ship", "hood", "ment", "ology", "itude", "acy", | |
| ) | |
| SUBORDINATORS = frozenset(""" | |
| which that because although though while whereas since when if unless | |
| provided assuming given whenever wherever whoever whichever | |
| """.split()) | |
| HEDGES = frozenset(""" | |
| may might maybe perhaps possibly probably typically usually often | |
| generally roughly approximately about somewhat rather | |
| """.split()) | |
| CONDITIONALS = frozenset(""" | |
| if when unless provided assuming suppose supposing | |
| """.split()) | |
| NEGATIONS = frozenset(""" | |
| not no never none without cannot can't don't doesn't won't | |
| isn't aren't wasn't weren't nor neither | |
| """.split()) | |
| BE_FORMS = frozenset(""" | |
| is are was were be been being am | |
| """.split()) | |
| REENTRY_MARKERS = frozenset(""" | |
| above below previous preceding following | |
| aforementioned noted mentioned discussed described | |
| stated referred earlier later | |
| """.split()) | |
| ARTICLES = frozenset(""" | |
| the this that these those such said aforementioned | |
| """.split()) | |
| UNIQUE_REFERENTS = frozenset(""" | |
| sun moon earth world sky ground horizon | |
| morning afternoon evening night noon midnight dawn dusk | |
| """.split()) | |
| CERTAINTY_MARKERS = frozenset(""" | |
| definitely definitively certainly obviously clearly undoubtedly | |
| unquestionably absolutely surely plainly evidently undeniably | |
| unmistakably decidedly categorically conclusively decisively | |
| resolutely proves proven prove proved proof impossible must always | |
| never guaranteed | |
| """.split()) | |
| CONCLUSION_MARKERS = ( | |
| "therefore", "thus", "hence", "consequently", "accordingly", | |
| "it follows that", "we conclude", "we can conclude", | |
| "this shows", "this demonstrates", "this proves", | |
| "in conclusion", "as a result", | |
| ) | |
| CONTRAST_MARKERS = ( | |
| "however", "but", "yet", "although", "though", "nevertheless", | |
| "nonetheless", "conversely", "on the contrary", "in contrast", | |
| "on the other hand", "by contrast", "notwithstanding", | |
| "despite this", "even so", | |
| ) | |
| EVIDENCE_MARKERS = ( | |
| "according to", "studies show", "studies suggest", | |
| "research shows", "research suggests", | |
| "data show", "data suggest", "evidence indicates", | |
| "we measured", "we observed", "we found", | |
| "for example", "for instance", "specifically", "namely", | |
| "for one", "in fact", "as measured", | |
| ) | |
| QUESTION_RAISERS = ( | |
| "why", "how", "whether", "what caused", "the reason", | |
| "unclear", "unknown", "remains to be determined", | |
| "remains unclear", "puzzling", "mysterious", | |
| "unexplained", "open question", "puzzle", | |
| ) | |
| ANSWER_MARKERS = ( | |
| "because", "since", "as a result", "due to", "explained by", | |
| "the reason is", "this explains", "the cause", "attributable to", | |
| "results from", "arises from", "the mechanism is", | |
| ) | |
| DIRECTION_GROUPS = { | |
| "up": frozenset(""" | |
| increase increases increased increasing rise rises rose risen | |
| grow grows grew grown growth expand expands expanded expansion | |
| raise raises raised raising improve improves improved improving | |
| gain gains gained gaining positive higher highest more most | |
| upward up climb climbs climbed climbing | |
| """.split()), | |
| "down": frozenset(""" | |
| decrease decreases decreased decreasing fall falls fell fallen | |
| shrink shrinks shrank shrunk contract contracts contracted | |
| lower lowers lowered lowering reduce reduces reduced reducing | |
| worsen worsens worsened worsening lose loses lost losing | |
| negative lower lowest less least downward down decline declines | |
| declined declining drop drops dropped dropping diminish | |
| """.split()), | |
| "cause": frozenset(""" | |
| cause causes caused causing produce produces produced producing | |
| create creates created creating induce induces induced | |
| trigger triggers triggered triggering generate | |
| """.split()), | |
| "prevent": frozenset(""" | |
| prevent prevents prevented preventing avoid avoids avoided | |
| block blocks blocked blocking inhibit inhibits inhibited | |
| prohibit prohibits prohibited prohibiting stop stops stopped | |
| """.split()), | |
| "support": frozenset(""" | |
| support supports supported supporting confirm confirms confirmed | |
| agree agrees agreed approve approves approved accept accepts accepted | |
| affirm affirms affirmed | |
| """.split()), | |
| "oppose": frozenset(""" | |
| oppose opposes opposed opposing deny denies denied contradict | |
| contradicts contradicted disagree disagrees disagreed | |
| reject rejects rejected refuse refuses refused | |
| """.split()), | |
| } | |
| GROUP_OPPOSITES = frozenset([ | |
| ("up", "down"), ("down", "up"), | |
| ("cause", "prevent"), ("prevent", "cause"), | |
| ("support", "oppose"), ("oppose", "support"), | |
| ]) | |
| DISCOURSE_MARKERS = frozenset(""" | |
| therefore thus hence consequently accordingly however but yet although | |
| though nevertheless nonetheless conversely meanwhile similarly | |
| moreover furthermore additionally | |
| """.split()) | |
| # ============================================================================ | |
| # Regexes | |
| # ============================================================================ | |
| _WORD_RE = re.compile(r"[A-Za-z][A-Za-z'\-]*") | |
| _SENT_SPLIT_RE = re.compile(r"(?<=[.!?])\s+(?=[A-Z\"'(])") | |
| _NUMBER_RE = re.compile(r"\b\d+(?:[.,]\d+)*\b") | |
| _STANDALONE_DEMON_RE = re.compile( | |
| r'^\s*(this|that|these|those)\s*[.,!?;:]*\s*$', re.IGNORECASE | |
| ) | |
| # ============================================================================ | |
| # Tokenization helpers | |
| # ============================================================================ | |
| def _stem(w: str) -> str: | |
| w = w.lower() | |
| if len(w) <= 4: | |
| return w | |
| for suffix in ("ingly", "edly", "ing", "ed", "ly", "es", "s"): | |
| if w.endswith(suffix) and len(w) - len(suffix) >= 3: | |
| base = w[: -len(suffix)] | |
| if len(base) >= 2 and base[-1] == base[-2] and base[-1] not in "aeiou": | |
| base = base[:-1] | |
| return base | |
| return w | |
| def _words(text: str) -> List[str]: | |
| return _WORD_RE.findall(text) | |
| def _content_words(text: str) -> List[str]: | |
| return [ | |
| w.lower() for w in _words(text) | |
| if w.lower() not in COMMON_WORDS | |
| and w.lower() not in DISCOURSE_MARKERS | |
| and len(w) >= 3 | |
| ] | |
| def _sentences(text: str) -> List[str]: | |
| return [s.strip() for s in _SENT_SPLIT_RE.split(text) if s.strip()] | |
| def _is_common(w: str) -> bool: | |
| lw = w.lower() | |
| if lw in COMMON_WORDS: | |
| return True | |
| return _stem(lw) in COMMON_WORDS | |
| def _is_rare(w: str) -> bool: | |
| return len(w) >= 7 and not _is_common(w) | |
| def _has_conclusion_marker(text: str) -> bool: | |
| low = text.lower().strip() | |
| for m in CONCLUSION_MARKERS: | |
| if low.startswith(m): | |
| return True | |
| if f" {m} " in f" {low} ": | |
| return True | |
| return False | |
| def _has_contrast_marker(text: str) -> Tuple[bool, str]: | |
| low = text.lower().strip() | |
| for m in CONTRAST_MARKERS: | |
| if low.startswith(m): | |
| return True, m | |
| if f" {m} " in f" {low} ": | |
| return True, m | |
| return False, "" | |
| def _has_evidence_marker(text: str) -> bool: | |
| low = text.lower() | |
| return any(m in low for m in EVIDENCE_MARKERS) | |
| # ============================================================================ | |
| # Config | |
| # ============================================================================ | |
| class HVReaderConfig: | |
| # Pace (hv-tempo weights) | |
| baseline_wpm: float = 220.0 | |
| w_sentence_len_excess: float = 0.40 | |
| w_clause_rate: float = 0.08 | |
| w_rare_rate: float = 0.70 | |
| w_abstract_rate: float = 0.40 | |
| w_digit_rate: float = 0.30 | |
| w_negation_rate: float = 0.40 | |
| w_hedge_rate: float = 0.40 | |
| w_conditional_rate: float = 0.60 | |
| w_passive_rate: float = 0.30 | |
| w_list_bonus: float = -0.50 | |
| max_log_slowdown: float = 1.5 | |
| # Memory (hv-forget) | |
| memory_target_days: float = 7.0 | |
| memory_stability_base: float = 1.0 | |
| memory_stability_density: float = 5.0 | |
| memory_stability_rare: float = 3.0 | |
| memory_stability_salience: float = 2.0 | |
| # Fold (hv-fold) | |
| fold_demon_weight: float = 2.0 | |
| fold_definite_only: float = 0.5 | |
| fold_reentry: float = 1.0 | |
| fold_forward: float = 0.5 | |
| fold_resolved: float = 0.1 | |
| exempt_first_sentence: bool = False | |
| # Slip (hv-slip) | |
| monotone_window: int = 3 | |
| slip_monotone_weight: float = 0.4 | |
| slip_repetition_weight: float = 0.3 | |
| slip_absence_weight: float = 0.3 | |
| # Hunger (hv-hunger) | |
| hunger_question_weight: float = 1.0 | |
| hunger_answer_weight: float = 1.0 | |
| # Wall (hv-wall) | |
| wall_threshold: float = 0.35 | |
| certainty_scale: float = 6.0 | |
| hedge_scale: float = 5.0 | |
| contradiction_scale: float = 0.4 | |
| specificity_scale: float = 5.0 | |
| gap_confidence_jump_weight: float = 0.5 | |
| contrast_bonus: float = 0.15 | |
| contradiction_min: float = 0.30 | |
| overclaim_min: float = 0.30 | |
| gap_min: float = 0.25 | |
| # Reporting | |
| min_sentence_words: int = 3 | |
| top_reasons: int = 3 | |
| version: str = "0.1.1" | |
| # ============================================================================ | |
| # Report dataclasses | |
| # ============================================================================ | |
| class SpanProfile: | |
| id: int | |
| text: str | |
| span: Tuple[int, int] | |
| n_words: int | |
| wpm: float | |
| slowdown: float | |
| density: float | |
| stability_days: float | |
| retention_week: float | |
| fold_load: float | |
| slip: float | |
| hunger_delta: float | |
| hunger: float | |
| wall: float | |
| wall_type: str | |
| reasons: List[str] = field(default_factory=list) | |
| def to_dict(self) -> dict: | |
| return { | |
| "id": self.id, | |
| "text": self.text, | |
| "span": list(self.span), | |
| "n_words": self.n_words, | |
| "wpm": round(self.wpm, 2), | |
| "slowdown": round(self.slowdown, 3), | |
| "density": round(self.density, 4), | |
| "stability_days": round(self.stability_days, 3), | |
| "retention_week": round(self.retention_week, 4), | |
| "fold_load": round(self.fold_load, 3), | |
| "slip": round(self.slip, 3), | |
| "hunger_delta": round(self.hunger_delta, 3), | |
| "hunger": round(self.hunger, 3), | |
| "wall": round(self.wall, 4), | |
| "wall_type": self.wall_type, | |
| "reasons": list(self.reasons), | |
| } | |
| class ReadingProfile: | |
| text: str | |
| n_sentences: int | |
| n_words: int | |
| pace: float | |
| memory: float | |
| passes: float | |
| slip: float | |
| wall: float | |
| ttu_s: float | |
| mean_wpm: float | |
| hunger_final: float | |
| spans: List[SpanProfile] | |
| slowest_span: Optional[SpanProfile] | |
| wall_span: Optional[SpanProfile] | |
| summary: str | |
| def to_dict(self) -> dict: | |
| return { | |
| "text": self.text, | |
| "n_sentences": self.n_sentences, | |
| "n_words": self.n_words, | |
| "profile": { | |
| "pace": round(self.pace, 4), | |
| "memory": round(self.memory, 4), | |
| "passes": round(self.passes, 4), | |
| "slip": round(self.slip, 4), | |
| "wall": round(self.wall, 4), | |
| }, | |
| "ttu_s": round(self.ttu_s, 2), | |
| "mean_wpm": round(self.mean_wpm, 2), | |
| "hunger_final": round(self.hunger_final, 3), | |
| "slowest_span_id": self.slowest_span.id if self.slowest_span else None, | |
| "wall_span_id": self.wall_span.id if self.wall_span else None, | |
| "summary": self.summary, | |
| "spans": [s.to_dict() for s in self.spans], | |
| } | |
| # ============================================================================ | |
| # Pace (hv-tempo) | |
| # ============================================================================ | |
| def _count_list_markers(text: str) -> int: | |
| bullets = len(re.findall(r"(?:^|\n)\s*[-*•]\s+\S", text)) | |
| numbered = re.findall(r"(?:^|[\s;:.])\d+[.)]\s+\S", text) | |
| n_numbered = len(numbered) if len(numbered) >= 2 else 0 | |
| return max(bullets, n_numbered) | |
| def _pace_features(sentence: str) -> Dict[str, float]: | |
| words = _words(sentence) | |
| n_words = len(words) | |
| if n_words == 0: | |
| return {k: 0.0 for k in [ | |
| "n_words", "mean_sentence_len", "sentence_len_signed", | |
| "clauses_per_sentence", "rare_rate", "abstract_rate", | |
| "digit_rate", "negation_rate", "hedge_rate", | |
| "conditional_rate", "passive_rate", "list_rate", | |
| ]} | |
| n_sentences = max(1, len([s for s in _SENT_SPLIT_RE.split(sentence) if s.strip()])) | |
| mean_len = n_words / n_sentences | |
| signed_len = (mean_len - 15.0) / 10.0 | |
| n_clauses = ( | |
| sentence.count(",") + sentence.count(";") + sentence.count(":") | |
| + sum(1 for w in words if w.lower() in SUBORDINATORS) | |
| ) | |
| clauses_per_sentence = n_clauses / n_sentences | |
| n_rare = sum(1 for w in words if len(w) >= 7 and not _is_common(w)) | |
| rare_rate = n_rare / n_words | |
| n_abstract = sum( | |
| 1 for w in words | |
| if len(w) > 5 and w.lower().endswith(ABSTRACT_SUFFIXES) | |
| ) | |
| abstract_rate = n_abstract / n_words | |
| digit_rate = len(_NUMBER_RE.findall(sentence)) / n_words | |
| negation_rate = sum(1 for w in words if w.lower() in NEGATIONS) / n_words | |
| hedge_rate = sum(1 for w in words if w.lower() in HEDGES) / n_words | |
| conditional_rate = sum(1 for w in words if w.lower() in CONDITIONALS) / n_words | |
| n_passive = 0 | |
| for i, w in enumerate(words): | |
| if w.lower() in BE_FORMS and i + 1 < len(words): | |
| nxt = words[i + 1].lower() | |
| if (nxt.endswith("ed") and len(nxt) > 3) or nxt in ( | |
| "gone", "seen", "written", "taken", "made", "known", | |
| "found", "given", "held", "sent", "left", "kept", | |
| ): | |
| n_passive += 1 | |
| passive_rate = n_passive / n_sentences | |
| n_list = _count_list_markers(sentence) | |
| list_rate = n_list / n_sentences | |
| return { | |
| "n_words": float(n_words), | |
| "mean_sentence_len": mean_len, | |
| "sentence_len_signed": signed_len, | |
| "clauses_per_sentence": clauses_per_sentence, | |
| "rare_rate": rare_rate, | |
| "abstract_rate": abstract_rate, | |
| "digit_rate": digit_rate, | |
| "negation_rate": negation_rate, | |
| "hedge_rate": hedge_rate, | |
| "conditional_rate": conditional_rate, | |
| "passive_rate": passive_rate, | |
| "list_rate": list_rate, | |
| } | |
| def _pace_slowdown( | |
| f: Dict[str, float], cfg: HVReaderConfig | |
| ) -> Tuple[float, Dict[str, float]]: | |
| contrib = { | |
| "sentence_length": cfg.w_sentence_len_excess * f["sentence_len_signed"], | |
| "clause_density": cfg.w_clause_rate * f["clauses_per_sentence"], | |
| "rare_words": cfg.w_rare_rate * f["rare_rate"], | |
| "abstract_terms": cfg.w_abstract_rate * f["abstract_rate"], | |
| "numerals": cfg.w_digit_rate * f["digit_rate"], | |
| "negation": cfg.w_negation_rate * f["negation_rate"], | |
| "hedging": cfg.w_hedge_rate * f["hedge_rate"], | |
| "conditionals": cfg.w_conditional_rate * f["conditional_rate"], | |
| "passive_voice": cfg.w_passive_rate * f["passive_rate"], | |
| "list_structure": cfg.w_list_bonus * f["list_rate"], | |
| } | |
| log_slowdown = sum(contrib.values()) | |
| log_slowdown = max( | |
| -cfg.max_log_slowdown, min(cfg.max_log_slowdown, log_slowdown) | |
| ) | |
| return math.exp(log_slowdown), contrib | |
| # ============================================================================ | |
| # Memory (hv-forget) | |
| # ============================================================================ | |
| def _memory_stability( | |
| sentence: str, density: float, cfg: HVReaderConfig | |
| ) -> Tuple[float, float]: | |
| words = _words(sentence) | |
| n_content = max(1, len(_content_words(sentence))) | |
| n_rare = sum(1 for w in words if _is_rare(w)) | |
| rare_ratio = n_rare / n_content | |
| n_digits = len(_NUMBER_RE.findall(sentence)) | |
| n_proper = sum( | |
| 1 for i, w in enumerate(words) | |
| if i > 0 and w[0].isupper() | |
| and w.lower() not in COMMON_WORDS | |
| and w.lower() not in DISCOURSE_MARKERS | |
| and len(w) >= 3 | |
| ) | |
| salience = min(1.0, (n_digits + n_proper) / n_content) | |
| stability = ( | |
| cfg.memory_stability_base | |
| + cfg.memory_stability_density * density | |
| + cfg.memory_stability_rare * rare_ratio | |
| + cfg.memory_stability_salience * salience | |
| ) | |
| return stability, salience | |
| def _memory_retention(stability: float, days: float) -> float: | |
| if stability <= 0: | |
| return 0.0 | |
| return math.exp(-days / stability) | |
| # ============================================================================ | |
| # Density | |
| # ============================================================================ | |
| def _density_per_sentence(sentences: List[str]) -> List[float]: | |
| seen: Set[str] = set() | |
| out: List[float] = [] | |
| for i, s in enumerate(sentences): | |
| content = [_stem(w) for w in _content_words(s)] | |
| if not content: | |
| out.append(0.0) | |
| continue | |
| if i == 0: | |
| out.append(1.0) | |
| else: | |
| novel = [w for w in content if w not in seen] | |
| out.append(len(novel) / len(content)) | |
| seen.update(content) | |
| return out | |
| # ============================================================================ | |
| # Fold (hv-fold) | |
| # ============================================================================ | |
| def _definite_nps(sentence: str) -> List[Tuple[str, str]]: | |
| out: List[Tuple[str, str]] = [] | |
| words = list(_WORD_RE.finditer(sentence)) | |
| for idx, w in enumerate(words): | |
| if w.group(0).lower() not in ARTICLES: | |
| continue | |
| tail = words[idx + 1: idx + 3] | |
| content = [ | |
| tw.group(0).lower() for tw in tail | |
| if tw.group(0).lower() not in COMMON_WORDS | |
| and len(tw.group(0)) >= 3 | |
| ] | |
| if not content: | |
| continue | |
| if any(c in UNIQUE_REFERENTS for c in content): | |
| continue | |
| head = content[-1] | |
| end = tail[-1].end() if tail else w.end() | |
| out.append((sentence[w.start():end], head)) | |
| return out | |
| def _fold_loads(sentences: List[str], cfg: HVReaderConfig) -> List[float]: | |
| n = len(sentences) | |
| word_sentences: Dict[str, Set[int]] = defaultdict(set) | |
| for i, s in enumerate(sentences): | |
| for w in _words(s): | |
| lw = w.lower() | |
| if lw not in COMMON_WORDS and len(lw) >= 3: | |
| word_sentences[_stem(lw)].add(i) | |
| loads: List[float] = [] | |
| for i, s in enumerate(sentences): | |
| load = 0.0 | |
| if _STANDALONE_DEMON_RE.match(s.strip()): | |
| load += cfg.fold_demon_weight | |
| for _, head in _definite_nps(s): | |
| stem = _stem(head) | |
| occ = word_sentences.get(stem, set()) | |
| prior = [j for j in occ if j < i] | |
| later = [j for j in occ if j > i] | |
| if prior: | |
| load += cfg.fold_resolved | |
| elif later: | |
| delay = min(later) - i | |
| load += cfg.fold_forward * delay | |
| else: | |
| if i == 0 and cfg.exempt_first_sentence: | |
| pass | |
| else: | |
| load += cfg.fold_definite_only | |
| if any(w.lower() in REENTRY_MARKERS for w in _words(s)): | |
| load += cfg.fold_reentry | |
| loads.append(load) | |
| return loads | |
| # ============================================================================ | |
| # Slip (hv-slip) | |
| # ============================================================================ | |
| def _slip_per_sentence( | |
| sentences: List[str], cfg: HVReaderConfig | |
| ) -> List[float]: | |
| n = len(sentences) | |
| if n == 0: | |
| return [] | |
| lengths = [len(_words(s)) for s in sentences] | |
| slips: List[float] = [] | |
| prev_content: Set[str] = set() | |
| for i, s in enumerate(sentences): | |
| lo = max(0, i - cfg.monotone_window) | |
| hi = min(n, i + cfg.monotone_window + 1) | |
| window = lengths[lo:hi] | |
| if len(window) > 1: | |
| mean = sum(window) / len(window) | |
| var = sum((x - mean) ** 2 for x in window) / len(window) | |
| std = math.sqrt(var) | |
| monotone = max(0.0, 1.0 - std / 8.0) | |
| else: | |
| monotone = 0.0 | |
| content = {_stem(w) for w in _content_words(s)} | |
| if prev_content and content: | |
| overlap = len(content & prev_content) / max(1, len(content)) | |
| else: | |
| overlap = 0.0 | |
| prev_content = content | |
| n_digits = len(_NUMBER_RE.findall(s)) | |
| words_s = _words(s) | |
| n_proper = sum( | |
| 1 for j, w in enumerate(words_s) | |
| if j > 0 and w[0].isupper() | |
| and w.lower() not in COMMON_WORDS | |
| and len(w) >= 3 | |
| ) | |
| absence = 1.0 if (n_digits + n_proper) == 0 else 0.0 | |
| slip = ( | |
| cfg.slip_monotone_weight * monotone | |
| + cfg.slip_repetition_weight * overlap | |
| + cfg.slip_absence_weight * absence | |
| ) | |
| slips.append(max(0.0, min(1.0, slip))) | |
| return slips | |
| # ============================================================================ | |
| # Hunger (hv-hunger) | |
| # ============================================================================ | |
| def _hunger_deltas(sentences: List[str], cfg: HVReaderConfig) -> List[float]: | |
| deltas: List[float] = [] | |
| for s in sentences: | |
| low = s.lower() | |
| raised = sum(1 for m in QUESTION_RAISERS if m in low) | |
| answered = sum(1 for m in ANSWER_MARKERS if m in low) | |
| delta = ( | |
| cfg.hunger_question_weight * raised | |
| - cfg.hunger_answer_weight * answered | |
| ) | |
| deltas.append(delta) | |
| return deltas | |
| # ============================================================================ | |
| # Wall (hv-wall) | |
| # ============================================================================ | |
| def _direction_of(word: str) -> List[str]: | |
| w = word.lower() | |
| sw = _stem(w) | |
| out = [] | |
| for g, words in DIRECTION_GROUPS.items(): | |
| if w in words or sw in words: | |
| out.append(g) | |
| return out | |
| def _closest_topic(words: List[str], idx: int) -> Optional[int]: | |
| best = None | |
| best_key: Tuple[int, int] = (10, 1) | |
| for j in range(max(0, idx - 3), min(len(words), idx + 4)): | |
| if j == idx: | |
| continue | |
| cand = words[j] | |
| if cand in COMMON_WORDS or len(cand) < 4: | |
| continue | |
| if _direction_of(cand): | |
| continue | |
| if cand in DISCOURSE_MARKERS: | |
| continue | |
| after = 0 if j > idx else 1 | |
| key = (abs(j - idx), after) | |
| if key < best_key: | |
| best_key = key | |
| best = j | |
| return best | |
| def _direction_pairs(sentence: str) -> List[Tuple[str, str, str]]: | |
| words = [w.lower() for w in _words(sentence)] | |
| out: List[Tuple[str, str, str]] = [] | |
| for i, w in enumerate(words): | |
| groups = _direction_of(w) | |
| if not groups: | |
| continue | |
| j = _closest_topic(words, i) | |
| if j is None: | |
| continue | |
| topic = _stem(words[j]) | |
| for g in groups: | |
| out.append((topic, g, w)) | |
| return out | |
| def _wall_confidence( | |
| sentence: str, cfg: HVReaderConfig | |
| ) -> Tuple[float, int, int]: | |
| words = [w.lower() for w in _words(sentence)] | |
| n = max(1, len(words)) | |
| cert = sum(1 for w in words if w in CERTAINTY_MARKERS) | |
| hedg = sum(1 for w in words if w in HEDGES) | |
| conf = min(1.0, cfg.certainty_scale * cert / n) | |
| hedge = min(1.0, cfg.hedge_scale * hedg / n) | |
| return max(0.0, conf - 0.5 * hedge), cert, hedg | |
| def _wall_evidence( | |
| sentence: str, cfg: HVReaderConfig | |
| ) -> Tuple[float, float, bool]: | |
| words = _words(sentence) | |
| content = _content_words(sentence) | |
| n_content = max(1, len(content)) | |
| n_digits = len(_NUMBER_RE.findall(sentence)) | |
| n_proper = sum( | |
| 1 for i, w in enumerate(words) | |
| if i > 0 and w[0].isupper() | |
| and w.lower() not in COMMON_WORDS | |
| and w.lower() not in DISCOURSE_MARKERS | |
| and len(w) >= 3 | |
| ) | |
| n_rare = sum(1 for w in content if len(w) >= 8) | |
| spec = ( | |
| 0.5 * (n_digits / n_content) | |
| + 0.3 * (n_proper / n_content) | |
| + 0.2 * (n_rare / n_content) | |
| ) | |
| spec = min(1.0, cfg.specificity_scale * spec) | |
| attribution = _has_evidence_marker(sentence) | |
| evidence = 0.6 * spec + 0.4 * (1.0 if attribution else 0.0) | |
| return evidence, spec, attribution | |
| def _wall_per_sentence( | |
| sentences: List[str], cfg: HVReaderConfig | |
| ) -> List[Tuple[float, str, List[str]]]: | |
| n = len(sentences) | |
| out: List[Tuple[float, str, List[str]]] = [] | |
| prior_sentences: List[str] = [] | |
| prior_conf: List[float] = [] | |
| prior_evid: List[float] = [] | |
| for i, s in enumerate(sentences): | |
| confidence, cert_count, _ = _wall_confidence(s, cfg) | |
| evidence, _spec, attribution = _wall_evidence(s, cfg) | |
| n_words = len(_words(s)) | |
| overclaim = ( | |
| max(0.0, confidence - evidence) | |
| if n_words >= cfg.min_sentence_words | |
| else 0.0 | |
| ) | |
| # Contradiction via direction-group conflicts. | |
| contradiction = 0.0 | |
| contra_idx: Optional[int] = None | |
| pairs: List[Tuple[str, str]] = [] | |
| if prior_sentences: | |
| curr_pairs = _direction_pairs(s) | |
| prior_dirs: Dict[str, List[Tuple[int, str, str]]] = {} | |
| for j, p in enumerate(prior_sentences): | |
| for topic, g, src in _direction_pairs(p): | |
| prior_dirs.setdefault(topic, []).append((j, g, src)) | |
| for topic, curr_g, curr_src in curr_pairs: | |
| for prior_idx, prior_g, prior_src in prior_dirs.get(topic, []): | |
| if (prior_g, curr_g) in GROUP_OPPOSITES: | |
| pairs.append(( | |
| f"{topic}:{prior_src}↔{curr_src}", | |
| f"({prior_g} vs {curr_g})", | |
| )) | |
| contra_idx = prior_idx | |
| if pairs: | |
| contradiction = min( | |
| 1.0, cfg.contradiction_scale * math.sqrt(len(pairs)) | |
| ) | |
| # Gap. | |
| gap = 0.0 | |
| has_conclusion = _has_conclusion_marker(s) | |
| if has_conclusion and prior_evid: | |
| prior_ev_mean = sum(prior_evid) / len(prior_evid) | |
| ev_deficit = max(0.0, confidence - prior_ev_mean) | |
| prior_conf_mean = ( | |
| sum(prior_conf) / len(prior_conf) if prior_conf else 0.0 | |
| ) | |
| conf_jump = max(0.0, confidence - prior_conf_mean) | |
| gap = ev_deficit + cfg.gap_confidence_jump_weight * conf_jump | |
| # Priority ordering. | |
| if contradiction >= cfg.contradiction_min: | |
| wall, wtype = contradiction, "contradiction" | |
| elif overclaim >= cfg.overclaim_min: | |
| wall, wtype = overclaim, "overclaim" | |
| elif gap >= cfg.gap_min: | |
| wall, wtype = gap, "gap" | |
| else: | |
| best = max(overclaim, contradiction, gap) | |
| if best <= 0.0: | |
| wall, wtype = 0.0, "neutral" | |
| else: | |
| wall = best | |
| if best == contradiction: | |
| wtype = "contradiction" | |
| elif best == overclaim: | |
| wtype = "overclaim" | |
| else: | |
| wtype = "gap" | |
| wall = min(1.0, wall) | |
| has_contrast, _contrast_word = _has_contrast_marker(s) | |
| if has_contrast and wall > 0.05: | |
| wall = min(1.0, wall + cfg.contrast_bonus) | |
| reasons: List[str] = [] | |
| if wtype == "contradiction" and pairs: | |
| reasons.append(f"conflict with sentence {contra_idx}: {pairs[0][0]}") | |
| elif wtype == "overclaim": | |
| if cert_count: | |
| matched = [ | |
| w.lower() for w in _words(s) | |
| if w.lower() in CERTAINTY_MARKERS | |
| ] | |
| reasons.append( | |
| f"certainty markers: " | |
| f"{', '.join(repr(m) for m in matched[:3])}" | |
| ) | |
| if not attribution: | |
| reasons.append("no attribution marker") | |
| elif wtype == "gap": | |
| if has_conclusion: | |
| reasons.append("conclusion marker with weak prior evidence") | |
| out.append((wall, wtype if wall > 0.0 else "neutral", reasons)) | |
| prior_sentences.append(s) | |
| prior_conf.append(confidence) | |
| prior_evid.append(evidence) | |
| return out | |
| # ============================================================================ | |
| # The model | |
| # ============================================================================ | |
| class HVReader: | |
| """Unified reading-experience model.""" | |
| def __init__(self, config: Optional[HVReaderConfig] = None): | |
| self.config = config or HVReaderConfig() | |
| self._obs = 0 | |
| def __repr__(self) -> str: | |
| return ( | |
| f"HVReader(baseline_wpm={self.config.baseline_wpm}, " | |
| f"wall_threshold={self.config.wall_threshold}, " | |
| f"version={self.config.version})" | |
| ) | |
| def analyze(self, text: str) -> ReadingProfile: | |
| if not text or not text.strip(): | |
| return self._empty(text) | |
| sentences = _sentences(text) | |
| n = len(sentences) | |
| if n == 0: | |
| return self._empty(text) | |
| spans: List[Tuple[int, int]] = [] | |
| cursor = 0 | |
| for s in sentences: | |
| i = text.find(s, cursor) | |
| if i < 0: | |
| i = cursor | |
| spans.append((i, i + len(s))) | |
| cursor = i + len(s) | |
| pace_feats = [_pace_features(s) for s in sentences] | |
| slowdowns: List[float] = [] | |
| wpms: List[float] = [] | |
| for f in pace_feats: | |
| sd, _ = _pace_slowdown(f, self.config) | |
| slowdowns.append(sd) | |
| wpms.append( | |
| self.config.baseline_wpm / sd | |
| if sd > 0 else self.config.baseline_wpm | |
| ) | |
| densities = _density_per_sentence(sentences) | |
| fold_loads = _fold_loads(sentences, self.config) | |
| slips = _slip_per_sentence(sentences, self.config) | |
| hunger_deltas = _hunger_deltas(sentences, self.config) | |
| hunger_cumulative: List[float] = [] | |
| h = 0.0 | |
| for d in hunger_deltas: | |
| h += d | |
| hunger_cumulative.append(h) | |
| stabilities: List[float] = [] | |
| retentions: List[float] = [] | |
| for s, dens in zip(sentences, densities): | |
| stab, _ = _memory_stability(s, dens, self.config) | |
| stabilities.append(stab) | |
| retentions.append( | |
| _memory_retention(stab, self.config.memory_target_days) | |
| ) | |
| wall_per = _wall_per_sentence(sentences, self.config) | |
| span_profiles: List[SpanProfile] = [] | |
| for i in range(n): | |
| wall_score, wall_type, reasons = wall_per[i] | |
| n_words_span = int(pace_feats[i]["n_words"]) | |
| span_profiles.append(SpanProfile( | |
| id=i, | |
| text=sentences[i], | |
| span=spans[i], | |
| n_words=n_words_span, | |
| wpm=wpms[i], | |
| slowdown=slowdowns[i], | |
| density=densities[i], | |
| stability_days=stabilities[i], | |
| retention_week=retentions[i], | |
| fold_load=fold_loads[i], | |
| slip=slips[i], | |
| hunger_delta=hunger_deltas[i], | |
| hunger=hunger_cumulative[i], | |
| wall=wall_score, | |
| wall_type=wall_type, | |
| reasons=reasons[: self.config.top_reasons], | |
| )) | |
| n_words_total = sum(s.n_words for s in span_profiles) | |
| mean_wpm = ( | |
| sum(s.wpm * s.n_words for s in span_profiles) | |
| / max(1, n_words_total) | |
| ) | |
| pace = max(0.0, min(1.0, mean_wpm / 400.0)) | |
| memory = sum(retentions) / n if n else 0.0 | |
| total_load = sum(fold_loads) | |
| passes = 1.0 + total_load / max(1, n) | |
| slip = sum(slips) / n if n else 0.0 | |
| wall_scores = [s.wall for s in span_profiles] | |
| wall_max = max(wall_scores) if wall_scores else 0.0 | |
| wall = wall_max | |
| wps = mean_wpm / 60.0 | |
| ttu_s = n_words_total / wps if wps > 0 else 0.0 | |
| hunger_final = hunger_cumulative[-1] if hunger_cumulative else 0.0 | |
| slowest = ( | |
| min(span_profiles, key=lambda s: s.wpm) | |
| if span_profiles else None | |
| ) | |
| wall_span: Optional[SpanProfile] = None | |
| if wall > 0.15: | |
| wall_span = max(span_profiles, key=lambda s: s.wall) | |
| summary = self._summary( | |
| span_profiles, pace, memory, passes, slip, wall | |
| ) | |
| self._obs += 1 | |
| return ReadingProfile( | |
| text=text, | |
| n_sentences=n, | |
| n_words=n_words_total, | |
| pace=pace, | |
| memory=memory, | |
| passes=passes, | |
| slip=slip, | |
| wall=wall, | |
| ttu_s=ttu_s, | |
| mean_wpm=mean_wpm, | |
| hunger_final=hunger_final, | |
| spans=span_profiles, | |
| slowest_span=slowest, | |
| wall_span=wall_span, | |
| summary=summary, | |
| ) | |
| def _empty(self, text: str) -> ReadingProfile: | |
| return ReadingProfile( | |
| text=text, | |
| n_sentences=0, | |
| n_words=0, | |
| pace=0.0, | |
| memory=0.0, | |
| passes=1.0, | |
| slip=0.0, | |
| wall=0.0, | |
| ttu_s=0.0, | |
| mean_wpm=0.0, | |
| hunger_final=0.0, | |
| spans=[], | |
| slowest_span=None, | |
| wall_span=None, | |
| summary="Empty text.", | |
| ) | |
| def _summary( | |
| spans: List[SpanProfile], | |
| pace: float, | |
| memory: float, | |
| passes: float, | |
| slip: float, | |
| wall: float, | |
| ) -> str: | |
| if not spans: | |
| return "Empty text." | |
| parts = [] | |
| parts.append(f"reads at {pace * 400:.0f} WPM (pace {pace:.2f})") | |
| parts.append(f"memory after a week: {memory * 100:.0f}%") | |
| parts.append(f"requires {passes:.2f} passes") | |
| parts.append(f"slip probability: {slip:.2f}") | |
| if wall > 0.15: | |
| wall_span = max(spans, key=lambda s: s.wall) | |
| parts.append( | |
| f"wall at sentence {wall_span.id} " | |
| f"({wall_span.wall_type}, {wall:.2f})" | |
| ) | |
| else: | |
| parts.append("no wall") | |
| return ". ".join(parts).capitalize() + "." | |
| def render(self, profile: ReadingProfile) -> str: | |
| lines: List[str] = [] | |
| bar = "=" * 72 | |
| lines.append(bar) | |
| lines.append("hv-reader — the reading experience") | |
| lines.append(bar) | |
| lines.append("") | |
| lines.append(f" text : {profile.n_sentences} sentences, " | |
| f"{profile.n_words} words") | |
| lines.append("") | |
| lines.append(" READING PROFILE") | |
| lines.append(" " + "-" * 68) | |
| lines.append(f" pace {profile.pace:>6.3f} " | |
| f"({profile.mean_wpm:.0f} WPM)") | |
| lines.append(f" memory {profile.memory:>6.3f} " | |
| f"(fraction surviving 1 week)") | |
| lines.append(f" passes {profile.passes:>6.3f} " | |
| f"(reads needed)") | |
| lines.append(f" slip {profile.slip:>6.3f} " | |
| f"(attention-lapse probability)") | |
| lines.append(f" wall {profile.wall:>6.3f} " | |
| f"(reader-refusal probability)") | |
| lines.append("") | |
| lines.append(f" ttu : {profile.ttu_s:.1f} s " | |
| f"(total reading time)") | |
| lines.append(f" hunger : {profile.hunger_final:+.2f} " | |
| f"(unresolved questions)") | |
| lines.append("") | |
| if not profile.spans: | |
| lines.append(" (no content)") | |
| return "\n".join(lines) | |
| lines.append(" PER-SENTENCE") | |
| lines.append(" " + "-" * 68) | |
| lines.append( | |
| f" {'id':>3} {'wpm':>5} {'dens':>5} {'ret':>5} " | |
| f"{'fold':>5} {'slip':>5} {'wall':>5} type" | |
| ) | |
| for s in profile.spans: | |
| marker = ( | |
| "*" if (profile.wall_span and s.id == profile.wall_span.id) | |
| else " " | |
| ) | |
| lines.append( | |
| f" {marker}{s.id:>2} {s.wpm:>5.0f} {s.density:>5.2f} " | |
| f"{s.retention_week:>5.2f} {s.fold_load:>5.2f} " | |
| f"{s.slip:>5.2f} {s.wall:>5.2f} {s.wall_type}" | |
| ) | |
| lines.append("") | |
| if profile.slowest_span: | |
| s = profile.slowest_span | |
| lines.append(" SLOWEST SPAN") | |
| lines.append(" " + "-" * 68) | |
| lines.append(f" [{s.id}] {s.wpm:.0f} WPM " | |
| f"(slowdown {s.slowdown:.2f}x)") | |
| lines.append(f" \"{self._shorten(s.text, 60)}\"") | |
| lines.append("") | |
| if profile.wall_span: | |
| s = profile.wall_span | |
| lines.append(" WALL SPAN") | |
| lines.append(" " + "-" * 68) | |
| lines.append(f" [{s.id}] {s.wall_type} " | |
| f"(score {s.wall:.3f})") | |
| lines.append(f" \"{self._shorten(s.text, 60)}\"") | |
| for r in s.reasons: | |
| lines.append(f" - {r}") | |
| lines.append("") | |
| lines.append(" SUMMARY") | |
| lines.append(" " + "-" * 68) | |
| lines.append(f" {profile.summary}") | |
| lines.append("") | |
| return "\n".join(lines) | |
| def _shorten(s: str, n: int) -> str: | |
| s = s.strip().replace("\n", " ") | |
| if len(s) <= n: | |
| return s | |
| return s[: n - 1].rsplit(" ", 1)[0] + "…" | |
| def save_pretrained(self, save_dir: str) -> None: | |
| os.makedirs(save_dir, exist_ok=True) | |
| payload = { | |
| "config": asdict(self.config), | |
| "observations": self._obs, | |
| } | |
| with open(os.path.join(save_dir, "config.json"), "w") as f: | |
| json.dump(payload, f, indent=2) | |
| def from_pretrained(cls, save_dir: str) -> "HVReader": | |
| with open(os.path.join(save_dir, "config.json"), "r") as f: | |
| payload = json.load(f) | |
| cfg_dict = payload.get("config", {}) | |
| known = {f.name for f in HVReaderConfig.__dataclass_fields__.values()} | |
| cfg_dict = {k: v for k, v in cfg_dict.items() if k in known} | |
| cfg = HVReaderConfig(**cfg_dict) | |
| obj = cls(config=cfg) | |
| obj._obs = int(payload.get("observations", 0)) | |
| return obj | |
| # ============================================================================ | |
| # Demo | |
| # ============================================================================ | |
| SAMPLE_FICTION = ( | |
| "The old man walked slowly to the boat. He stopped, looked at the " | |
| "water, and then continued. The sea was quiet that morning. He " | |
| "pushed the boat into the water and climbed in. The oars were cold " | |
| "in his hands. He rowed out past the harbor and into the open sea." | |
| ) | |
| SAMPLE_ACADEMIC = ( | |
| "A black hole is a region of spacetime where gravity is so strong " | |
| "that nothing — no particles or even electromagnetic radiation such " | |
| "as light — can escape from it. The theory of general relativity " | |
| "predicts that a sufficiently compact mass can deform spacetime to " | |
| "form a black hole. The boundary of the region from which no escape " | |
| "is possible is called the event horizon. Although the event horizon " | |
| "has profound effects on the fate of an object that crosses it, it " | |
| "has no locally detectable features. A black hole acts as a perfect " | |
| "black body, and moreover, it emits Hawking radiation." | |
| ) | |
| SAMPLE_OVERCLAIM = ( | |
| "The data suggests some correlation between the policy and the outcome. " | |
| "Results appear to indicate a modest effect in some subpopulations. " | |
| "The mechanism remains unclear, and further work is needed to establish " | |
| "causality. " | |
| "Therefore, the policy definitively causes the outcome in all cases, " | |
| "and this is unquestionably proven by the evidence." | |
| ) | |
| SAMPLE_MONOTONE = ( | |
| "The system processes the input. The system processes the data. " | |
| "The system processes the output. The system processes the result. " | |
| "The system processes the value. The system processes the record." | |
| ) | |
| def _demo(output_dir: str = "./hv_reader_output") -> None: | |
| os.makedirs(output_dir, exist_ok=True) | |
| m = HVReader() | |
| samples = [ | |
| ("Fiction", SAMPLE_FICTION), | |
| ("Academic", SAMPLE_ACADEMIC), | |
| ("Overclaim", SAMPLE_OVERCLAIM), | |
| ("Monotone", SAMPLE_MONOTONE), | |
| ] | |
| for name, text in samples: | |
| print() | |
| print("#" * 72) | |
| print(f"# {name}") | |
| print("#" * 72) | |
| print(m.render(m.analyze(text))) | |
| print() | |
| print("=" * 72) | |
| print("Summary across samples") | |
| print("=" * 72) | |
| print( | |
| f" {'sample':<12} {'words':>6} {'pace':>6} {'mem':>6} " | |
| f"{'pass':>5} {'slip':>6} {'wall':>5} {'ttu':>6}" | |
| ) | |
| print(" " + "-" * 68) | |
| for name, text in samples: | |
| r = m.analyze(text) | |
| print( | |
| f" {name:<12} {r.n_words:>6} {r.pace:>6.3f} " | |
| f"{r.memory:>6.3f} {r.passes:>5.2f} " | |
| f"{r.slip:>6.3f} {r.wall:>5.2f} {r.ttu_s:>5.1f}s" | |
| ) | |
| print() | |
| print("=" * 72) | |
| print("Save / load round trip") | |
| print("=" * 72) | |
| save_path = os.path.join(output_dir, "hv_reader_model") | |
| m.save_pretrained(save_path) | |
| m2 = HVReader.from_pretrained(save_path) | |
| print(f" saved to : {save_path}") | |
| print(f" reloaded : {m2!r}") | |
| a = m.analyze(SAMPLE_OVERCLAIM) | |
| b = m2.analyze(SAMPLE_OVERCLAIM) | |
| print(f" pace : {a.pace:.4f}") | |
| print(f" wall : {a.wall:.4f}") | |
| print(f" identical : " | |
| f"{abs(a.pace - b.pace) < 1e-9 and abs(a.wall - b.wall) < 1e-9}") | |
| print() | |
| # ============================================================================ | |
| # CLI | |
| # ============================================================================ | |
| def _cli() -> None: | |
| p = argparse.ArgumentParser( | |
| description="hv-reader: the reading experience in one call." | |
| ) | |
| p.add_argument("--text", type=str, default="", | |
| help="text to analyze (or '-' to read stdin)") | |
| p.add_argument("--json", action="store_true", | |
| help="output JSON instead of a rendered report") | |
| p.add_argument("--profile", action="store_true", | |
| help="print only the five-axis profile") | |
| p.add_argument("--save-to", type=str, default="", | |
| help="save the model to this directory") | |
| p.add_argument("--outdir", type=str, default="./hv_reader_output") | |
| args = p.parse_args() | |
| text = sys.stdin.read() if args.text == "-" else args.text | |
| m = HVReader() | |
| if args.save_to: | |
| m.save_pretrained(args.save_to) | |
| print(f"saved to {args.save_to}", file=sys.stderr) | |
| if not text: | |
| _demo(args.outdir) | |
| return | |
| profile = m.analyze(text) | |
| if args.profile: | |
| print(f"pace {profile.pace:.3f}") | |
| print(f"memory {profile.memory:.3f}") | |
| print(f"passes {profile.passes:.3f}") | |
| print(f"slip {profile.slip:.3f}") | |
| print(f"wall {profile.wall:.3f}") | |
| return | |
| if args.json: | |
| print(json.dumps(profile.to_dict(), indent=2, ensure_ascii=False)) | |
| else: | |
| print(m.render(profile)) | |
| if __name__ == "__main__": | |
| _cli() |