| """analysis/textdiff.py — verbatim text delta signals for the Analyst Edge layer. |
| |
| Pure Python + sentence-transformers, zero LLM calls. |
| Compares the most recent filing period against the prior period for a ticker |
| and surfaces verbatim before→after fragments for the most material changes: |
| |
| 1. risk_reworded / risk_added / risk_removed — risk-factor diffs |
| 2. term_frequency — analyst-lexicon count deltas |
| 3. guidance_language_shift — hedge/modal word shifts in MD&A |
| 4. kpi_dropped — metric mentioned prior, absent now |
| 5. compute_lexicon_trend — multi-quarter lexicon term trend (n-quarter monotone run) |
| |
| Usage: |
| from analysis.textdiff import compute |
| signals = compute("NVDA") |
| """ |
| from __future__ import annotations |
|
|
| import re |
| from typing import Optional |
|
|
| import numpy as np |
|
|
| from analysis.signals import QuarterDelta |
| from storage.sections_db import get_section, get_periods_for_ticker |
|
|
| |
| |
| |
|
|
| _REWORD_THRESHOLD = 0.70 |
| _NEW_RISK_THRESHOLD = 0.40 |
| _IDENTICAL_THRESHOLD = 0.93 |
|
|
| _MIN_ITEM_WORDS = 25 |
|
|
| |
| _LEXICON: list[tuple[str, str]] = [ |
| |
| (r"\btariff\b", "tariff"), |
| (r"\bexport control\b", "export control"), |
| (r"\bheadwind\b", "headwind"), |
| (r"\buncertainty\b", "uncertainty"), |
| (r"\bsoftness\b", "softness"), |
| (r"\bslowing\b", "slowing"), |
| (r"\bdecelerat\w*", "deceleration"), |
| (r"\bcautious\b", "cautious"), |
| (r"\bpressure\b", "pressure"), |
| (r"\bai\b", "AI"), |
| (r"\bbuyback\b", "buyback"), |
| (r"\blayoff\b", "layoff"), |
| (r"\brestructur\w*", "restructuring"), |
| (r"\bimpairment\b", "impairment"), |
| (r"\blitigation\b", "litigation"), |
| (r"\bchinese? market\b", "China market"), |
| (r"\bsanction\b", "sanction"), |
| (r"\brecession\b", "recession"), |
| ] |
|
|
| |
| _FREQ_RATIO_THRESHOLD = 2.0 |
| _FREQ_ABS_THRESHOLD = 2 |
|
|
| |
| _KPI_PATTERNS: list[tuple[str, str]] = [ |
| (r"\b(?:gross\s+)?margins?\b", "gross margin"), |
| (r"\b(?:operating\s+)?margins?\b", "operating margin"), |
| (r"\bfree\s+cash\s+flow\b", "free cash flow"), |
| (r"\bdays?\s+sales?\s+outstanding\b|\bdso\b", "DSO"), |
| (r"\bdays?\s+inventory\s+outstanding\b|\bdio\b", "DIO"), |
| (r"\bdays?\s+payable\s+outstanding\b|\bdpo\b", "DPO"), |
| (r"\bshare\s+(?:repurchase|buyback)\b", "share repurchase"), |
| (r"\bdividend\b", "dividend"), |
| (r"\bguidance\b", "guidance"), |
| (r"\bbacklog\b", "backlog"), |
| (r"\bdeferred\s+revenue\b", "deferred revenue"), |
| (r"\bnet\s+retention\s+rate\b", "net retention rate"), |
| ] |
|
|
| |
| _HEDGE_WORDS = [ |
| "expect to grow", "expect growth", "expects to grow", "expects growth", |
| "anticipate", "plan to", "target", "forecast", |
| "moderate", "soften", "decline", "reduce", "headwind", "challenge", |
| "cautious", "uncertain", "volatile", |
| ] |
|
|
| |
| |
| |
|
|
| _encoder = None |
|
|
|
|
| def _get_encoder(): |
| global _encoder |
| if _encoder is None: |
| from sentence_transformers import SentenceTransformer |
| _encoder = SentenceTransformer("all-MiniLM-L6-v2", device="cpu") |
| return _encoder |
|
|
|
|
| def _embed(texts: list[str]) -> np.ndarray: |
| enc = _get_encoder() |
| vecs = enc.encode(texts, convert_to_numpy=True, show_progress_bar=False) |
| |
| norms = np.linalg.norm(vecs, axis=1, keepdims=True) |
| norms = np.where(norms < 1e-8, 1.0, norms) |
| return vecs / norms |
|
|
|
|
| def _cosine(a: np.ndarray, b: np.ndarray) -> float: |
| return float(np.dot(a, b)) |
|
|
|
|
| |
| |
| |
|
|
| def _split_into_items(text: str, min_words: int = _MIN_ITEM_WORDS) -> list[str]: |
| """Split a section text into logical chunks (risk items / paragraphs). |
| |
| Uses double-newline paragraph boundaries. Merges short lines (headers) |
| with the following paragraph. Returns only chunks >= min_words. |
| """ |
| raw = re.split(r"\n{2,}", text.strip()) |
| items: list[str] = [] |
| buffer = "" |
| for para in raw: |
| para = para.strip() |
| if not para: |
| continue |
| word_count = len(para.split()) |
| if word_count < 8: |
| |
| buffer = para + " " |
| else: |
| combined = (buffer + para).strip() |
| buffer = "" |
| if len(combined.split()) >= min_words: |
| items.append(combined) |
| if buffer.strip() and len(buffer.split()) >= min_words: |
| items.append(buffer.strip()) |
| return items |
|
|
|
|
| def _split_sentences(text: str) -> list[str]: |
| """Simple sentence splitter (no NLTK dependency).""" |
| sentences = re.split(r"(?<=[.!?])\s+", text) |
| return [s.strip() for s in sentences if len(s.split()) >= 5] |
|
|
|
|
| |
| |
| |
|
|
| def _align_items( |
| current_items: list[str], |
| prior_items: list[str], |
| current_vecs: np.ndarray, |
| prior_vecs: np.ndarray, |
| ) -> tuple[dict[int, int], dict[int, float]]: |
| """Greedy one-to-one alignment: each current item → best prior item. |
| |
| Returns: |
| matches: {current_idx: prior_idx} |
| scores: {current_idx: cosine_similarity} |
| """ |
| if len(current_items) == 0 or len(prior_items) == 0: |
| return {}, {} |
|
|
| |
| sim_matrix = current_vecs @ prior_vecs.T |
|
|
| matches: dict[int, int] = {} |
| scores: dict[int, float] = {} |
| used_prior: set[int] = set() |
|
|
| |
| for ci in range(len(current_items)): |
| row = sim_matrix[ci] |
| |
| masked = [(row[pi], pi) for pi in range(len(prior_items)) if pi not in used_prior] |
| if not masked: |
| break |
| best_score, best_pi = max(masked) |
| matches[ci] = best_pi |
| scores[ci] = best_score |
| if best_score >= _NEW_RISK_THRESHOLD: |
| used_prior.add(best_pi) |
|
|
| return matches, scores |
|
|
|
|
| |
| |
| |
|
|
| def compute_risk_deltas( |
| current_text: str, |
| prior_text: str, |
| period_from: str, |
| period_to: str, |
| form_type: str, |
| ) -> list[QuarterDelta]: |
| """Align risk-factor items across two periods and classify changes.""" |
| if not current_text or not prior_text: |
| return [] |
|
|
| current_items = _split_into_items(current_text) |
| prior_items = _split_into_items(prior_text) |
| if not current_items or not prior_items: |
| return [] |
|
|
| current_vecs = _embed(current_items) |
| prior_vecs = _embed(prior_items) |
|
|
| matches, scores = _align_items(current_items, prior_items, current_vecs, prior_vecs) |
|
|
| matched_prior_indices: set[int] = set() |
| deltas: list[QuarterDelta] = [] |
| source_lit = "10-K" if "10-K" in form_type.upper() else "10-Q" |
|
|
| for ci, item in enumerate(current_items): |
| pi = matches.get(ci) |
| score = scores.get(ci, 0.0) |
|
|
| if pi is not None and score >= _NEW_RISK_THRESHOLD: |
| matched_prior_indices.add(pi) |
| if score >= _IDENTICAL_THRESHOLD: |
| continue |
|
|
| |
| before = _truncate(prior_items[pi], 120) |
| after = _truncate(item, 120) |
| sig = "HIGH" if score < 0.80 else "MEDIUM" |
| deltas.append(QuarterDelta( |
| kind="risk_reworded", |
| period_from=period_from, |
| period_to=period_to, |
| before_text=before, |
| after_text=after, |
| computed_metric=f"similarity {score:.2f}", |
| source=source_lit, |
| significance=sig, |
| term="", |
| )) |
| else: |
| |
| after = _truncate(item, 120) |
| deltas.append(QuarterDelta( |
| kind="risk_added", |
| period_from=period_from, |
| period_to=period_to, |
| before_text="", |
| after_text=after, |
| computed_metric="", |
| source=source_lit, |
| significance="HIGH", |
| term="", |
| )) |
|
|
| |
| for pi, item in enumerate(prior_items): |
| if pi not in matched_prior_indices: |
| before = _truncate(item, 120) |
| deltas.append(QuarterDelta( |
| kind="risk_removed", |
| period_from=period_from, |
| period_to=period_to, |
| before_text=before, |
| after_text="", |
| computed_metric="", |
| source=source_lit, |
| significance="MEDIUM", |
| term="", |
| )) |
|
|
| |
| order = {"HIGH": 0, "MEDIUM": 1, "LOW": 2} |
| deltas.sort(key=lambda d: (order[d.significance], d.kind)) |
| return deltas[:6] |
|
|
|
|
| |
| |
| |
|
|
| def compute_lexicon_deltas( |
| current_text: str, |
| prior_text: str, |
| period_from: str, |
| period_to: str, |
| form_type: str, |
| ) -> list[QuarterDelta]: |
| """Count analyst-lexicon term occurrences and flag large swings.""" |
| if not current_text or not prior_text: |
| return [] |
|
|
| source_lit = "10-K" if "10-K" in form_type.upper() else "10-Q" |
| cur_lower = current_text.lower() |
| pri_lower = prior_text.lower() |
|
|
| deltas: list[QuarterDelta] = [] |
|
|
| for pattern, label in _LEXICON: |
| cur_count = len(re.findall(pattern, cur_lower, re.IGNORECASE)) |
| pri_count = len(re.findall(pattern, pri_lower, re.IGNORECASE)) |
|
|
| if cur_count == 0 and pri_count == 0: |
| continue |
|
|
| abs_diff = abs(cur_count - pri_count) |
| if abs_diff < _FREQ_ABS_THRESHOLD: |
| continue |
|
|
| |
| max_count = max(cur_count, pri_count) |
| min_count = min(cur_count, pri_count) or 0.5 |
| ratio = max_count / min_count |
| if ratio < _FREQ_RATIO_THRESHOLD: |
| continue |
|
|
| direction = "up" if cur_count > pri_count else "down" |
| pct = (cur_count - pri_count) / (pri_count or 1) * 100 |
| metric = f"{pri_count}→{cur_count} occurrences ({pct:+.0f}%)" |
|
|
| |
| sig = "HIGH" if (ratio >= 3.0 or abs_diff >= 5) else "MEDIUM" |
|
|
| |
| after_ctx = _find_context_sentence(current_text, pattern) if cur_count > 0 else "" |
| before_ctx = _find_context_sentence(prior_text, pattern) if pri_count > 0 else "" |
|
|
| deltas.append(QuarterDelta( |
| kind="term_frequency", |
| period_from=period_from, |
| period_to=period_to, |
| before_text=before_ctx, |
| after_text=after_ctx, |
| computed_metric=metric, |
| source=source_lit, |
| significance=sig, |
| term=label, |
| )) |
|
|
| deltas.sort(key=lambda d: {"HIGH": 0, "MEDIUM": 1}.get(d.significance, 2)) |
| return deltas[:5] |
|
|
|
|
| def _find_context_sentence(text: str, pattern: str) -> str: |
| """Return the first sentence containing a match for `pattern`.""" |
| sentences = _split_sentences(text) |
| for sent in sentences: |
| if re.search(pattern, sent, re.IGNORECASE): |
| return _truncate(sent, 100) |
| return "" |
|
|
|
|
| |
| |
| |
|
|
| def _detect_trend(counts: list[int]) -> str | None: |
| """Detect the longest strictly monotone run at the tail of a count series. |
| |
| Args: |
| counts: term occurrence counts in **chronological order** (oldest first). |
| |
| Returns: |
| ``"rising N quarters"`` if the last N≥3 values are strictly increasing, |
| ``"falling N quarters"`` if the last N≥3 values are strictly decreasing, |
| ``None`` otherwise. |
| |
| Examples: |
| >>> _detect_trend([1, 3, 5, 8]) |
| 'rising 4 quarters' |
| >>> _detect_trend([8, 5, 3, 1]) |
| 'falling 4 quarters' |
| >>> _detect_trend([1, 5, 2, 4, 6]) |
| 'rising 3 quarters' |
| >>> _detect_trend([1, 2, 2, 4]) # plateau breaks strict run |
| >>> _detect_trend([1, 3]) # only 2 values |
| """ |
| if len(counts) < 3: |
| return None |
|
|
| |
| |
| n = len(counts) |
|
|
| |
| if counts[-1] > counts[-2]: |
| direction = "rising" |
| elif counts[-1] < counts[-2]: |
| direction = "falling" |
| else: |
| return None |
|
|
| |
| run_length = 2 |
| for i in range(n - 2, 0, -1): |
| if direction == "rising" and counts[i] > counts[i - 1]: |
| run_length += 1 |
| elif direction == "falling" and counts[i] < counts[i - 1]: |
| run_length += 1 |
| else: |
| break |
|
|
| if run_length < 3: |
| return None |
|
|
| return f"{direction} {run_length} quarters" |
|
|
|
|
| def compute_lexicon_trend(ticker: str, n: int = 4) -> list[QuarterDelta]: |
| """Detect multi-quarter monotone trends for each analyst-lexicon term. |
| |
| Looks back up to *n* 10-Q periods and surfaces terms whose occurrence |
| count has been strictly rising or falling for 3+ consecutive quarters — |
| a more durable signal than a single quarter-over-quarter spike. |
| |
| Args: |
| ticker: uppercase ticker symbol. |
| n: maximum number of recent 10-Q periods to examine (default 4). |
| |
| Returns: |
| Up to 5 ``QuarterDelta`` objects (HIGH-significance first), one per |
| term that shows a multi-quarter trend. Returns ``[]`` if fewer than |
| 3 periods are available or no trends are detected. |
| """ |
| ticker = ticker.upper() |
| periods = get_periods_for_ticker(ticker, form_type="10-Q") |
| if len(periods) < 3: |
| return [] |
|
|
| n = min(n, len(periods)) |
| |
| selected = list(reversed(periods[:n])) |
|
|
| |
| period_texts: list[str] = [] |
| for period in selected: |
| mda = get_section(ticker, period, "mda") or "" |
| risk = get_section(ticker, period, "risk_factors") or "" |
| period_texts.append((mda + "\n\n" + risk).strip()) |
|
|
| deltas: list[QuarterDelta] = [] |
|
|
| for pattern, label in _LEXICON: |
| counts = [ |
| len(re.findall(pattern, text, re.IGNORECASE)) |
| for text in period_texts |
| ] |
|
|
| trend = _detect_trend(counts) |
| if trend is None: |
| continue |
|
|
| |
| oldest_period = selected[0] |
| newest_period = selected[-1] |
|
|
| first_count = counts[0] |
| last_count = counts[-1] |
| metric = ( |
| f"{first_count}→{last_count} occurrences over " |
| f"{oldest_period}→{newest_period} ({trend})" |
| ) |
|
|
| oldest_text = period_texts[0] |
| newest_text = period_texts[-1] |
| before_ctx = _find_context_sentence(oldest_text, pattern) if first_count > 0 else "" |
| after_ctx = _find_context_sentence(newest_text, pattern) if last_count > 0 else "" |
|
|
| |
| run_quarters = int(trend.split()[1]) |
| sig = "HIGH" if run_quarters >= 4 else "MEDIUM" |
|
|
| deltas.append(QuarterDelta( |
| kind="term_frequency", |
| period_from=oldest_period, |
| period_to=newest_period, |
| before_text=before_ctx, |
| after_text=after_ctx, |
| computed_metric=metric, |
| source="10-Q", |
| significance=sig, |
| term=label, |
| )) |
|
|
| |
| deltas.sort(key=lambda d: {"HIGH": 0, "MEDIUM": 1}.get(d.significance, 2)) |
| return deltas[:5] |
|
|
|
|
| |
| |
| |
|
|
| def compute_guidance_shifts( |
| current_mda: str, |
| prior_mda: str, |
| period_from: str, |
| period_to: str, |
| form_type: str, |
| ) -> list[QuarterDelta]: |
| """Detect forward-looking language becoming more cautious or more bullish.""" |
| if not current_mda or not prior_mda: |
| return [] |
|
|
| source_lit = "10-K" if "10-K" in form_type.upper() else "10-Q" |
|
|
| |
| cur_fwd = _forward_looking_sentences(current_mda) |
| pri_fwd = _forward_looking_sentences(prior_mda) |
|
|
| if not cur_fwd or not pri_fwd: |
| return [] |
|
|
| |
| cur_hedge = _count_hedge(cur_fwd) |
| pri_hedge = _count_hedge(pri_fwd) |
|
|
| abs_diff = abs(cur_hedge - pri_hedge) |
| if abs_diff < 2: |
| return [] |
|
|
| direction = "more cautious" if cur_hedge > pri_hedge else "more confident" |
| pct = (cur_hedge - pri_hedge) / (pri_hedge or 1) * 100 |
| metric = f"{pri_hedge}→{cur_hedge} hedge-word occurrences ({pct:+.0f}%) → {direction}" |
|
|
| |
| before_sent = _pick_representative(pri_fwd, prior_mda) |
| after_sent = _pick_representative(cur_fwd, current_mda) |
|
|
| sig = "HIGH" if abs_diff >= 5 else "MEDIUM" |
|
|
| return [QuarterDelta( |
| kind="guidance_language_shift", |
| period_from=period_from, |
| period_to=period_to, |
| before_text=before_sent, |
| after_text=after_sent, |
| computed_metric=metric, |
| source=source_lit, |
| significance=sig, |
| term="guidance tone", |
| )] |
|
|
|
|
| _FWD_PATTERNS = re.compile( |
| r"\b(expect|anticipate|forecast|guidance|outlook|project|target|plan\s+to|" |
| r"will\s+(?:grow|increase|decrease|decline|moderate)|believe\s+(?:we|our))\b", |
| re.IGNORECASE, |
| ) |
|
|
|
|
| def _forward_looking_sentences(text: str) -> list[str]: |
| sentences = _split_sentences(text) |
| return [s for s in sentences if _FWD_PATTERNS.search(s)] |
|
|
|
|
| def _count_hedge(sentences: list[str]) -> int: |
| joined = " ".join(sentences).lower() |
| return sum(1 for w in _HEDGE_WORDS if w in joined) |
|
|
|
|
| def _pick_representative(sentences: list[str], full_text: str) -> str: |
| """Return the shortest guidance sentence (most quotable) that contains a hedge word.""" |
| hedge_sents = [ |
| s for s in sentences |
| if any(h in s.lower() for h in _HEDGE_WORDS) |
| ] |
| pool = hedge_sents if hedge_sents else sentences |
| pool_sorted = sorted(pool, key=lambda s: len(s.split())) |
| if pool_sorted: |
| return _truncate(pool_sorted[0], 100) |
| return _truncate(sentences[0], 100) if sentences else "" |
|
|
|
|
| |
| |
| |
|
|
| def compute_kpi_drops( |
| current_mda: str, |
| prior_mda: str, |
| period_from: str, |
| period_to: str, |
| form_type: str, |
| ) -> list[QuarterDelta]: |
| """Flag a KPI / metric label that appears in prior MD&A but not in current.""" |
| if not current_mda or not prior_mda: |
| return [] |
|
|
| source_lit = "10-K" if "10-K" in form_type.upper() else "10-Q" |
| cur_lower = current_mda.lower() |
| pri_lower = prior_mda.lower() |
|
|
| deltas: list[QuarterDelta] = [] |
| for pattern, label in _KPI_PATTERNS: |
| in_current = bool(re.search(pattern, cur_lower, re.IGNORECASE)) |
| in_prior = bool(re.search(pattern, pri_lower, re.IGNORECASE)) |
|
|
| if in_prior and not in_current: |
| ctx = _find_context_sentence(prior_mda, pattern) |
| deltas.append(QuarterDelta( |
| kind="kpi_dropped", |
| period_from=period_from, |
| period_to=period_to, |
| before_text=ctx, |
| after_text="", |
| computed_metric=f"'{label}' mentioned in {period_from} MD&A, absent from {period_to}", |
| source=source_lit, |
| significance="MEDIUM", |
| term=label, |
| )) |
|
|
| return deltas[:3] |
|
|
|
|
| |
| |
| |
|
|
| def _truncate(text: str, max_words: int) -> str: |
| words = text.split() |
| if len(words) <= max_words: |
| return text |
| return " ".join(words[:max_words]) + "…" |
|
|
|
|
| |
| |
| |
|
|
| def compute(ticker: str, current_period: Optional[str] = None) -> list[QuarterDelta]: |
| """Compute all text delta signals for a ticker. |
| |
| Compares the current period (latest ingested 10-Q) against the prior |
| period (previous 10-Q). Returns an empty list if sections are missing |
| or an error occurs — never raises. |
| |
| Args: |
| ticker: uppercase ticker symbol. |
| current_period: override the current period (default: latest in DB). |
| """ |
| try: |
| return _compute_inner(ticker, current_period) |
| except Exception as exc: |
| import sys |
| print(f"[textdiff] Error computing deltas for {ticker}: {exc}", file=sys.stderr) |
| return [] |
|
|
|
|
| def _compute_inner(ticker: str, current_period: Optional[str]) -> list[QuarterDelta]: |
| ticker = ticker.upper() |
|
|
| |
| periods = get_periods_for_ticker(ticker, form_type="10-Q") |
| if len(periods) < 2: |
| return [] |
|
|
| period_to = current_period if current_period else periods[0] |
| |
| if period_to in periods: |
| idx = periods.index(period_to) |
| if idx + 1 >= len(periods): |
| return [] |
| period_from = periods[idx + 1] |
| else: |
| period_from = periods[1] |
|
|
| |
| |
| |
| form_type = "10-Q" |
|
|
| |
| cur_risk = get_section(ticker, period_to, "risk_factors") or "" |
| pri_risk = get_section(ticker, period_from, "risk_factors") or "" |
| cur_mda = get_section(ticker, period_to, "mda") or "" |
| pri_mda = get_section(ticker, period_from, "mda") or "" |
|
|
| if not cur_risk and not cur_mda: |
| return [] |
|
|
| all_deltas: list[QuarterDelta] = [] |
|
|
| |
| if cur_risk and pri_risk: |
| all_deltas.extend(compute_risk_deltas(cur_risk, pri_risk, period_from, period_to, form_type)) |
|
|
| |
| cur_full = (cur_mda + "\n\n" + cur_risk).strip() |
| pri_full = (pri_mda + "\n\n" + pri_risk).strip() |
| if cur_full and pri_full: |
| all_deltas.extend(compute_lexicon_deltas(cur_full, pri_full, period_from, period_to, form_type)) |
|
|
| |
| if cur_mda and pri_mda: |
| all_deltas.extend(compute_guidance_shifts(cur_mda, pri_mda, period_from, period_to, form_type)) |
|
|
| |
| if cur_mda and pri_mda: |
| all_deltas.extend(compute_kpi_drops(cur_mda, pri_mda, period_from, period_to, form_type)) |
|
|
| |
| trend_deltas = compute_lexicon_trend(ticker) |
| all_deltas.extend(trend_deltas) |
|
|
| |
| |
| |
| trend_terms: set[str] = { |
| d.term |
| for d in trend_deltas |
| if d.kind == "term_frequency" and "quarters" in (d.computed_metric or "") |
| } |
| if trend_terms: |
| all_deltas = [ |
| d for d in all_deltas |
| if not ( |
| d.kind == "term_frequency" |
| and d.term in trend_terms |
| and "quarters" not in (d.computed_metric or "") |
| ) |
| ] |
|
|
| |
| seen: set[str] = set() |
| deduped: list[QuarterDelta] = [] |
| for d in all_deltas: |
| key = f"{d.kind}:{d.term}:{d.before_text[:40]}" |
| if key not in seen: |
| seen.add(key) |
| deduped.append(d) |
|
|
| order = {"HIGH": 0, "MEDIUM": 1, "LOW": 2} |
| deduped.sort(key=lambda d: (order[d.significance], d.kind)) |
| return deduped |
|
|