Spaces:
Running on Zero
Running on Zero
Download finbot/analytics.py from spacedout-bits/Oracle: direct link, hf CLI and curl.
- Browser
- Download file 9.35 kB
-
https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/finbot/analytics.py
- Command line
-
hf download hf://spaces/spacedout-bits/Oracle/finbot/analytics.py
-
curl -L -o analytics.py https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/finbot/analytics.py
9.35 kB
| """Aggregation over the ledger. | |
| All pure functions over lists of expenses -- no I/O, no model calls -- so the | |
| numbers are testable and always reproducible. Everything is computed | |
| per-currency: summing INR and USD into one figure would be meaningless, so a | |
| multi-currency ledger yields one summary per currency rather than a fictional | |
| total. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from calendar import monthrange | |
| from collections import defaultdict | |
| from dataclasses import dataclass | |
| from datetime import date, datetime, timedelta | |
| from typing import Iterable, Optional | |
| from zoneinfo import ZoneInfo | |
| from .models import Expense, Summary | |
| PERIODS = ("today", "yesterday", "week", "month", "last_month", "year", "30d", "all") | |
| def today_in(timezone: str) -> date: | |
| """Current date in the user's timezone. | |
| The container runs in UTC; for a user in Asia/Kolkata a spend logged at | |
| 2am local would otherwise land on the previous day. | |
| """ | |
| try: | |
| return datetime.now(ZoneInfo(timezone)).date() | |
| except Exception: | |
| return date.today() | |
| def month_bounds(day: date) -> tuple[date, date]: | |
| return day.replace(day=1), day.replace(day=monthrange(day.year, day.month)[1]) | |
| def previous_month_bounds(day: date) -> tuple[date, date]: | |
| first = day.replace(day=1) | |
| last_prev = first - timedelta(days=1) | |
| return month_bounds(last_prev) | |
| def period_bounds(name: str, today: date) -> tuple[date, date, str]: | |
| """Resolve a period keyword to (start, end, human label).""" | |
| name = (name or "month").lower().strip() | |
| if name == "today": | |
| return today, today, "Today" | |
| if name == "yesterday": | |
| day = today - timedelta(days=1) | |
| return day, day, "Yesterday" | |
| if name == "week": | |
| start = today - timedelta(days=today.weekday()) | |
| return start, today, "This week" | |
| if name == "last_month": | |
| start, end = previous_month_bounds(today) | |
| return start, end, start.strftime("%B %Y") | |
| if name == "year": | |
| return today.replace(month=1, day=1), today, str(today.year) | |
| if name == "30d": | |
| return today - timedelta(days=29), today, "Last 30 days" | |
| if name == "all": | |
| return date(1970, 1, 1), today, "All time" | |
| start, _ = month_bounds(today) | |
| return start, today, today.strftime("%B %Y") | |
| def split_by_currency(expenses: Iterable[Expense]) -> dict[str, list[Expense]]: | |
| buckets: dict[str, list[Expense]] = defaultdict(list) | |
| for expense in expenses: | |
| buckets[expense.currency.upper()].append(expense) | |
| return dict(buckets) | |
| def dominant_currency(expenses: Iterable[Expense], fallback: str) -> str: | |
| buckets = split_by_currency(expenses) | |
| if not buckets: | |
| return fallback.upper() | |
| return max(buckets.items(), key=lambda kv: (len(kv[1]), sum(e.amount_minor for e in kv[1])))[0] | |
| def total_minor(expenses: Iterable[Expense]) -> int: | |
| return sum(e.amount_minor for e in expenses) | |
| def by_category(expenses: Iterable[Expense]) -> list[tuple[str, int]]: | |
| totals: dict[str, int] = defaultdict(int) | |
| for expense in expenses: | |
| totals[expense.category] += expense.amount_minor | |
| return sorted(totals.items(), key=lambda kv: kv[1], reverse=True) | |
| def summarise( | |
| expenses: list[Expense], | |
| currency: str, | |
| label: str, | |
| start: date, | |
| end: date, | |
| previous: Optional[list[Expense]] = None, | |
| ) -> Summary: | |
| scoped = [e for e in expenses if e.currency.upper() == currency.upper()] | |
| previous_total = None | |
| if previous is not None: | |
| previous_total = total_minor( | |
| [e for e in previous if e.currency.upper() == currency.upper()] | |
| ) | |
| return Summary( | |
| label=label, | |
| start=start, | |
| end=end, | |
| currency=currency.upper(), | |
| total_minor=total_minor(scoped), | |
| count=len(scoped), | |
| by_category=by_category(scoped), | |
| previous_total_minor=previous_total, | |
| ) | |
| # ----------------------------------------------------------------- recurring | |
| class Recurring: | |
| label: str | |
| category: str | |
| currency: str | |
| typical_minor: int | |
| occurrences: int | |
| months: int | |
| last_seen: date | |
| def annualised_minor(self) -> int: | |
| return self.typical_minor * 12 | |
| _NOISE_RE = re.compile(r"\b(?:\d{4,}|ref|txn|upi|no|id)\b[-\s]*\w*", re.I) | |
| _TLD_RE = re.compile(r"\.(?:com|in|co|net|org|io|app|me|shop|store)\b", re.I) | |
| #: Structural filler that appears in card narrations regardless of merchant. | |
| #: Deliberately not locale-specific -- no city or country names, which would | |
| #: only work for whichever market we happened to hardcode. | |
| _NOISE_TOKENS = frozenset( | |
| { | |
| "com", "www", "net", "org", "ltd", "pvt", "inc", "llc", "limited", | |
| "the", "pos", "pay", "payment", "purchase", "online", "debit", | |
| "card", "transaction", "auto", "recurring", | |
| } | |
| ) | |
| def _merchant_key(expense: Expense) -> str: | |
| """Normalise a merchant so 'NETFLIX.COM 447192' and 'Netflix' collapse. | |
| Purely lexical, so it will not unify names that share no words | |
| ("AMZN Mktp" vs "Amazon"). Keeping only the first few significant tokens | |
| absorbs the trailing reference numbers and city suffixes banks append. | |
| """ | |
| raw = (expense.merchant or expense.description or "").lower() | |
| raw = _TLD_RE.sub(" ", raw) | |
| raw = _NOISE_RE.sub(" ", raw) | |
| raw = re.sub(r"[^a-z ]+", " ", raw) | |
| tokens = [ | |
| t for t in raw.split() if len(t) > 2 and t not in _NOISE_TOKENS | |
| ] | |
| return " ".join(tokens[:3]) | |
| def detect_recurring( | |
| expenses: Iterable[Expense], | |
| min_months: int = 3, | |
| tolerance: float = 0.20, | |
| ) -> list[Recurring]: | |
| """Find charges that repeat monthly at a similar amount. | |
| Requires hits in at least ``min_months`` distinct calendar months so that | |
| three coffees in one week don't register as a subscription. | |
| """ | |
| groups: dict[tuple[str, str], list[Expense]] = defaultdict(list) | |
| for expense in expenses: | |
| key = _merchant_key(expense) | |
| if not key: | |
| continue | |
| groups[(key, expense.currency.upper())].append(expense) | |
| found: list[Recurring] = [] | |
| for (key, currency), items in groups.items(): | |
| months = {(e.occurred_on.year, e.occurred_on.month) for e in items} | |
| if len(months) < min_months: | |
| continue | |
| amounts = sorted(e.amount_minor for e in items) | |
| median = amounts[len(amounts) // 2] | |
| if median <= 0: | |
| continue | |
| # Consistent amount is what separates a subscription from ordinary | |
| # repeat spending at the same shop. | |
| consistent = [a for a in amounts if abs(a - median) <= median * tolerance] | |
| if len(consistent) < len(amounts) * 0.6: | |
| continue | |
| best = max(items, key=lambda e: e.occurred_on) | |
| found.append( | |
| Recurring( | |
| label=(best.merchant or best.description or key).strip()[:60], | |
| category=best.category, | |
| currency=currency, | |
| typical_minor=median, | |
| occurrences=len(items), | |
| months=len(months), | |
| last_seen=best.occurred_on, | |
| ) | |
| ) | |
| return sorted(found, key=lambda r: r.typical_minor, reverse=True) | |
| # ------------------------------------------------------------------- budgets | |
| class BudgetStatus: | |
| category: str | |
| currency: str | |
| limit_minor: int | |
| spent_minor: int | |
| def pct(self) -> float: | |
| if self.limit_minor <= 0: | |
| return 0.0 | |
| return self.spent_minor / self.limit_minor * 100.0 | |
| def remaining_minor(self) -> int: | |
| return self.limit_minor - self.spent_minor | |
| def over(self) -> bool: | |
| return self.spent_minor > self.limit_minor | |
| def budget_status( | |
| budgets: Iterable, month_expenses: Iterable[Expense] | |
| ) -> list[BudgetStatus]: | |
| spent: dict[tuple[str, str], int] = defaultdict(int) | |
| for expense in month_expenses: | |
| spent[(expense.category, expense.currency.upper())] += expense.amount_minor | |
| statuses = [ | |
| BudgetStatus( | |
| category=budget.category, | |
| currency=budget.currency.upper(), | |
| limit_minor=budget.amount_minor, | |
| spent_minor=spent.get((budget.category, budget.currency.upper()), 0), | |
| ) | |
| for budget in budgets | |
| ] | |
| return sorted(statuses, key=lambda s: s.pct, reverse=True) | |
| def daily_burn(expenses: list[Expense], start: date, end: date) -> float: | |
| """Average spend per elapsed day, in minor units.""" | |
| days = max(1, (end - start).days + 1) | |
| return total_minor(expenses) / days | |
| def project_month_end(expenses: list[Expense], today: date) -> int: | |
| """Straight-line projection of this month's total from spend so far.""" | |
| start, end = month_bounds(today) | |
| elapsed = max(1, (today - start).days + 1) | |
| days_in_month = (end - start).days + 1 | |
| return int(total_minor(expenses) / elapsed * days_in_month) | |
| def top_merchants(expenses: Iterable[Expense], limit: int = 5) -> list[tuple[str, int]]: | |
| totals: dict[str, int] = defaultdict(int) | |
| for expense in expenses: | |
| label = (expense.merchant or expense.description or "unlabelled").strip()[:40] | |
| totals[label] += expense.amount_minor | |
| return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)[:limit] | |