"""Aggregation over the ledger. All pure functions over lists of expenses -- no I/O, no model calls -- so the numbers are testable and always reproducible. Everything is computed per-currency: summing INR and USD into one figure would be meaningless, so a multi-currency ledger yields one summary per currency rather than a fictional total. """ from __future__ import annotations import re from calendar import monthrange from collections import defaultdict from dataclasses import dataclass from datetime import date, datetime, timedelta from typing import Iterable, Optional from zoneinfo import ZoneInfo from .models import Expense, Summary PERIODS = ("today", "yesterday", "week", "month", "last_month", "year", "30d", "all") def today_in(timezone: str) -> date: """Current date in the user's timezone. The container runs in UTC; for a user in Asia/Kolkata a spend logged at 2am local would otherwise land on the previous day. """ try: return datetime.now(ZoneInfo(timezone)).date() except Exception: return date.today() def month_bounds(day: date) -> tuple[date, date]: return day.replace(day=1), day.replace(day=monthrange(day.year, day.month)[1]) def previous_month_bounds(day: date) -> tuple[date, date]: first = day.replace(day=1) last_prev = first - timedelta(days=1) return month_bounds(last_prev) def period_bounds(name: str, today: date) -> tuple[date, date, str]: """Resolve a period keyword to (start, end, human label).""" name = (name or "month").lower().strip() if name == "today": return today, today, "Today" if name == "yesterday": day = today - timedelta(days=1) return day, day, "Yesterday" if name == "week": start = today - timedelta(days=today.weekday()) return start, today, "This week" if name == "last_month": start, end = previous_month_bounds(today) return start, end, start.strftime("%B %Y") if name == "year": return today.replace(month=1, day=1), today, str(today.year) if name == "30d": return today - timedelta(days=29), today, "Last 30 days" if name == "all": return date(1970, 1, 1), today, "All time" start, _ = month_bounds(today) return start, today, today.strftime("%B %Y") def split_by_currency(expenses: Iterable[Expense]) -> dict[str, list[Expense]]: buckets: dict[str, list[Expense]] = defaultdict(list) for expense in expenses: buckets[expense.currency.upper()].append(expense) return dict(buckets) def dominant_currency(expenses: Iterable[Expense], fallback: str) -> str: buckets = split_by_currency(expenses) if not buckets: return fallback.upper() return max(buckets.items(), key=lambda kv: (len(kv[1]), sum(e.amount_minor for e in kv[1])))[0] def total_minor(expenses: Iterable[Expense]) -> int: return sum(e.amount_minor for e in expenses) def by_category(expenses: Iterable[Expense]) -> list[tuple[str, int]]: totals: dict[str, int] = defaultdict(int) for expense in expenses: totals[expense.category] += expense.amount_minor return sorted(totals.items(), key=lambda kv: kv[1], reverse=True) def summarise( expenses: list[Expense], currency: str, label: str, start: date, end: date, previous: Optional[list[Expense]] = None, ) -> Summary: scoped = [e for e in expenses if e.currency.upper() == currency.upper()] previous_total = None if previous is not None: previous_total = total_minor( [e for e in previous if e.currency.upper() == currency.upper()] ) return Summary( label=label, start=start, end=end, currency=currency.upper(), total_minor=total_minor(scoped), count=len(scoped), by_category=by_category(scoped), previous_total_minor=previous_total, ) # ----------------------------------------------------------------- recurring @dataclass class Recurring: label: str category: str currency: str typical_minor: int occurrences: int months: int last_seen: date @property def annualised_minor(self) -> int: return self.typical_minor * 12 _NOISE_RE = re.compile(r"\b(?:\d{4,}|ref|txn|upi|no|id)\b[-\s]*\w*", re.I) _TLD_RE = re.compile(r"\.(?:com|in|co|net|org|io|app|me|shop|store)\b", re.I) #: Structural filler that appears in card narrations regardless of merchant. #: Deliberately not locale-specific -- no city or country names, which would #: only work for whichever market we happened to hardcode. _NOISE_TOKENS = frozenset( { "com", "www", "net", "org", "ltd", "pvt", "inc", "llc", "limited", "the", "pos", "pay", "payment", "purchase", "online", "debit", "card", "transaction", "auto", "recurring", } ) def _merchant_key(expense: Expense) -> str: """Normalise a merchant so 'NETFLIX.COM 447192' and 'Netflix' collapse. Purely lexical, so it will not unify names that share no words ("AMZN Mktp" vs "Amazon"). Keeping only the first few significant tokens absorbs the trailing reference numbers and city suffixes banks append. """ raw = (expense.merchant or expense.description or "").lower() raw = _TLD_RE.sub(" ", raw) raw = _NOISE_RE.sub(" ", raw) raw = re.sub(r"[^a-z ]+", " ", raw) tokens = [ t for t in raw.split() if len(t) > 2 and t not in _NOISE_TOKENS ] return " ".join(tokens[:3]) def detect_recurring( expenses: Iterable[Expense], min_months: int = 3, tolerance: float = 0.20, ) -> list[Recurring]: """Find charges that repeat monthly at a similar amount. Requires hits in at least ``min_months`` distinct calendar months so that three coffees in one week don't register as a subscription. """ groups: dict[tuple[str, str], list[Expense]] = defaultdict(list) for expense in expenses: key = _merchant_key(expense) if not key: continue groups[(key, expense.currency.upper())].append(expense) found: list[Recurring] = [] for (key, currency), items in groups.items(): months = {(e.occurred_on.year, e.occurred_on.month) for e in items} if len(months) < min_months: continue amounts = sorted(e.amount_minor for e in items) median = amounts[len(amounts) // 2] if median <= 0: continue # Consistent amount is what separates a subscription from ordinary # repeat spending at the same shop. consistent = [a for a in amounts if abs(a - median) <= median * tolerance] if len(consistent) < len(amounts) * 0.6: continue best = max(items, key=lambda e: e.occurred_on) found.append( Recurring( label=(best.merchant or best.description or key).strip()[:60], category=best.category, currency=currency, typical_minor=median, occurrences=len(items), months=len(months), last_seen=best.occurred_on, ) ) return sorted(found, key=lambda r: r.typical_minor, reverse=True) # ------------------------------------------------------------------- budgets @dataclass class BudgetStatus: category: str currency: str limit_minor: int spent_minor: int @property def pct(self) -> float: if self.limit_minor <= 0: return 0.0 return self.spent_minor / self.limit_minor * 100.0 @property def remaining_minor(self) -> int: return self.limit_minor - self.spent_minor @property def over(self) -> bool: return self.spent_minor > self.limit_minor def budget_status( budgets: Iterable, month_expenses: Iterable[Expense] ) -> list[BudgetStatus]: spent: dict[tuple[str, str], int] = defaultdict(int) for expense in month_expenses: spent[(expense.category, expense.currency.upper())] += expense.amount_minor statuses = [ BudgetStatus( category=budget.category, currency=budget.currency.upper(), limit_minor=budget.amount_minor, spent_minor=spent.get((budget.category, budget.currency.upper()), 0), ) for budget in budgets ] return sorted(statuses, key=lambda s: s.pct, reverse=True) def daily_burn(expenses: list[Expense], start: date, end: date) -> float: """Average spend per elapsed day, in minor units.""" days = max(1, (end - start).days + 1) return total_minor(expenses) / days def project_month_end(expenses: list[Expense], today: date) -> int: """Straight-line projection of this month's total from spend so far.""" start, end = month_bounds(today) elapsed = max(1, (today - start).days + 1) days_in_month = (end - start).days + 1 return int(total_minor(expenses) / elapsed * days_in_month) def top_merchants(expenses: Iterable[Expense], limit: int = 5) -> list[tuple[str, int]]: totals: dict[str, int] = defaultdict(int) for expense in expenses: label = (expense.merchant or expense.description or "unlabelled").strip()[:40] totals[label] += expense.amount_minor return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)[:limit]