Oracle / finbot /analytics.py
spacedout-bits's picture
Deploy coco-finbot
bfd6783 verified
Raw History Blame Contribute Delete
9.35 kB
"""Aggregation over the ledger.
All pure functions over lists of expenses -- no I/O, no model calls -- so the
numbers are testable and always reproducible. Everything is computed
per-currency: summing INR and USD into one figure would be meaningless, so a
multi-currency ledger yields one summary per currency rather than a fictional
total.
"""
from __future__ import annotations
import re
from calendar import monthrange
from collections import defaultdict
from dataclasses import dataclass
from datetime import date, datetime, timedelta
from typing import Iterable, Optional
from zoneinfo import ZoneInfo
from .models import Expense, Summary
PERIODS = ("today", "yesterday", "week", "month", "last_month", "year", "30d", "all")
def today_in(timezone: str) -> date:
"""Current date in the user's timezone.
The container runs in UTC; for a user in Asia/Kolkata a spend logged at
2am local would otherwise land on the previous day.
"""
try:
return datetime.now(ZoneInfo(timezone)).date()
except Exception:
return date.today()
def month_bounds(day: date) -> tuple[date, date]:
return day.replace(day=1), day.replace(day=monthrange(day.year, day.month)[1])
def previous_month_bounds(day: date) -> tuple[date, date]:
first = day.replace(day=1)
last_prev = first - timedelta(days=1)
return month_bounds(last_prev)
def period_bounds(name: str, today: date) -> tuple[date, date, str]:
"""Resolve a period keyword to (start, end, human label)."""
name = (name or "month").lower().strip()
if name == "today":
return today, today, "Today"
if name == "yesterday":
day = today - timedelta(days=1)
return day, day, "Yesterday"
if name == "week":
start = today - timedelta(days=today.weekday())
return start, today, "This week"
if name == "last_month":
start, end = previous_month_bounds(today)
return start, end, start.strftime("%B %Y")
if name == "year":
return today.replace(month=1, day=1), today, str(today.year)
if name == "30d":
return today - timedelta(days=29), today, "Last 30 days"
if name == "all":
return date(1970, 1, 1), today, "All time"
start, _ = month_bounds(today)
return start, today, today.strftime("%B %Y")
def split_by_currency(expenses: Iterable[Expense]) -> dict[str, list[Expense]]:
buckets: dict[str, list[Expense]] = defaultdict(list)
for expense in expenses:
buckets[expense.currency.upper()].append(expense)
return dict(buckets)
def dominant_currency(expenses: Iterable[Expense], fallback: str) -> str:
buckets = split_by_currency(expenses)
if not buckets:
return fallback.upper()
return max(buckets.items(), key=lambda kv: (len(kv[1]), sum(e.amount_minor for e in kv[1])))[0]
def total_minor(expenses: Iterable[Expense]) -> int:
return sum(e.amount_minor for e in expenses)
def by_category(expenses: Iterable[Expense]) -> list[tuple[str, int]]:
totals: dict[str, int] = defaultdict(int)
for expense in expenses:
totals[expense.category] += expense.amount_minor
return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)
def summarise(
expenses: list[Expense],
currency: str,
label: str,
start: date,
end: date,
previous: Optional[list[Expense]] = None,
) -> Summary:
scoped = [e for e in expenses if e.currency.upper() == currency.upper()]
previous_total = None
if previous is not None:
previous_total = total_minor(
[e for e in previous if e.currency.upper() == currency.upper()]
)
return Summary(
label=label,
start=start,
end=end,
currency=currency.upper(),
total_minor=total_minor(scoped),
count=len(scoped),
by_category=by_category(scoped),
previous_total_minor=previous_total,
)
# ----------------------------------------------------------------- recurring
@dataclass
class Recurring:
label: str
category: str
currency: str
typical_minor: int
occurrences: int
months: int
last_seen: date
@property
def annualised_minor(self) -> int:
return self.typical_minor * 12
_NOISE_RE = re.compile(r"\b(?:\d{4,}|ref|txn|upi|no|id)\b[-\s]*\w*", re.I)
_TLD_RE = re.compile(r"\.(?:com|in|co|net|org|io|app|me|shop|store)\b", re.I)
#: Structural filler that appears in card narrations regardless of merchant.
#: Deliberately not locale-specific -- no city or country names, which would
#: only work for whichever market we happened to hardcode.
_NOISE_TOKENS = frozenset(
{
"com", "www", "net", "org", "ltd", "pvt", "inc", "llc", "limited",
"the", "pos", "pay", "payment", "purchase", "online", "debit",
"card", "transaction", "auto", "recurring",
}
)
def _merchant_key(expense: Expense) -> str:
"""Normalise a merchant so 'NETFLIX.COM 447192' and 'Netflix' collapse.
Purely lexical, so it will not unify names that share no words
("AMZN Mktp" vs "Amazon"). Keeping only the first few significant tokens
absorbs the trailing reference numbers and city suffixes banks append.
"""
raw = (expense.merchant or expense.description or "").lower()
raw = _TLD_RE.sub(" ", raw)
raw = _NOISE_RE.sub(" ", raw)
raw = re.sub(r"[^a-z ]+", " ", raw)
tokens = [
t for t in raw.split() if len(t) > 2 and t not in _NOISE_TOKENS
]
return " ".join(tokens[:3])
def detect_recurring(
expenses: Iterable[Expense],
min_months: int = 3,
tolerance: float = 0.20,
) -> list[Recurring]:
"""Find charges that repeat monthly at a similar amount.
Requires hits in at least ``min_months`` distinct calendar months so that
three coffees in one week don't register as a subscription.
"""
groups: dict[tuple[str, str], list[Expense]] = defaultdict(list)
for expense in expenses:
key = _merchant_key(expense)
if not key:
continue
groups[(key, expense.currency.upper())].append(expense)
found: list[Recurring] = []
for (key, currency), items in groups.items():
months = {(e.occurred_on.year, e.occurred_on.month) for e in items}
if len(months) < min_months:
continue
amounts = sorted(e.amount_minor for e in items)
median = amounts[len(amounts) // 2]
if median <= 0:
continue
# Consistent amount is what separates a subscription from ordinary
# repeat spending at the same shop.
consistent = [a for a in amounts if abs(a - median) <= median * tolerance]
if len(consistent) < len(amounts) * 0.6:
continue
best = max(items, key=lambda e: e.occurred_on)
found.append(
Recurring(
label=(best.merchant or best.description or key).strip()[:60],
category=best.category,
currency=currency,
typical_minor=median,
occurrences=len(items),
months=len(months),
last_seen=best.occurred_on,
)
)
return sorted(found, key=lambda r: r.typical_minor, reverse=True)
# ------------------------------------------------------------------- budgets
@dataclass
class BudgetStatus:
category: str
currency: str
limit_minor: int
spent_minor: int
@property
def pct(self) -> float:
if self.limit_minor <= 0:
return 0.0
return self.spent_minor / self.limit_minor * 100.0
@property
def remaining_minor(self) -> int:
return self.limit_minor - self.spent_minor
@property
def over(self) -> bool:
return self.spent_minor > self.limit_minor
def budget_status(
budgets: Iterable, month_expenses: Iterable[Expense]
) -> list[BudgetStatus]:
spent: dict[tuple[str, str], int] = defaultdict(int)
for expense in month_expenses:
spent[(expense.category, expense.currency.upper())] += expense.amount_minor
statuses = [
BudgetStatus(
category=budget.category,
currency=budget.currency.upper(),
limit_minor=budget.amount_minor,
spent_minor=spent.get((budget.category, budget.currency.upper()), 0),
)
for budget in budgets
]
return sorted(statuses, key=lambda s: s.pct, reverse=True)
def daily_burn(expenses: list[Expense], start: date, end: date) -> float:
"""Average spend per elapsed day, in minor units."""
days = max(1, (end - start).days + 1)
return total_minor(expenses) / days
def project_month_end(expenses: list[Expense], today: date) -> int:
"""Straight-line projection of this month's total from spend so far."""
start, end = month_bounds(today)
elapsed = max(1, (today - start).days + 1)
days_in_month = (end - start).days + 1
return int(total_minor(expenses) / elapsed * days_in_month)
def top_merchants(expenses: Iterable[Expense], limit: int = 5) -> list[tuple[str, int]]:
totals: dict[str, int] = defaultdict(int)
for expense in expenses:
label = (expense.merchant or expense.description or "unlabelled").strip()[:40]
totals[label] += expense.amount_minor
return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)[:limit]