Spaces:
Running on Zero
Running on Zero
File size: 6,815 Bytes
bfd6783 260fe4e bfd6783 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 | """Deterministic parsing of free-text expense entry.
This is the fast path: "coffee 250", "₹1,200 groceries at bigbasket",
"spent 45.50 on lunch yesterday". It runs before any model call, costs
nothing, and is reproducible. Anything it cannot make sense of is handed to
the LLM by :mod:`finbot.ingest`.
"""
from __future__ import annotations
import re
from datetime import date, timedelta
from typing import Optional
from .categories import categorise, coerce
from .models import ExpenseDraft
from .money import parse_amount
_WEEKDAYS = {
"monday": 0, "mon": 0,
"tuesday": 1, "tue": 1, "tues": 1,
"wednesday": 2, "wed": 2,
"thursday": 3, "thu": 3, "thurs": 3,
"friday": 4, "fri": 4,
"saturday": 5, "sat": 5,
"sunday": 6, "sun": 6,
}
_MONTHS = {
"jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3,
"apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7,
"aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9,
"oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12,
}
# Words that carry no meaning in a description once the amount is removed.
_FILLER = {
"spent", "spend", "paid", "pay", "bought", "buy", "for", "on", "at", "in",
"of", "the", "a", "an", "to", "rs", "rs.", "inr", "cost", "costs", "was",
"were", "i", "my", "me", "some", "and", "with", "got",
}
_ISO_DATE_RE = re.compile(r"\b(\d{4})-(\d{1,2})-(\d{1,2})\b")
_DMY_RE = re.compile(r"\b(\d{1,2})[/-](\d{1,2})(?:[/-](\d{2,4}))?\b")
_DAY_MONTH_RE = re.compile(
r"\b(\d{1,2})(?:st|nd|rd|th)?\s+(" + "|".join(_MONTHS) + r")\b", re.I
)
_MONTH_DAY_RE = re.compile(
r"\b(" + "|".join(_MONTHS) + r")\s+(\d{1,2})(?:st|nd|rd|th)?\b", re.I
)
_DAYS_AGO_RE = re.compile(r"\b(\d{1,3})\s+days?\s+ago\b", re.I)
_LAST_WEEKDAY_RE = re.compile(
r"\b(?:last\s+)?(" + "|".join(_WEEKDAYS) + r")\b", re.I
)
_TAG_RE = re.compile(r"#([a-z_][a-z0-9_]*)", re.I)
def _clamp_year(year: int) -> int:
if year < 100:
return 2000 + year
return year
def parse_date_hint(
text: str, today: Optional[date] = None
) -> tuple[date, Optional[tuple[int, int]]]:
"""Resolve a date reference in ``text``.
Returns (resolved_date, span_of_the_phrase). Span is None when no hint was
found, in which case the date defaults to today. Dates that would land in
the future are pulled back a year -- "12 dec" typed in August means last
December, not a spend that has not happened yet.
"""
today = today or date.today()
lowered = text.lower()
if m := re.search(r"\bday before yesterday\b", lowered):
return today - timedelta(days=2), m.span()
if m := re.search(r"\byesterday\b", lowered):
return today - timedelta(days=1), m.span()
if m := re.search(r"\btoday\b", lowered):
return today, m.span()
if m := _DAYS_AGO_RE.search(lowered):
return today - timedelta(days=int(m.group(1))), m.span()
if m := _ISO_DATE_RE.search(lowered):
try:
return date(int(m.group(1)), int(m.group(2)), int(m.group(3))), m.span()
except ValueError:
pass
if m := _DAY_MONTH_RE.search(lowered):
day, month = int(m.group(1)), _MONTHS[m.group(2).lower()]
try:
candidate = date(today.year, month, day)
if candidate > today:
candidate = date(today.year - 1, month, day)
return candidate, m.span()
except ValueError:
pass
if m := _MONTH_DAY_RE.search(lowered):
month, day = _MONTHS[m.group(1).lower()], int(m.group(2))
try:
candidate = date(today.year, month, day)
if candidate > today:
candidate = date(today.year - 1, month, day)
return candidate, m.span()
except ValueError:
pass
if m := _DMY_RE.search(lowered):
first, second, year_s = int(m.group(1)), int(m.group(2)), m.group(3)
year = _clamp_year(int(year_s)) if year_s else today.year
# Ambiguous d/m vs m/d. Prefer day-first, which covers most of the
# world, and fall back to month-first when day-first is impossible.
for day, month in ((first, second), (second, first)):
try:
candidate = date(year, month, day)
except ValueError:
continue
if not year_s and candidate > today:
candidate = date(year - 1, month, day)
return candidate, m.span()
if m := _LAST_WEEKDAY_RE.search(lowered):
target = _WEEKDAYS[m.group(1).lower()]
delta = (today.weekday() - target) % 7
delta = delta or 7 # bare weekday name means the most recent past one
return today - timedelta(days=delta), m.span()
return today, None
def _clean_description(text: str, cuts: list[tuple[int, int]]) -> str:
"""Remove consumed spans, currency noise and filler words."""
chars = list(text)
for start, end in cuts:
for i in range(start, min(end, len(chars))):
chars[i] = " "
remaining = "".join(chars)
remaining = re.sub(r"[₹$€£¥₽₩₪₫₺₦₴₸฿₡₱﷼]", " ", remaining)
remaining = re.sub(r"\b[A-Z]{3}\b", " ", remaining)
remaining = _TAG_RE.sub(" ", remaining)
remaining = re.sub(r"[^\w\s&'-]", " ", remaining)
words = [w for w in remaining.split() if w.lower() not in _FILLER]
return re.sub(r"\s+", " ", " ".join(words)).strip()
def parse_expense(
text: str, default_currency: str = "INR", today: Optional[date] = None
) -> Optional[ExpenseDraft]:
"""Parse one free-text line into a draft, or None if no amount is present."""
if not text or not text.strip():
return None
amount = parse_amount(text, default_currency)
# A zero amount is not an expense. Without this, ordinary sentences that
# happen to contain a 0 ("question 0", "flight AI 0") get logged as ₹0.00
# entries instead of being treated as conversation.
if amount is None or amount.minor <= 0:
return None
occurred_on, date_span = parse_date_hint(text, today)
cuts = [amount.span]
if date_span:
cuts.append(date_span)
description = _clean_description(text, cuts)
# An explicit #tag overrides keyword matching.
tag_match = _TAG_RE.search(text)
if tag_match:
category = coerce(tag_match.group(1))
confidence = 1.0
else:
category, confidence = categorise(description)
return ExpenseDraft(
amount_minor=amount.minor,
currency=amount.currency or default_currency.upper(),
description=description or "unlabelled",
occurred_on=occurred_on,
category=category,
confidence=1.0 if confidence > 0 else 0.6,
raw=text.strip(),
)
|