File size: 6,815 Bytes
bfd6783
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
260fe4e
 
 
 
bfd6783
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
"""Deterministic parsing of free-text expense entry.

This is the fast path: "coffee 250", "₹1,200 groceries at bigbasket",
"spent 45.50 on lunch yesterday". It runs before any model call, costs
nothing, and is reproducible. Anything it cannot make sense of is handed to
the LLM by :mod:`finbot.ingest`.
"""

from __future__ import annotations

import re
from datetime import date, timedelta
from typing import Optional

from .categories import categorise, coerce
from .models import ExpenseDraft
from .money import parse_amount

_WEEKDAYS = {
    "monday": 0, "mon": 0,
    "tuesday": 1, "tue": 1, "tues": 1,
    "wednesday": 2, "wed": 2,
    "thursday": 3, "thu": 3, "thurs": 3,
    "friday": 4, "fri": 4,
    "saturday": 5, "sat": 5,
    "sunday": 6, "sun": 6,
}

_MONTHS = {
    "jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3,
    "apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7,
    "aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9,
    "oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12,
}

# Words that carry no meaning in a description once the amount is removed.
_FILLER = {
    "spent", "spend", "paid", "pay", "bought", "buy", "for", "on", "at", "in",
    "of", "the", "a", "an", "to", "rs", "rs.", "inr", "cost", "costs", "was",
    "were", "i", "my", "me", "some", "and", "with", "got",
}

_ISO_DATE_RE = re.compile(r"\b(\d{4})-(\d{1,2})-(\d{1,2})\b")
_DMY_RE = re.compile(r"\b(\d{1,2})[/-](\d{1,2})(?:[/-](\d{2,4}))?\b")
_DAY_MONTH_RE = re.compile(
    r"\b(\d{1,2})(?:st|nd|rd|th)?\s+(" + "|".join(_MONTHS) + r")\b", re.I
)
_MONTH_DAY_RE = re.compile(
    r"\b(" + "|".join(_MONTHS) + r")\s+(\d{1,2})(?:st|nd|rd|th)?\b", re.I
)
_DAYS_AGO_RE = re.compile(r"\b(\d{1,3})\s+days?\s+ago\b", re.I)
_LAST_WEEKDAY_RE = re.compile(
    r"\b(?:last\s+)?(" + "|".join(_WEEKDAYS) + r")\b", re.I
)
_TAG_RE = re.compile(r"#([a-z_][a-z0-9_]*)", re.I)


def _clamp_year(year: int) -> int:
    if year < 100:
        return 2000 + year
    return year


def parse_date_hint(
    text: str, today: Optional[date] = None
) -> tuple[date, Optional[tuple[int, int]]]:
    """Resolve a date reference in ``text``.

    Returns (resolved_date, span_of_the_phrase). Span is None when no hint was
    found, in which case the date defaults to today. Dates that would land in
    the future are pulled back a year -- "12 dec" typed in August means last
    December, not a spend that has not happened yet.
    """
    today = today or date.today()
    lowered = text.lower()

    if m := re.search(r"\bday before yesterday\b", lowered):
        return today - timedelta(days=2), m.span()
    if m := re.search(r"\byesterday\b", lowered):
        return today - timedelta(days=1), m.span()
    if m := re.search(r"\btoday\b", lowered):
        return today, m.span()

    if m := _DAYS_AGO_RE.search(lowered):
        return today - timedelta(days=int(m.group(1))), m.span()

    if m := _ISO_DATE_RE.search(lowered):
        try:
            return date(int(m.group(1)), int(m.group(2)), int(m.group(3))), m.span()
        except ValueError:
            pass

    if m := _DAY_MONTH_RE.search(lowered):
        day, month = int(m.group(1)), _MONTHS[m.group(2).lower()]
        try:
            candidate = date(today.year, month, day)
            if candidate > today:
                candidate = date(today.year - 1, month, day)
            return candidate, m.span()
        except ValueError:
            pass

    if m := _MONTH_DAY_RE.search(lowered):
        month, day = _MONTHS[m.group(1).lower()], int(m.group(2))
        try:
            candidate = date(today.year, month, day)
            if candidate > today:
                candidate = date(today.year - 1, month, day)
            return candidate, m.span()
        except ValueError:
            pass

    if m := _DMY_RE.search(lowered):
        first, second, year_s = int(m.group(1)), int(m.group(2)), m.group(3)
        year = _clamp_year(int(year_s)) if year_s else today.year
        # Ambiguous d/m vs m/d. Prefer day-first, which covers most of the
        # world, and fall back to month-first when day-first is impossible.
        for day, month in ((first, second), (second, first)):
            try:
                candidate = date(year, month, day)
            except ValueError:
                continue
            if not year_s and candidate > today:
                candidate = date(year - 1, month, day)
            return candidate, m.span()

    if m := _LAST_WEEKDAY_RE.search(lowered):
        target = _WEEKDAYS[m.group(1).lower()]
        delta = (today.weekday() - target) % 7
        delta = delta or 7  # bare weekday name means the most recent past one
        return today - timedelta(days=delta), m.span()

    return today, None


def _clean_description(text: str, cuts: list[tuple[int, int]]) -> str:
    """Remove consumed spans, currency noise and filler words."""
    chars = list(text)
    for start, end in cuts:
        for i in range(start, min(end, len(chars))):
            chars[i] = " "
    remaining = "".join(chars)

    remaining = re.sub(r"[₹$€£¥₽₩₪₫₺₦₴₸฿₡₱﷼]", " ", remaining)
    remaining = re.sub(r"\b[A-Z]{3}\b", " ", remaining)
    remaining = _TAG_RE.sub(" ", remaining)
    remaining = re.sub(r"[^\w\s&'-]", " ", remaining)

    words = [w for w in remaining.split() if w.lower() not in _FILLER]
    return re.sub(r"\s+", " ", " ".join(words)).strip()


def parse_expense(
    text: str, default_currency: str = "INR", today: Optional[date] = None
) -> Optional[ExpenseDraft]:
    """Parse one free-text line into a draft, or None if no amount is present."""
    if not text or not text.strip():
        return None

    amount = parse_amount(text, default_currency)
    # A zero amount is not an expense. Without this, ordinary sentences that
    # happen to contain a 0 ("question 0", "flight AI 0") get logged as ₹0.00
    # entries instead of being treated as conversation.
    if amount is None or amount.minor <= 0:
        return None

    occurred_on, date_span = parse_date_hint(text, today)

    cuts = [amount.span]
    if date_span:
        cuts.append(date_span)

    description = _clean_description(text, cuts)

    # An explicit #tag overrides keyword matching.
    tag_match = _TAG_RE.search(text)
    if tag_match:
        category = coerce(tag_match.group(1))
        confidence = 1.0
    else:
        category, confidence = categorise(description)

    return ExpenseDraft(
        amount_minor=amount.minor,
        currency=amount.currency or default_currency.upper(),
        description=description or "unlabelled",
        occurred_on=occurred_on,
        category=category,
        confidence=1.0 if confidence > 0 else 0.6,
        raw=text.strip(),
    )