"""Deterministic parsing of free-text expense entry. This is the fast path: "coffee 250", "₹1,200 groceries at bigbasket", "spent 45.50 on lunch yesterday". It runs before any model call, costs nothing, and is reproducible. Anything it cannot make sense of is handed to the LLM by :mod:`finbot.ingest`. """ from __future__ import annotations import re from datetime import date, timedelta from typing import Optional from .categories import categorise, coerce from .models import ExpenseDraft from .money import parse_amount _WEEKDAYS = { "monday": 0, "mon": 0, "tuesday": 1, "tue": 1, "tues": 1, "wednesday": 2, "wed": 2, "thursday": 3, "thu": 3, "thurs": 3, "friday": 4, "fri": 4, "saturday": 5, "sat": 5, "sunday": 6, "sun": 6, } _MONTHS = { "jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3, "apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7, "aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9, "oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12, } # Words that carry no meaning in a description once the amount is removed. _FILLER = { "spent", "spend", "paid", "pay", "bought", "buy", "for", "on", "at", "in", "of", "the", "a", "an", "to", "rs", "rs.", "inr", "cost", "costs", "was", "were", "i", "my", "me", "some", "and", "with", "got", } _ISO_DATE_RE = re.compile(r"\b(\d{4})-(\d{1,2})-(\d{1,2})\b") _DMY_RE = re.compile(r"\b(\d{1,2})[/-](\d{1,2})(?:[/-](\d{2,4}))?\b") _DAY_MONTH_RE = re.compile( r"\b(\d{1,2})(?:st|nd|rd|th)?\s+(" + "|".join(_MONTHS) + r")\b", re.I ) _MONTH_DAY_RE = re.compile( r"\b(" + "|".join(_MONTHS) + r")\s+(\d{1,2})(?:st|nd|rd|th)?\b", re.I ) _DAYS_AGO_RE = re.compile(r"\b(\d{1,3})\s+days?\s+ago\b", re.I) _LAST_WEEKDAY_RE = re.compile( r"\b(?:last\s+)?(" + "|".join(_WEEKDAYS) + r")\b", re.I ) _TAG_RE = re.compile(r"#([a-z_][a-z0-9_]*)", re.I) def _clamp_year(year: int) -> int: if year < 100: return 2000 + year return year def parse_date_hint( text: str, today: Optional[date] = None ) -> tuple[date, Optional[tuple[int, int]]]: """Resolve a date reference in ``text``. Returns (resolved_date, span_of_the_phrase). Span is None when no hint was found, in which case the date defaults to today. Dates that would land in the future are pulled back a year -- "12 dec" typed in August means last December, not a spend that has not happened yet. """ today = today or date.today() lowered = text.lower() if m := re.search(r"\bday before yesterday\b", lowered): return today - timedelta(days=2), m.span() if m := re.search(r"\byesterday\b", lowered): return today - timedelta(days=1), m.span() if m := re.search(r"\btoday\b", lowered): return today, m.span() if m := _DAYS_AGO_RE.search(lowered): return today - timedelta(days=int(m.group(1))), m.span() if m := _ISO_DATE_RE.search(lowered): try: return date(int(m.group(1)), int(m.group(2)), int(m.group(3))), m.span() except ValueError: pass if m := _DAY_MONTH_RE.search(lowered): day, month = int(m.group(1)), _MONTHS[m.group(2).lower()] try: candidate = date(today.year, month, day) if candidate > today: candidate = date(today.year - 1, month, day) return candidate, m.span() except ValueError: pass if m := _MONTH_DAY_RE.search(lowered): month, day = _MONTHS[m.group(1).lower()], int(m.group(2)) try: candidate = date(today.year, month, day) if candidate > today: candidate = date(today.year - 1, month, day) return candidate, m.span() except ValueError: pass if m := _DMY_RE.search(lowered): first, second, year_s = int(m.group(1)), int(m.group(2)), m.group(3) year = _clamp_year(int(year_s)) if year_s else today.year # Ambiguous d/m vs m/d. Prefer day-first, which covers most of the # world, and fall back to month-first when day-first is impossible. for day, month in ((first, second), (second, first)): try: candidate = date(year, month, day) except ValueError: continue if not year_s and candidate > today: candidate = date(year - 1, month, day) return candidate, m.span() if m := _LAST_WEEKDAY_RE.search(lowered): target = _WEEKDAYS[m.group(1).lower()] delta = (today.weekday() - target) % 7 delta = delta or 7 # bare weekday name means the most recent past one return today - timedelta(days=delta), m.span() return today, None def _clean_description(text: str, cuts: list[tuple[int, int]]) -> str: """Remove consumed spans, currency noise and filler words.""" chars = list(text) for start, end in cuts: for i in range(start, min(end, len(chars))): chars[i] = " " remaining = "".join(chars) remaining = re.sub(r"[₹$€£¥₽₩₪₫₺₦₴₸฿₡₱﷼]", " ", remaining) remaining = re.sub(r"\b[A-Z]{3}\b", " ", remaining) remaining = _TAG_RE.sub(" ", remaining) remaining = re.sub(r"[^\w\s&'-]", " ", remaining) words = [w for w in remaining.split() if w.lower() not in _FILLER] return re.sub(r"\s+", " ", " ".join(words)).strip() def parse_expense( text: str, default_currency: str = "INR", today: Optional[date] = None ) -> Optional[ExpenseDraft]: """Parse one free-text line into a draft, or None if no amount is present.""" if not text or not text.strip(): return None amount = parse_amount(text, default_currency) # A zero amount is not an expense. Without this, ordinary sentences that # happen to contain a 0 ("question 0", "flight AI 0") get logged as ₹0.00 # entries instead of being treated as conversation. if amount is None or amount.minor <= 0: return None occurred_on, date_span = parse_date_hint(text, today) cuts = [amount.span] if date_span: cuts.append(date_span) description = _clean_description(text, cuts) # An explicit #tag overrides keyword matching. tag_match = _TAG_RE.search(text) if tag_match: category = coerce(tag_match.group(1)) confidence = 1.0 else: category, confidence = categorise(description) return ExpenseDraft( amount_minor=amount.minor, currency=amount.currency or default_currency.upper(), description=description or "unlabelled", occurred_on=occurred_on, category=category, confidence=1.0 if confidence > 0 else 0.6, raw=text.strip(), )