Oracle / finbot /parser.py
spacedout-bits's picture
Conversational fallback with memory; reject zero-amount expenses
260fe4e verified
Raw History Blame Contribute Delete
6.82 kB
"""Deterministic parsing of free-text expense entry.
This is the fast path: "coffee 250", "₹1,200 groceries at bigbasket",
"spent 45.50 on lunch yesterday". It runs before any model call, costs
nothing, and is reproducible. Anything it cannot make sense of is handed to
the LLM by :mod:`finbot.ingest`.
"""
from __future__ import annotations
import re
from datetime import date, timedelta
from typing import Optional
from .categories import categorise, coerce
from .models import ExpenseDraft
from .money import parse_amount
_WEEKDAYS = {
"monday": 0, "mon": 0,
"tuesday": 1, "tue": 1, "tues": 1,
"wednesday": 2, "wed": 2,
"thursday": 3, "thu": 3, "thurs": 3,
"friday": 4, "fri": 4,
"saturday": 5, "sat": 5,
"sunday": 6, "sun": 6,
}
_MONTHS = {
"jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3,
"apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7,
"aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9,
"oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12,
}
# Words that carry no meaning in a description once the amount is removed.
_FILLER = {
"spent", "spend", "paid", "pay", "bought", "buy", "for", "on", "at", "in",
"of", "the", "a", "an", "to", "rs", "rs.", "inr", "cost", "costs", "was",
"were", "i", "my", "me", "some", "and", "with", "got",
}
_ISO_DATE_RE = re.compile(r"\b(\d{4})-(\d{1,2})-(\d{1,2})\b")
_DMY_RE = re.compile(r"\b(\d{1,2})[/-](\d{1,2})(?:[/-](\d{2,4}))?\b")
_DAY_MONTH_RE = re.compile(
r"\b(\d{1,2})(?:st|nd|rd|th)?\s+(" + "|".join(_MONTHS) + r")\b", re.I
)
_MONTH_DAY_RE = re.compile(
r"\b(" + "|".join(_MONTHS) + r")\s+(\d{1,2})(?:st|nd|rd|th)?\b", re.I
)
_DAYS_AGO_RE = re.compile(r"\b(\d{1,3})\s+days?\s+ago\b", re.I)
_LAST_WEEKDAY_RE = re.compile(
r"\b(?:last\s+)?(" + "|".join(_WEEKDAYS) + r")\b", re.I
)
_TAG_RE = re.compile(r"#([a-z_][a-z0-9_]*)", re.I)
def _clamp_year(year: int) -> int:
if year < 100:
return 2000 + year
return year
def parse_date_hint(
text: str, today: Optional[date] = None
) -> tuple[date, Optional[tuple[int, int]]]:
"""Resolve a date reference in ``text``.
Returns (resolved_date, span_of_the_phrase). Span is None when no hint was
found, in which case the date defaults to today. Dates that would land in
the future are pulled back a year -- "12 dec" typed in August means last
December, not a spend that has not happened yet.
"""
today = today or date.today()
lowered = text.lower()
if m := re.search(r"\bday before yesterday\b", lowered):
return today - timedelta(days=2), m.span()
if m := re.search(r"\byesterday\b", lowered):
return today - timedelta(days=1), m.span()
if m := re.search(r"\btoday\b", lowered):
return today, m.span()
if m := _DAYS_AGO_RE.search(lowered):
return today - timedelta(days=int(m.group(1))), m.span()
if m := _ISO_DATE_RE.search(lowered):
try:
return date(int(m.group(1)), int(m.group(2)), int(m.group(3))), m.span()
except ValueError:
pass
if m := _DAY_MONTH_RE.search(lowered):
day, month = int(m.group(1)), _MONTHS[m.group(2).lower()]
try:
candidate = date(today.year, month, day)
if candidate > today:
candidate = date(today.year - 1, month, day)
return candidate, m.span()
except ValueError:
pass
if m := _MONTH_DAY_RE.search(lowered):
month, day = _MONTHS[m.group(1).lower()], int(m.group(2))
try:
candidate = date(today.year, month, day)
if candidate > today:
candidate = date(today.year - 1, month, day)
return candidate, m.span()
except ValueError:
pass
if m := _DMY_RE.search(lowered):
first, second, year_s = int(m.group(1)), int(m.group(2)), m.group(3)
year = _clamp_year(int(year_s)) if year_s else today.year
# Ambiguous d/m vs m/d. Prefer day-first, which covers most of the
# world, and fall back to month-first when day-first is impossible.
for day, month in ((first, second), (second, first)):
try:
candidate = date(year, month, day)
except ValueError:
continue
if not year_s and candidate > today:
candidate = date(year - 1, month, day)
return candidate, m.span()
if m := _LAST_WEEKDAY_RE.search(lowered):
target = _WEEKDAYS[m.group(1).lower()]
delta = (today.weekday() - target) % 7
delta = delta or 7 # bare weekday name means the most recent past one
return today - timedelta(days=delta), m.span()
return today, None
def _clean_description(text: str, cuts: list[tuple[int, int]]) -> str:
"""Remove consumed spans, currency noise and filler words."""
chars = list(text)
for start, end in cuts:
for i in range(start, min(end, len(chars))):
chars[i] = " "
remaining = "".join(chars)
remaining = re.sub(r"[₹$€£¥₽₩₪₫₺₦₴₸฿₡₱﷼]", " ", remaining)
remaining = re.sub(r"\b[A-Z]{3}\b", " ", remaining)
remaining = _TAG_RE.sub(" ", remaining)
remaining = re.sub(r"[^\w\s&'-]", " ", remaining)
words = [w for w in remaining.split() if w.lower() not in _FILLER]
return re.sub(r"\s+", " ", " ".join(words)).strip()
def parse_expense(
text: str, default_currency: str = "INR", today: Optional[date] = None
) -> Optional[ExpenseDraft]:
"""Parse one free-text line into a draft, or None if no amount is present."""
if not text or not text.strip():
return None
amount = parse_amount(text, default_currency)
# A zero amount is not an expense. Without this, ordinary sentences that
# happen to contain a 0 ("question 0", "flight AI 0") get logged as ₹0.00
# entries instead of being treated as conversation.
if amount is None or amount.minor <= 0:
return None
occurred_on, date_span = parse_date_hint(text, today)
cuts = [amount.span]
if date_span:
cuts.append(date_span)
description = _clean_description(text, cuts)
# An explicit #tag overrides keyword matching.
tag_match = _TAG_RE.search(text)
if tag_match:
category = coerce(tag_match.group(1))
confidence = 1.0
else:
category, confidence = categorise(description)
return ExpenseDraft(
amount_minor=amount.minor,
currency=amount.currency or default_currency.upper(),
description=description or "unlabelled",
occurred_on=occurred_on,
category=category,
confidence=1.0 if confidence > 0 else 0.6,
raw=text.strip(),
)