Spaces:
Running on Zero
Running on Zero
Download finbot/parser.py from spacedout-bits/Oracle: direct link, hf CLI and curl.
- Browser
- Download file 6.82 kB
-
https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/finbot/parser.py
- Command line
-
hf download hf://spaces/spacedout-bits/Oracle/finbot/parser.py
-
curl -L -o parser.py https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/finbot/parser.py
6.82 kB
| """Deterministic parsing of free-text expense entry. | |
| This is the fast path: "coffee 250", "₹1,200 groceries at bigbasket", | |
| "spent 45.50 on lunch yesterday". It runs before any model call, costs | |
| nothing, and is reproducible. Anything it cannot make sense of is handed to | |
| the LLM by :mod:`finbot.ingest`. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from datetime import date, timedelta | |
| from typing import Optional | |
| from .categories import categorise, coerce | |
| from .models import ExpenseDraft | |
| from .money import parse_amount | |
| _WEEKDAYS = { | |
| "monday": 0, "mon": 0, | |
| "tuesday": 1, "tue": 1, "tues": 1, | |
| "wednesday": 2, "wed": 2, | |
| "thursday": 3, "thu": 3, "thurs": 3, | |
| "friday": 4, "fri": 4, | |
| "saturday": 5, "sat": 5, | |
| "sunday": 6, "sun": 6, | |
| } | |
| _MONTHS = { | |
| "jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3, | |
| "apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7, | |
| "aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9, | |
| "oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12, | |
| } | |
| # Words that carry no meaning in a description once the amount is removed. | |
| _FILLER = { | |
| "spent", "spend", "paid", "pay", "bought", "buy", "for", "on", "at", "in", | |
| "of", "the", "a", "an", "to", "rs", "rs.", "inr", "cost", "costs", "was", | |
| "were", "i", "my", "me", "some", "and", "with", "got", | |
| } | |
| _ISO_DATE_RE = re.compile(r"\b(\d{4})-(\d{1,2})-(\d{1,2})\b") | |
| _DMY_RE = re.compile(r"\b(\d{1,2})[/-](\d{1,2})(?:[/-](\d{2,4}))?\b") | |
| _DAY_MONTH_RE = re.compile( | |
| r"\b(\d{1,2})(?:st|nd|rd|th)?\s+(" + "|".join(_MONTHS) + r")\b", re.I | |
| ) | |
| _MONTH_DAY_RE = re.compile( | |
| r"\b(" + "|".join(_MONTHS) + r")\s+(\d{1,2})(?:st|nd|rd|th)?\b", re.I | |
| ) | |
| _DAYS_AGO_RE = re.compile(r"\b(\d{1,3})\s+days?\s+ago\b", re.I) | |
| _LAST_WEEKDAY_RE = re.compile( | |
| r"\b(?:last\s+)?(" + "|".join(_WEEKDAYS) + r")\b", re.I | |
| ) | |
| _TAG_RE = re.compile(r"#([a-z_][a-z0-9_]*)", re.I) | |
| def _clamp_year(year: int) -> int: | |
| if year < 100: | |
| return 2000 + year | |
| return year | |
| def parse_date_hint( | |
| text: str, today: Optional[date] = None | |
| ) -> tuple[date, Optional[tuple[int, int]]]: | |
| """Resolve a date reference in ``text``. | |
| Returns (resolved_date, span_of_the_phrase). Span is None when no hint was | |
| found, in which case the date defaults to today. Dates that would land in | |
| the future are pulled back a year -- "12 dec" typed in August means last | |
| December, not a spend that has not happened yet. | |
| """ | |
| today = today or date.today() | |
| lowered = text.lower() | |
| if m := re.search(r"\bday before yesterday\b", lowered): | |
| return today - timedelta(days=2), m.span() | |
| if m := re.search(r"\byesterday\b", lowered): | |
| return today - timedelta(days=1), m.span() | |
| if m := re.search(r"\btoday\b", lowered): | |
| return today, m.span() | |
| if m := _DAYS_AGO_RE.search(lowered): | |
| return today - timedelta(days=int(m.group(1))), m.span() | |
| if m := _ISO_DATE_RE.search(lowered): | |
| try: | |
| return date(int(m.group(1)), int(m.group(2)), int(m.group(3))), m.span() | |
| except ValueError: | |
| pass | |
| if m := _DAY_MONTH_RE.search(lowered): | |
| day, month = int(m.group(1)), _MONTHS[m.group(2).lower()] | |
| try: | |
| candidate = date(today.year, month, day) | |
| if candidate > today: | |
| candidate = date(today.year - 1, month, day) | |
| return candidate, m.span() | |
| except ValueError: | |
| pass | |
| if m := _MONTH_DAY_RE.search(lowered): | |
| month, day = _MONTHS[m.group(1).lower()], int(m.group(2)) | |
| try: | |
| candidate = date(today.year, month, day) | |
| if candidate > today: | |
| candidate = date(today.year - 1, month, day) | |
| return candidate, m.span() | |
| except ValueError: | |
| pass | |
| if m := _DMY_RE.search(lowered): | |
| first, second, year_s = int(m.group(1)), int(m.group(2)), m.group(3) | |
| year = _clamp_year(int(year_s)) if year_s else today.year | |
| # Ambiguous d/m vs m/d. Prefer day-first, which covers most of the | |
| # world, and fall back to month-first when day-first is impossible. | |
| for day, month in ((first, second), (second, first)): | |
| try: | |
| candidate = date(year, month, day) | |
| except ValueError: | |
| continue | |
| if not year_s and candidate > today: | |
| candidate = date(year - 1, month, day) | |
| return candidate, m.span() | |
| if m := _LAST_WEEKDAY_RE.search(lowered): | |
| target = _WEEKDAYS[m.group(1).lower()] | |
| delta = (today.weekday() - target) % 7 | |
| delta = delta or 7 # bare weekday name means the most recent past one | |
| return today - timedelta(days=delta), m.span() | |
| return today, None | |
| def _clean_description(text: str, cuts: list[tuple[int, int]]) -> str: | |
| """Remove consumed spans, currency noise and filler words.""" | |
| chars = list(text) | |
| for start, end in cuts: | |
| for i in range(start, min(end, len(chars))): | |
| chars[i] = " " | |
| remaining = "".join(chars) | |
| remaining = re.sub(r"[₹$€£¥₽₩₪₫₺₦₴₸฿₡₱﷼]", " ", remaining) | |
| remaining = re.sub(r"\b[A-Z]{3}\b", " ", remaining) | |
| remaining = _TAG_RE.sub(" ", remaining) | |
| remaining = re.sub(r"[^\w\s&'-]", " ", remaining) | |
| words = [w for w in remaining.split() if w.lower() not in _FILLER] | |
| return re.sub(r"\s+", " ", " ".join(words)).strip() | |
| def parse_expense( | |
| text: str, default_currency: str = "INR", today: Optional[date] = None | |
| ) -> Optional[ExpenseDraft]: | |
| """Parse one free-text line into a draft, or None if no amount is present.""" | |
| if not text or not text.strip(): | |
| return None | |
| amount = parse_amount(text, default_currency) | |
| # A zero amount is not an expense. Without this, ordinary sentences that | |
| # happen to contain a 0 ("question 0", "flight AI 0") get logged as ₹0.00 | |
| # entries instead of being treated as conversation. | |
| if amount is None or amount.minor <= 0: | |
| return None | |
| occurred_on, date_span = parse_date_hint(text, today) | |
| cuts = [amount.span] | |
| if date_span: | |
| cuts.append(date_span) | |
| description = _clean_description(text, cuts) | |
| # An explicit #tag overrides keyword matching. | |
| tag_match = _TAG_RE.search(text) | |
| if tag_match: | |
| category = coerce(tag_match.group(1)) | |
| confidence = 1.0 | |
| else: | |
| category, confidence = categorise(description) | |
| return ExpenseDraft( | |
| amount_minor=amount.minor, | |
| currency=amount.currency or default_currency.upper(), | |
| description=description or "unlabelled", | |
| occurred_on=occurred_on, | |
| category=category, | |
| confidence=1.0 if confidence > 0 else 0.6, | |
| raw=text.strip(), | |
| ) | |