stock_report_robust / classifier.py
iammraat's picture
Update classifier.py
669e51b verified
Raw
History Blame Contribute Delete
3.04 kB
import pandas as pd
import pdfplumber
def classify_pdf(pdf_path: str) -> str:
"""
Peeks at the first page of the PDF to determine its origin.
Returns 'MARG', 'TALLY', or 'UNKNOWN'.
"""
with pdfplumber.open(pdf_path) as pdf:
if len(pdf.pages) == 0:
return "UNKNOWN"
first_page = pdf.pages[0]
words = first_page.extract_words(keep_blank_chars=False)
if not words:
return "UNKNOWN"
text_stream = " ".join([w["text"].lower() for w in words])
# --- 1. Check for Explicit Metadata/Branding ---
if "marg erp" in text_stream or "margcompusoft" in text_stream:
return "MARG"
if "tally prime" in text_stream or "tally solutions" in text_stream:
return "TALLY"
# --- 2. Check Structural Vocabulary (Heuristics) ---
marg_score = 0
tally_score = 0
# Marg Keywords (Strictly Marg-specific financial headers)
# Removed "batch", "exp", and "hsn" as they are shared with Tally
marg_keywords = ["s.no.", "mrp", "p.rate", "total stock", "op.stock", "cl.stock"]
for kw in marg_keywords:
if kw in text_stream:
marg_score += 1
# Tally Keywords (Hierarchical accounting)
# Added "godown" which is highly specific to Tally
tally_keywords = ["particulars", "opening balance", "inwards", "outwards", "closing balance", "grand total",
"godown"]
for kw in tally_keywords:
if kw in text_stream:
tally_score += 1
# Look for Marg's distinct dashed lines
dashed_lines = sum(1 for w in words if "----" in w["text"] or "====" in w["text"])
if dashed_lines > 3:
marg_score += 2
# Check for explicit Tally report titles combined with Particulars
if "stock summary" in text_stream and "particulars" in text_stream:
tally_score += 2
# --- 3. Final Decision ---
if marg_score > tally_score and marg_score >= 1:
return "MARG"
elif tally_score > marg_score and tally_score >= 1:
return "TALLY"
# Check for explicit drawn tables (Vanilla Grids)
if len(first_page.find_tables()) > 0:
return "GENERIC"
return "UNKNOWN"
def classify_dataframe(df: pd.DataFrame) -> str:
"""Scans the first 15 rows of a dataframe for Tally/Marg signatures."""
# Convert the top portion to a single lowercase string for easy keyword matching
top_rows_text = df.head(15).to_string().lower()
if "marg erp" in top_rows_text or "margcompusoft" in top_rows_text:
return "MARG"
if "tally prime" in top_rows_text or "tally solutions" in top_rows_text:
return "TALLY"
# Heuristics
if "particulars" in top_rows_text and "godown" in top_rows_text:
return "TALLY"
if "s.no" in top_rows_text and "mrp" in top_rows_text and "p.rate" in top_rows_text:
return "MARG"
return "GENERIC"