FinSight / backend /parser.py
Sanjam19's picture
Deploy FinSight demo (single-container Docker Space)
d4f8959 verified
Raw
History Blame Contribute Delete
6.47 kB
# backend/parser.py
from backend.metrics_extractor import extract_metrics, get_spacy_nlp
from backend.sector import detect_sector
import fitz # PyMuPDF
import re
nlp = None
def get_parser_nlp():
global nlp
if nlp is not None:
return nlp
nlp = get_spacy_nlp()
return nlp
def _clean(text: str) -> str:
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r" {2,}", " ", text)
return text.strip()
def pdf_to_pages(pdf_path: str) -> list:
"""Extract text per page (1-based page numbers preserved by index+1).
HTML files count as a single 'page'. Each page is cleaned individually
so char offsets computed AFTER joining stay valid for page mapping."""
path = str(pdf_path)
# HTML file
if path.endswith(".html") or path.endswith(".htm"):
with open(path, "r", encoding="utf-8", errors="ignore") as f:
raw = f.read()
# strip HTML tags
text = re.sub(r"<[^>]+>", " ", raw)
text = re.sub(r"&nbsp;", " ", text)
text = re.sub(r"&amp;", "&", text)
text = re.sub(r"&lt;", "<", text)
text = re.sub(r"&gt;", ">", text)
return [_clean(text)]
# PDF file
doc = fitz.open(path)
pages = [_clean(page.get_text()) for page in doc]
doc.close()
return pages
def pdf_to_text(pdf_path: str) -> str:
return "\n".join(pdf_to_pages(pdf_path))
def page_offsets(pages: list) -> list:
"""Start char offset of each page in the '\\n'.join(pages) text."""
offsets = []
pos = 0
for p in pages:
offsets.append(pos)
pos += len(p) + 1 # +1 for the joining \n
return offsets
def char_pos_to_page(char_pos: int, offsets: list) -> int | None:
"""Map a char position in the joined text to a 1-based page number."""
if char_pos is None or not offsets:
return None
page = 1
for i, start in enumerate(offsets):
if char_pos >= start:
page = i + 1
else:
break
return page
def extract_entities(text: str) -> list:
parser_nlp = get_parser_nlp()
if parser_nlp is None:
print("WARNING: spaCy model 'en_core_web_sm' not installed; extracted entities will be empty.")
return []
doc = parser_nlp(text[:50000]) # cap for speed
entities = []
seen = set()
for ent in doc.ents:
if ent.label_ in {"ORG", "GPE", "MONEY", "DATE", "PRODUCT", "PERSON"}:
key = (ent.text.strip(), ent.label_)
if key not in seen:
seen.add(key)
entities.append({"text": ent.text.strip(), "label": ent.label_})
return entities
def chunk_text(text: str, size: int = 512, overlap: int = 50) -> list:
words = text.split()
chunks = []
start = 0
while start < len(words):
end = min(start + size, len(words))
chunks.append(" ".join(words[start:end]))
if end == len(words):
break
start += size - overlap
return chunks
def chunk_pages(pages: list, size: int = 512, overlap: int = 50) -> list:
"""Chunk page-by-page so every chunk carries the page it came from.
Chunks never span pages — a slight retrieval-quality trade for exact
provenance, which is the point. Most report pages are under 512 words
so this usually means one chunk per page anyway.
Returns list of {"text": str, "page": int} (page is 1-based)."""
out = []
for page_no, page_text in enumerate(pages, start=1):
if not page_text.strip():
continue
for piece in chunk_text(page_text, size=size, overlap=overlap):
out.append({"text": piece, "page": page_no})
return out
def _attach_pages(metrics: dict, offsets: list) -> dict:
"""Convert each money-metric's char_pos (and its alternatives') into a
1-based PDF page number. Ratio metrics are bare floats — passed through."""
for value in metrics.values():
if not isinstance(value, dict):
continue
value["page"] = char_pos_to_page(value.pop("char_pos", None), offsets)
for alt in value.get("alternatives", []):
if isinstance(alt, dict):
alt["page"] = char_pos_to_page(alt.pop("char_pos", None), offsets)
return metrics
def parse_document(pdf_path: str, company: str, year: str) -> dict:
print(f"Parsing {pdf_path}...")
pages = pdf_to_pages(pdf_path)
offsets = page_offsets(pages)
text = "\n".join(pages)
print(f"Text length: {len(text)} chars across {len(pages)} pages")
# ascii-safe: PDF text often contains ligatures/₹ glyphs that crash
# print() on Windows cp1252 consoles
print("Sample:", text[:200].encode("ascii", "replace").decode("ascii"))
# NOTE: company is now passed through to extract_metrics so the
# entity-boundary detection in metrics_extractor.py can demote
# numbers found inside a bundled subsidiary's section instead of
# treating every match in the document as equally valid for `company`.
metrics = _attach_pages(extract_metrics(text, company=company), offsets)
# metric dicts carry raw PDF context strings — ascii-safe for the same
# cp1252 console reason as the sample print above
print("Metrics found:", str(metrics).encode("ascii", "replace").decode("ascii"))
sector = detect_sector(text)
print(f"Sector detected: {sector}")
entities = extract_entities(text)
print(f"Entities found: {len(entities)}")
chunks = chunk_pages(pages)
return {
"company": company,
"year": year,
"file": pdf_path,
"sector": sector,
"char_count": len(text),
"chunk_count": len(chunks),
"metrics": metrics,
"entities": entities,
"chunks": [
{
"chunk_id": f"{company}_{year}_{i:04d}",
"company": company,
"year": year,
"page": c["page"],
"text": c["text"]
}
for i, c in enumerate(chunks)
]
}
if __name__ == "__main__":
sample = """
Apple Inc. reported total revenues of $394.3 billion for fiscal year 2022.
Net income was $99.8 billion. Earnings per share reached $6.15.
Total assets stood at $352.6 billion. Cash and cash equivalents were $23.6 billion.
"""
print("Metrics:", extract_metrics(sample, company="Apple"))
print("Entities:", extract_entities(sample))
print("Chunks:", len(chunk_text(sample)))