""" Frox AI — File Analysis Tool Extracts text (and light structure) from common document formats. Each format's parser is optional-imported so the tool degrades to a clear error for a specific format rather than failing to import at all if one optional dependency is missing. """ from __future__ import annotations from pathlib import Path from typing import Optional from tools.registry import tool, ToolContext MAX_CONTENT_CHARS = 20000 # keep tool results from blowing the model's context budget class UnsupportedFileType(ValueError): pass def _parse_txt(path: Path) -> str: return path.read_text(encoding="utf-8", errors="replace") def _parse_pdf(path: Path) -> str: try: import pdfplumber except ImportError: raise RuntimeError("pdfplumber is required for PDF parsing — pip install pdfplumber") pages = [] with pdfplumber.open(str(path)) as pdf: for i, page in enumerate(pdf.pages): text = page.extract_text() or "" if text.strip(): pages.append(f"[Page {i + 1}]\n{text}") return "\n\n".join(pages) def _parse_docx(path: Path) -> str: try: import docx except ImportError: raise RuntimeError("python-docx is required for DOCX parsing — pip install python-docx") doc = docx.Document(str(path)) parts = [p.text for p in doc.paragraphs if p.text.strip()] for table in doc.tables: for row in table.rows: parts.append(" | ".join(cell.text for cell in row.cells)) return "\n".join(parts) def _parse_csv(path: Path, max_rows: int = 200) -> str: import csv with open(path, newline="", encoding="utf-8", errors="replace") as f: reader = csv.reader(f) rows = list(reader) header = rows[0] if rows else [] body = rows[1:max_rows + 1] lines = [", ".join(header)] lines += [", ".join(row) for row in body] truncated_note = f"\n[... {len(rows) - 1 - len(body)} more rows not shown]" if len(rows) - 1 > len(body) else "" return "\n".join(lines) + truncated_note def _parse_xlsx(path: Path, max_rows: int = 200) -> str: try: import openpyxl except ImportError: raise RuntimeError("openpyxl is required for XLSX parsing — pip install openpyxl") wb = openpyxl.load_workbook(str(path), data_only=True, read_only=True) sheets_out = [] for sheet in wb.worksheets: lines = [f"[Sheet: {sheet.title}]"] for i, row in enumerate(sheet.iter_rows(values_only=True)): if i >= max_rows: lines.append(f"[... more rows not shown]") break lines.append(", ".join(str(c) if c is not None else "" for c in row)) sheets_out.append("\n".join(lines)) return "\n\n".join(sheets_out) def _parse_pptx(path: Path) -> str: try: from pptx import Presentation except ImportError: raise RuntimeError("python-pptx is required for PPTX parsing — pip install python-pptx") prs = Presentation(str(path)) slides_out = [] for i, slide in enumerate(prs.slides): lines = [f"[Slide {i + 1}]"] for shape in slide.shapes: if shape.has_text_frame and shape.text_frame.text.strip(): lines.append(shape.text_frame.text) if shape.has_notes_frame if hasattr(shape, "has_notes_frame") else False: lines.append(f"(notes: {shape.notes_frame.text})") slides_out.append("\n".join(lines)) return "\n\n".join(slides_out) def _parse_md(path: Path) -> str: return path.read_text(encoding="utf-8", errors="replace") _PARSERS = { ".txt": _parse_txt, ".md": _parse_md, ".pdf": _parse_pdf, ".docx": _parse_docx, ".csv": _parse_csv, ".xlsx": _parse_xlsx, ".xls": _parse_xlsx, ".pptx": _parse_pptx, } def parse_file(path: str) -> dict: """Parse a file into extracted text, dispatching by extension.""" p = Path(path) if not p.exists(): raise FileNotFoundError(f"File not found: {path}") ext = p.suffix.lower() parser = _PARSERS.get(ext) if parser is None: raise UnsupportedFileType( f"Unsupported file type: {ext!r}. Supported: {', '.join(_PARSERS)}" ) text = parser(p) truncated = len(text) > MAX_CONTENT_CHARS return { "filename": p.name, "extension": ext, "text": text[:MAX_CONTENT_CHARS], "truncated": truncated, "char_count": len(text), } @tool( name="file_analysis", description="Extract text from a document file (PDF, DOCX, TXT, MD, CSV, XLSX, PPTX)", timeout=30.0, ) def file_analysis(ctx: ToolContext, file_path: str) -> dict: """ Args: file_path: Path to the uploaded file on disk. Plain `def`, not `async def`: file I/O and the parsing libraries here (pdfplumber, python-docx, openpyxl) are all blocking, so this is thread-offloaded by the registry. """ return parse_file(file_path)