Spaces:
Running
Running
File size: 6,386 Bytes
2ecc4a7 0e82360 d61e4dd 0e82360 2ecc4a7 d61e4dd 0e82360 d61e4dd 0e82360 d61e4dd 0e82360 8d465e5 0e82360 8d465e5 0e82360 8d465e5 0e82360 2ecc4a7 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 | import json
import re
from pathlib import Path
from typing import List, Dict, Any
from docx import Document
from .ocr_service import ocr_service
from .vision_service import vision_service
import logging
logger = logging.getLogger(__name__)
async def parse_file(file_path: str, file_type: str, job_id: str, ws_manager: Any, user_id: str) -> List[Dict[str, Any]]:
"""
Dispatcher for all file types.
Returns: [{"text": str, "metadata": {"source", "page", "section"}}]
"""
match file_type.lower():
case "pdf":
return await _parse_pdf(file_path, job_id, ws_manager, user_id)
case "jpg" | "jpeg" | "png":
return await _parse_image(file_path, job_id, ws_manager, user_id)
case "docx":
return _parse_docx(file_path)
case "txt":
return _parse_txt(file_path)
case "md":
return _parse_markdown(file_path)
case "json":
return _parse_json(file_path)
case _:
logger.warning(f"Unsupported file type: {file_type}")
return []
async def _parse_pdf(file_path: str, job_id: str, ws_manager: Any, user_id: str):
try:
await ws_manager.emit(job_id, user_id, {"step": "OCR_START", "color": "#8B5CF6", "detail": "Sending to Mistral OCR (Primary)..."})
results = await ocr_service.perform_ocr(file_path)
return [{"text": results['plain_text'], "metadata": {"source": Path(file_path).name, "page": 1}}]
except Exception as e:
logger.warning(f"Mistral OCR failed, falling back to PyMuPDF: {e}")
await ws_manager.emit(job_id, user_id, {"step": "FALLBACK", "color": "#F59E0B", "detail": "Mistral failed. Falling back to PyMuPDF..."})
import fitz # PyMuPDF
doc = fitz.open(file_path)
text = ""
for page in doc:
text += page.get_text()
return [{"text": text, "metadata": {"source": Path(file_path).name, "page": 1}}]
async def _parse_image(file_path: str, job_id: str, ws_manager: Any, user_id: str):
filename = Path(file_path).name
extracted_content = []
# 1. Tesseract OCR (Fast, Reliable Local OCR with grayscale preprocessing)
ocr_text = ""
try:
import pytesseract
from PIL import Image, ImageOps
img = Image.open(file_path)
# Convert to grayscale and enhance contrast for reliable OCR
gray_img = ImageOps.grayscale(img)
if gray_img.width < 1000:
gray_img = gray_img.resize((gray_img.width * 2, gray_img.height * 2), Image.Resampling.BILINEAR)
ocr_text = pytesseract.image_to_string(gray_img).strip()
if not ocr_text:
ocr_text = pytesseract.image_to_string(img).strip()
if ocr_text:
logger.info(f"OCR extracted {len(ocr_text)} characters from {filename}")
extracted_content.append(f"Visual Text Content Extracted via OCR:\n{ocr_text}")
except Exception as e:
logger.warning(f"Tesseract OCR failed: {e}")
# 2. Vision Space Analysis (with non-blocking 4s timeout)
try:
await ws_manager.emit(job_id, user_id, {"step": "IMAGE_ANALYZE", "color": "#8B5CF6", "detail": "Analyzing visual image contents..."})
import asyncio
description = await asyncio.wait_for(
asyncio.to_thread(vision_service.understand_image, file_path),
timeout=4.0
)
if description and "failed to understand" not in description.lower():
extracted_content.append(f"Visual Scene Description:\n{description}")
except Exception as e:
logger.warning(f"Vision space analysis skipped/timed out: {e}")
# 3. Fallback to image metadata if no text or vision
if not extracted_content:
try:
from PIL import Image
img = Image.open(file_path)
extracted_content.append(f"Image Document: {filename}\nResolution: {img.width}x{img.height}\nFormat: {img.format}\nMode: {img.mode}\nStatus: Visual image document indexed in knowledge base.")
except Exception:
extracted_content.append(f"Image Document: {filename}\nStatus: Visual image document indexed in knowledge base.")
final_text = "\n\n".join(extracted_content)
return [{"text": final_text, "metadata": {"source": filename, "page": 1}}]
def _parse_docx(file_path: str):
doc = Document(file_path)
sections, current_heading, current_text = [], "General", []
for para in doc.paragraphs:
if para.style.name.startswith('Heading'):
if current_text:
sections.append({"text": "\n".join(current_text), "metadata": {"source": Path(file_path).name, "section": current_heading}})
current_heading, current_text = para.text, []
elif para.text.strip():
current_text.append(para.text)
if current_text:
sections.append({"text": "\n".join(current_text), "metadata": {"source": Path(file_path).name, "section": current_heading}})
return sections
def _parse_markdown(file_path: str):
text = Path(file_path).read_text(encoding="utf-8")
parts = re.split(r'\n(?=#+\s)', text)
docs = []
for p in parts:
if not p.strip(): continue
match = re.match(r'^#+\s+(.*)', p)
section = match.group(1) if match else "General"
docs.append({"text": p.strip(), "metadata": {"source": Path(file_path).name, "section": section}})
return docs
def _parse_txt(file_path: str):
text = Path(file_path).read_text(encoding="utf-8")
paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()]
return [{"text": p, "metadata": {"source": Path(file_path).name}} for p in paragraphs]
def _parse_json(file_path: str):
data = json.loads(Path(file_path).read_text())
docs = []
# If it's a list, treat each item as a doc
if isinstance(data, list):
items = data
# If it's a dict, treat each top-level key-value pair as a doc
elif isinstance(data, dict):
items = [{"key": k, "value": v} for k, v in data.items()]
else:
items = [data]
for item in items:
if isinstance(item, (dict, list)):
text = json.dumps(item, indent=2)
else:
text = str(item)
if text.strip():
docs.append({"text": text, "metadata": {"source": Path(file_path).name}})
return docs
|