File size: 6,386 Bytes
2ecc4a7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0e82360
 
 
d61e4dd
0e82360
2ecc4a7
 
d61e4dd
0e82360
d61e4dd
 
 
 
 
 
 
0e82360
d61e4dd
0e82360
 
 
 
8d465e5
0e82360
 
8d465e5
 
 
 
 
0e82360
 
 
8d465e5
0e82360
 
 
 
 
 
 
 
 
 
 
 
2ecc4a7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
import json
import re
from pathlib import Path
from typing import List, Dict, Any
from docx import Document
from .ocr_service import ocr_service
from .vision_service import vision_service
import logging

logger = logging.getLogger(__name__)

async def parse_file(file_path: str, file_type: str, job_id: str, ws_manager: Any, user_id: str) -> List[Dict[str, Any]]:
    """
    Dispatcher for all file types.
    Returns: [{"text": str, "metadata": {"source", "page", "section"}}]
    """
    match file_type.lower():
        case "pdf":
            return await _parse_pdf(file_path, job_id, ws_manager, user_id)
        case "jpg" | "jpeg" | "png":
            return await _parse_image(file_path, job_id, ws_manager, user_id)
        case "docx":
            return _parse_docx(file_path)
        case "txt":
            return _parse_txt(file_path)
        case "md":
            return _parse_markdown(file_path)
        case "json":
            return _parse_json(file_path)
        case _:
            logger.warning(f"Unsupported file type: {file_type}")
            return []

async def _parse_pdf(file_path: str, job_id: str, ws_manager: Any, user_id: str):
    try:
        await ws_manager.emit(job_id, user_id, {"step": "OCR_START", "color": "#8B5CF6", "detail": "Sending to Mistral OCR (Primary)..."})
        results = await ocr_service.perform_ocr(file_path)
        return [{"text": results['plain_text'], "metadata": {"source": Path(file_path).name, "page": 1}}]
    except Exception as e:
        logger.warning(f"Mistral OCR failed, falling back to PyMuPDF: {e}")
        await ws_manager.emit(job_id, user_id, {"step": "FALLBACK", "color": "#F59E0B", "detail": "Mistral failed. Falling back to PyMuPDF..."})
        import fitz  # PyMuPDF
        doc = fitz.open(file_path)
        text = ""
        for page in doc:
            text += page.get_text()
        return [{"text": text, "metadata": {"source": Path(file_path).name, "page": 1}}]

async def _parse_image(file_path: str, job_id: str, ws_manager: Any, user_id: str):
    filename = Path(file_path).name
    extracted_content = []
    
    # 1. Tesseract OCR (Fast, Reliable Local OCR with grayscale preprocessing)
    ocr_text = ""
    try:
        import pytesseract
        from PIL import Image, ImageOps
        img = Image.open(file_path)
        # Convert to grayscale and enhance contrast for reliable OCR
        gray_img = ImageOps.grayscale(img)
        if gray_img.width < 1000:
            gray_img = gray_img.resize((gray_img.width * 2, gray_img.height * 2), Image.Resampling.BILINEAR)
        ocr_text = pytesseract.image_to_string(gray_img).strip()
        if not ocr_text:
            ocr_text = pytesseract.image_to_string(img).strip()
        if ocr_text:
            logger.info(f"OCR extracted {len(ocr_text)} characters from {filename}")
            extracted_content.append(f"Visual Text Content Extracted via OCR:\n{ocr_text}")
    except Exception as e:
        logger.warning(f"Tesseract OCR failed: {e}")

    # 2. Vision Space Analysis (with non-blocking 4s timeout)
    try:
        await ws_manager.emit(job_id, user_id, {"step": "IMAGE_ANALYZE", "color": "#8B5CF6", "detail": "Analyzing visual image contents..."})
        import asyncio
        description = await asyncio.wait_for(
            asyncio.to_thread(vision_service.understand_image, file_path),
            timeout=4.0
        )
        if description and "failed to understand" not in description.lower():
            extracted_content.append(f"Visual Scene Description:\n{description}")
    except Exception as e:
        logger.warning(f"Vision space analysis skipped/timed out: {e}")

    # 3. Fallback to image metadata if no text or vision
    if not extracted_content:
        try:
            from PIL import Image
            img = Image.open(file_path)
            extracted_content.append(f"Image Document: {filename}\nResolution: {img.width}x{img.height}\nFormat: {img.format}\nMode: {img.mode}\nStatus: Visual image document indexed in knowledge base.")
        except Exception:
            extracted_content.append(f"Image Document: {filename}\nStatus: Visual image document indexed in knowledge base.")

    final_text = "\n\n".join(extracted_content)
    return [{"text": final_text, "metadata": {"source": filename, "page": 1}}]

def _parse_docx(file_path: str):
    doc = Document(file_path)
    sections, current_heading, current_text = [], "General", []
    for para in doc.paragraphs:
        if para.style.name.startswith('Heading'):
            if current_text:
                sections.append({"text": "\n".join(current_text), "metadata": {"source": Path(file_path).name, "section": current_heading}})
            current_heading, current_text = para.text, []
        elif para.text.strip():
            current_text.append(para.text)
    if current_text:
        sections.append({"text": "\n".join(current_text), "metadata": {"source": Path(file_path).name, "section": current_heading}})
    return sections

def _parse_markdown(file_path: str):
    text = Path(file_path).read_text(encoding="utf-8")
    parts = re.split(r'\n(?=#+\s)', text)
    docs = []
    for p in parts:
        if not p.strip(): continue
        match = re.match(r'^#+\s+(.*)', p)
        section = match.group(1) if match else "General"
        docs.append({"text": p.strip(), "metadata": {"source": Path(file_path).name, "section": section}})
    return docs

def _parse_txt(file_path: str):
    text = Path(file_path).read_text(encoding="utf-8")
    paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()]
    return [{"text": p, "metadata": {"source": Path(file_path).name}} for p in paragraphs]

def _parse_json(file_path: str):
    data = json.loads(Path(file_path).read_text())
    docs = []
    
    # If it's a list, treat each item as a doc
    if isinstance(data, list):
        items = data
    # If it's a dict, treat each top-level key-value pair as a doc
    elif isinstance(data, dict):
        items = [{"key": k, "value": v} for k, v in data.items()]
    else:
        items = [data]

    for item in items:
        if isinstance(item, (dict, list)):
            text = json.dumps(item, indent=2)
        else:
            text = str(item)
            
        if text.strip():
            docs.append({"text": text, "metadata": {"source": Path(file_path).name}})
    
    return docs