MaduRox's picture
Upload folder using huggingface_hub
e3584eb verified
Raw
History Blame Contribute Delete
3.04 kB
import os
import json
import csv
from typing import Dict, Any
# Safe imports for optional file libraries
try:
import pypdf
PYPDF_AVAILABLE = True
except ImportError:
PYPDF_AVAILABLE = False
try:
import docx
DOCX_AVAILABLE = True
except ImportError:
DOCX_AVAILABLE = False
def extract_text_from_file(file_path: str) -> str:
"""
Reads the file path and returns extracted text content based on file extension.
"""
if not os.path.exists(file_path):
raise FileNotFoundError(f"File not found at: {file_path}")
_, ext = os.path.splitext(file_path.lower())
if ext == ".txt" or ext == ".md":
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
return f.read()
elif ext == ".csv":
extracted = []
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
reader = csv.reader(f)
for row in reader:
extracted.append(", ".join(row))
return "\n".join(extracted)
elif ext in [".json", ".jsonl"]:
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
try:
if ext == ".json":
data = json.load(f)
return json.dumps(data, indent=2)
else: # jsonl
lines = []
for line in f:
if line.strip():
lines.append(json.dumps(json.loads(line)))
return "\n".join(lines)
except Exception as e:
raise ValueError(f"Invalid JSON/JSONL format: {str(e)}")
elif ext == ".pdf":
if not PYPDF_AVAILABLE:
raise ImportError("PDF extraction requires the 'pypdf' package. Please install it.")
text_content = []
with open(file_path, "rb") as f:
reader = pypdf.PdfReader(f)
num_pages = len(reader.pages)
for i in range(num_pages):
page_text = reader.pages[i].extract_text()
if page_text:
text_content.append(page_text)
full_text = "\n".join(text_content).strip()
if not full_text:
raise ValueError("PDF text extraction returned empty text. This file appears to be image-only (scanned). OCR is not supported.")
return full_text
elif ext == ".docx":
if not DOCX_AVAILABLE:
raise ImportError("Word document extraction requires the 'python-docx' package. Please install it.")
doc = docx.Document(file_path)
paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
full_text = "\n".join(paragraphs).strip()
if not full_text:
raise ValueError("Word document is empty or has no readable text paragraphs.")
return full_text
else:
raise ValueError(f"Unsupported file format extension: {ext}")