RandomZ / app /ingest /parser_docx.py
StormShadow308's picture
feat(experiment-lab): rewrite RAG pipeline using LangChain framework
49f0cfb
Raw
History Blame Contribute Delete
1.96 kB
"""Parse .docx files using the LangChain Docx2txtLoader.
LangChain replaces the manual python-docx block extraction with a
standard ``Docx2txtLoader`` that returns a list of LangChain ``Document``
objects. Structural headings/lists are no longer separated into typed
blocks; the full text is returned as a single document and split later
by the ``RecursiveCharacterTextSplitter`` in the chunking stage.
Compared to master branch:
- ✅ Simpler code — no XML introspection
- ✅ Standard LangChain interface
- ⚠️ Less structural fidelity (heading levels not preserved)
"""
import logging
from pathlib import Path
from langchain_community.document_loaders import Docx2txtLoader
from langchain_core.documents import Document
logger = logging.getLogger(__name__)
def parse_docx(file_path: Path) -> list[Document]:
"""Load a ``.docx`` file and return its text as LangChain Documents.
Uses ``langchain_community.document_loaders.Docx2txtLoader`` which
extracts plain text from all paragraphs, headings, and tables via
the ``docx2txt`` library.
Args:
file_path: Absolute path to a ``.docx`` file.
Returns:
List of :class:`langchain_core.documents.Document` objects.
Typically one document per file with ``source`` metadata.
Raises:
FileNotFoundError: If ``file_path`` does not exist.
ValueError: If the file cannot be loaded as a valid ``.docx``.
Example::
docs = parse_docx(Path("/uploads/survey.docx"))
print(docs[0].page_content[:200])
"""
if not file_path.exists():
raise FileNotFoundError(f"File not found: {file_path}")
try:
loader = Docx2txtLoader(str(file_path))
docs = loader.load()
except Exception as exc:
raise ValueError(f"Cannot load docx '{file_path}': {exc}") from exc
logger.debug("Parsed %d document(s) from %s via LangChain Docx2txtLoader", len(docs), file_path.name)
return docs