quantforge-miner / utils /ingest_docs.py
DevWizard-Vandan
Harden alpha grammar and auth-aware research
599e860
Raw
History Blame Contribute Delete
3.98 kB
import os
import logging
from pathlib import Path
from typing import List
logger = logging.getLogger("DocIngestion")
def extract_pdf_text(pdf_path: Path) -> str:
"""
Safely extracts text from a PDF file using pypdf.
If pypdf is not installed, logs a warning and returns an empty string.
"""
try:
import pypdf
reader = pypdf.PdfReader(str(pdf_path))
text_parts = []
for i, page in enumerate(reader.pages):
page_text = page.extract_text()
if page_text:
text_parts.append(page_text)
return "\n".join(text_parts)
except ImportError:
logger.warning(
f"pypdf is not installed. Cannot extract text from PDF: {pdf_path.name}. "
"Please run 'pip install pypdf' to enable PDF ingestion."
)
return ""
except Exception as e:
logger.error(f"Error reading PDF file {pdf_path.name}: {e}", exc_info=True)
return ""
def compile_knowledge_context(directories: List[str] = ["docs", "knowledge"]) -> str:
"""
Scans the specified directories relative to the project root for PDF, Markdown,
and text files. Combines their contents into a single structured string block.
Target files are related to WorldQuant BRAIN documentation:
- Double Neutralization
- Fast D1 Documentation
- Sentiment1
- Model77
- earnings4
Returns:
A compiled string of all matching documents, formatted with markdown headers.
"""
project_root = Path(__file__).resolve().parent.parent
compiled_parts = []
for dir_name in directories:
target_dir = project_root / dir_name
if not target_dir.exists():
logger.debug(f"Directory {dir_name} does not exist at {target_dir}")
continue
logger.info(f"Ingesting documents from directory: {target_dir}")
# Search recursively or directly
for file_path in target_dir.rglob("*"):
if file_path.is_dir():
continue
suffix = file_path.suffix.lower()
if suffix not in (".md", ".txt", ".pdf"):
continue
logger.info(f"Processing doc: {file_path.name}")
content = ""
try:
if suffix in (".md", ".txt"):
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
content = f.read()
elif suffix == ".pdf":
content = extract_pdf_text(file_path)
except Exception as e:
logger.error(f"Failed to read file {file_path.name}: {e}")
continue
if content.strip():
# Keep the prompt bounded when consultant PDFs are present.
# Preserve both the opening definitions and closing tests;
# otherwise a full multi-megabyte corpus can crowd out the
# actual generation constraints sent to the model.
content = content.strip()
max_chars = 40000
if len(content) > max_chars:
head = 30000
content = (
content[:head]
+ "\n\n[DOCUMENT CONTENT TRUNCATED FOR PROMPT BUDGET]\n\n"
+ content[-(max_chars - head):]
)
logger.info("Truncated document %s to %s characters for model context.", file_path.name, max_chars)
doc_name = file_path.stem.replace("_", " ").title()
compiled_parts.append(
f"### DOCUMENT: {doc_name} ({file_path.name})\n\n{content.strip()}\n\n"
)
if not compiled_parts:
logger.warning("No knowledge documents were found or successfully ingested.")
return "No local platform documentation available in docs/ or knowledge/ directories."
return "\n".join(compiled_parts)