study-buddy / app /rag /ingestion.py
GitHub Actions
deploy d092bea3608b7a29952f16357fda39b7a29e399b
2e818da
Raw
History Blame Contribute Delete
2.99 kB
"""Compatibility entry points for structured academic-evidence ingestion.
Callers still use these functions while ingestion itself is owned by
``EvidenceIngestionService``. No fixed-width chunks are created here.
"""
import io
from typing import Literal, Optional
from app.rag.chromadb_client import ChromaDBClient
from app.rag.evidence_ingestion import EvidenceIngestionService
from app.rag.vector_indexes import PAPER_EVIDENCE_COLLECTION
ChunkType = Literal["content", "question"]
# Transitional name imported by existing retrieval call sites. They now point
# at the structured dense index rather than the retired flat ``library`` index.
LIBRARY_COLLECTION = PAPER_EVIDENCE_COLLECTION
def extract_document_structure(file_bytes: bytes, filename: str, max_chars: int = 6000) -> str:
"""Extract only the structural overview of a document for instant tree generation.
Returns headings/TOC and first-page text -> enough for Gemma 4 to derive a
curriculum tree without needing full chunking.
"""
ext = filename.rsplit(".", 1)[-1].lower()
if ext == "pdf":
from pypdf import PdfReader
reader = PdfReader(io.BytesIO(file_bytes))
# First 4 pages capture TOC + intro for most textbooks
pages = [reader.pages[i].extract_text() or "" for i in range(min(4, len(reader.pages)))]
return "\n".join(pages)[:max_chars]
elif ext == "docx":
import docx
doc = docx.Document(io.BytesIO(file_bytes))
# Only heading paragraphs for structure
headings = [
p.text for p in doc.paragraphs
if p.style and p.style.name and p.style.name.startswith("Heading")
]
if headings:
return "\n".join(headings)[:max_chars]
# Fall back to first 6000 chars of plain text
return "\n".join(p.text for p in doc.paragraphs)[:max_chars]
else:
return file_bytes.decode("utf-8", errors="replace")[:max_chars]
def ingest_text(
text: str,
source_label: str,
collection: str = LIBRARY_COLLECTION,
chunk_type: ChunkType = "content",
db: Optional[ChromaDBClient] = None,
project_id: str = "",
document_id: str = "",
) -> int:
del collection, chunk_type
service = EvidenceIngestionService(db=db or ChromaDBClient())
result = service.ingest_text(
text,
source_label,
project_id,
document_id=document_id or None,
)
return result.indexed_unit_count
def ingest_file(
file_bytes: bytes,
filename: str,
collection: str = LIBRARY_COLLECTION,
chunk_type: ChunkType = "content",
db: Optional[ChromaDBClient] = None,
skip_if_indexed: bool = True,
project_id: str = "",
) -> int:
"""Parse and index a file as typed, page-grounded evidence units."""
del collection, chunk_type, skip_if_indexed
service = EvidenceIngestionService(db=db or ChromaDBClient())
return service.ingest_file(file_bytes, filename, project_id).indexed_unit_count