amplegest / ingestion /embedder.py
Viney's picture
feat: add company primer research tab
d1e793b
Raw
History Blame Contribute Delete
3.64 kB
from storage import vector_store
def chunk_text(text: str, chunk_size: int = 500, overlap: int = 50) -> list[str]:
words = text.split()
if len(words) <= chunk_size:
return [text]
chunks, start = [], 0
while start < len(words):
end = min(start + chunk_size, len(words))
chunks.append(" ".join(words[start:end]))
start += chunk_size - overlap
return chunks
def clear_ticker_data(ticker: str) -> None:
"""Delete all existing chunks for a ticker from both Chroma collections."""
vector_store.delete_by_ticker("filings", ticker)
vector_store.delete_by_ticker("transcripts", ticker)
def embed_and_store_filing(
ticker: str,
company_name: str,
mda_text: str,
risk_text: str,
filing_date: str,
period: str,
form_type: str,
accession: str = "",
source_url: str = "",
document_id: str = "",
business_text: str = "",
segments_geography_text: str = "",
) -> None:
ticker = ticker.upper()
filing_document_id = document_id or f"sec:{ticker}:{accession or f'{form_type}:{period}:{filing_date}'}"
sections = [
("MD&A", "mda", mda_text),
("Risk Factors", "risk_factors", risk_text),
("Business", "business", business_text),
("Segments & Geography", "segments_geography", segments_geography_text),
]
for section, section_slug, text in sections:
if not text.strip():
print(f"WARNING: empty section '{section}' for {ticker} {period} ({form_type}) — Chroma chunk not written")
continue
chunks = chunk_text(text)
total = len(chunks)
chunk_ids = [f"{section_slug}:{i}" for i in range(total)]
metadatas = [
{
"ticker": ticker,
"company_name": company_name,
"source": form_type,
"filing_date": filing_date,
"period": period,
"section": section,
"accession": accession,
"source_url": source_url,
"document_id": filing_document_id,
"chunk_id": chunk_ids[i],
"chunk_context": f"{company_name} | {form_type} {period} | {section} | Chunk {i + 1}/{total}",
}
for i in range(total)
]
ids = [f"{ticker}-{form_type}-{period}-{section.replace(' ', '_')}-{i}" for i in range(total)]
vector_store.add_chunks("filings", chunks, metadatas, ids)
def embed_and_store_transcript(
ticker: str,
company_name: str,
transcript_text: str,
transcript_date: str,
period: str,
source_url: str = "",
document_id: str = "",
provider: str = "alphavantage",
) -> None:
ticker = ticker.upper()
if not transcript_text.strip():
return
chunks = chunk_text(transcript_text)
total = len(chunks)
transcript_document_id = document_id or f"transcript:{ticker}:{period}:{transcript_date}"
metadatas = [
{
"ticker": ticker,
"company_name": company_name,
"source": "transcript",
"date": transcript_date,
"period": period,
"speaker": "mixed",
"source_url": source_url,
"provider": provider,
"document_id": transcript_document_id,
"chunk_id": f"transcript:{i}",
"chunk_context": f"{company_name} | Earnings Call {period} | Chunk {i + 1}/{total}",
}
for i in range(total)
]
ids = [f"{ticker}-transcript-{period}-{i}" for i in range(total)]
vector_store.add_chunks("transcripts", chunks, metadatas, ids)