ai-interview-caller / src /resume_rag.py
ranjithkumar111's picture
Upload 8 files
b253d7e verified
Raw
History Blame Contribute Delete
2.89 kB
# File: src/resume_rag.py
# Purpose: Resume chunking, embedding (sentence-transformers), ChromaDB indexing,
# and skill extraction via Groq (no OpenAI required)
import json
import re
from typing import List
import chromadb
from chromadb.utils import embedding_functions
from groq import Groq
from config import (
GROQ_API_KEY, GROQ_LLM_MODEL,
EMBEDDING_MODEL, CHROMA_PERSIST_DIR,
CHUNK_SIZE, CHUNK_OVERLAP, TOP_K_RESUME_CHUNKS,
)
# Local sentence-transformers embeddings — no API key needed
_st_ef = embedding_functions.SentenceTransformerEmbeddingFunction(
model_name=EMBEDDING_MODEL
)
# Groq client
_groq = Groq(api_key=GROQ_API_KEY)
def _chunk_text(text: str, size: int = CHUNK_SIZE, overlap: int = CHUNK_OVERLAP) -> List[str]:
words = text.split()
chunks, i = [], 0
while i < len(words):
chunks.append(" ".join(words[i: i + size]))
i += size - overlap
return chunks
def build_resume_index(session_id: int, resume_text: str) -> chromadb.Collection:
client = chromadb.PersistentClient(path=CHROMA_PERSIST_DIR)
collection_name = f"resume_{session_id}"
try:
client.delete_collection(collection_name)
except Exception:
pass
collection = client.create_collection(
name=collection_name,
embedding_function=_st_ef,
)
chunks = _chunk_text(resume_text)
collection.add(
documents=chunks,
ids=[f"chunk_{i}" for i in range(len(chunks))],
)
return collection
def retrieve_resume_context(session_id: int, query: str, top_k: int = TOP_K_RESUME_CHUNKS) -> str:
client = chromadb.PersistentClient(path=CHROMA_PERSIST_DIR)
collection_name = f"resume_{session_id}"
try:
collection = client.get_collection(
name=collection_name,
embedding_function=_st_ef,
)
except Exception:
return ""
results = collection.query(
query_texts=[query],
n_results=min(top_k, collection.count()),
)
docs = results.get("documents", [[]])[0]
return "\n".join(docs)
def extract_skills_from_resume(resume_text: str) -> List[str]:
prompt = (
"Extract a concise list of technical skills, tools, and frameworks mentioned in the "
"following resume. Return ONLY a valid JSON array of strings. "
"No explanation, no markdown, no extra text.\n\n"
f"Resume:\n{resume_text[:3000]}"
)
response = _groq.chat.completions.create(
model=GROQ_LLM_MODEL,
messages=[{"role": "user", "content": prompt}],
temperature=0,
max_tokens=400,
)
raw = response.choices[0].message.content.strip()
raw = re.sub(r"```json|```", "", raw).strip()
try:
return json.loads(raw)
except Exception:
return []