"""SLM generation with transformers — ZeroGPU-compatible (fully lazy).""" from typing import List, Dict from config import LLM_MODEL_ID, LLM_MAX_NEW_TOKENS, LLM_TEMPERATURE, LLM_DO_SAMPLE class AethronGenerator: """Portfolio response generator — model loads lazily on first generate() call.""" SYSTEM_PROMPT = """You are Aethron, the portfolio intelligence agent for Arash Nicoomanesh. You answer questions about Arash's experience, skills, projects, and architecture philosophy based STRICTLY on the provided context. CRITICAL RULES: 1. STRICT factual GROUNDING: Answer using ONLY information explicitly written in the provided Context chunks. Do NOT invent, extrapolate, or infer dates, tenure lengths, years of experience, or company names that are not explicitly stated in the context. If a specific number or date is not in the context, do not guess — say what IS known and stop. 2. NO unsolicited contact info: Do NOT append email addresses or contact details (e.g., anicomanesh@gmail.com) unless the user explicitly asks how to contact, hire, or reach Arash. Focus strictly on answering the query. 3. CONCISE & COMPLETE: Provide clear, complete, and scannable answers using Markdown bullet points where appropriate. Finish your thoughts completely without truncation. 4. UNCERTAINTY: If the context does not contain enough information to answer fully, state clearly what is known from the context and stop. Do not pad with invented details. 5. For technical questions, reference specific projects, repositories, or architecture layers when relevant. 6. For recruiter questions, be concise and highlight technologies and roles explicitly mentioned in the context.""" RECRUITER_PROMPT = "You are speaking with a recruiter or hiring manager. Emphasize: years of experience, specific technologies, remote work availability, and education. Be professional and concise." TECHNICAL_PROMPT = "You are speaking with a technical partner or peer. Emphasize: architecture decisions, implementation details, repository names, and design philosophy. Be precise and technically deep." def __init__(self): self.pipe = None self.tokenizer = None def _load_model(self): """Load model and pipeline — called lazily inside @spaces.GPU context.""" if self.pipe is not None: return import torch from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline print(f"Loading SLM: {LLM_MODEL_ID}...") self.tokenizer = AutoTokenizer.from_pretrained(LLM_MODEL_ID) model = AutoModelForCausalLM.from_pretrained( LLM_MODEL_ID, dtype=torch.float16, device_map="auto" ) self.pipe = pipeline( "text-generation", model=model, tokenizer=self.tokenizer, return_full_text=False, do_sample=LLM_DO_SAMPLE, temperature=LLM_TEMPERATURE if LLM_DO_SAMPLE else None, max_new_tokens=LLM_MAX_NEW_TOKENS, pad_token_id=self.tokenizer.eos_token_id, ) print("SLM loaded successfully") def generate(self, query: str, chunks: List[Dict], persona: str = "general") -> Dict: """Generate response from retrieved context.""" self._load_model() context_text = "\n\n".join([ f"[Source: {c['section_type']} - {c['title']}]\n{c['text']}" for c in chunks ]) persona_addon = "" if persona == "recruiter": persona_addon = self.RECRUITER_PROMPT elif persona == "technical": persona_addon = self.TECHNICAL_PROMPT system_content = self.SYSTEM_PROMPT if persona_addon: system_content += "\n\n" + persona_addon user_content = f"Context:\n{context_text}\n\nQuestion: {query}\n\nAnswer based strictly on the above context:" messages = [ {"role": "system", "content": system_content}, {"role": "user", "content": user_content} ] prompt = self.tokenizer.apply_chat_template( messages, tokenize=False, add_generation_prompt=True ) outputs = self.pipe(prompt) response_text = outputs[0]["generated_text"].strip() response_text = self._clean_response(response_text) guard_passed = self._guard_check(response_text, chunks) unique_sources = self._deduplicate_sources(chunks) return { "answer": response_text, "sources": unique_sources, "persona": persona, "guard_passed": guard_passed, } def _deduplicate_sources(self, chunks: List[Dict]) -> List[str]: """Extract and deduplicate chunk source titles, preserving retrieval order.""" raw_titles = [] for c in chunks: title = c.get("title") or c.get("section_type") or c.get("source") if title: raw_titles.append(str(title).strip()) return list(dict.fromkeys(raw_titles)) def _clean_response(self, text: str) -> str: stops = ["<|end|>", "<|user|>", "<|system|>", "<|assistant|>", "Question:", "Context:"] for stop in stops: if stop in text: text = text[:text.index(stop)].strip() return text def _guard_check(self, response: str, chunks: List[Dict]) -> bool: refusal_phrases = ["don't have", "not in the portfolio", "cannot find"] if any(p in response.lower() for p in refusal_phrases): return True self_references = ["i am a", "i have", "i work", "my experience", "my skills"] if any(p in response.lower() for p in self_references): return False return True