import os import logging import json import re from collections import Counter from difflib import SequenceMatcher import google.generativeai as genai logger = logging.getLogger(__name__) STOPWORDS = { "a", "an", "and", "are", "as", "at", "be", "by", "for", "from", "has", "have", "in", "is", "it", "its", "of", "on", "or", "that", "the", "to", "was", "were", "will", "with", "you", "your", "we", "our", "us", "their", "they", "this", "these", "those", "job", "role", "candidate", "experience", "work", "ability", "skills", "skill", "years", "year", "required", "preferred", "strong", "knowledge", "using", "must", "including", "etc", "good", "plus", "team", "responsible", "responsibilities" } class ATSEvaluator: """ Singleton wrapper for ATS Semantic Evaluation. Ensures heavy ML models are only loaded into memory once and only when needed. """ _instance = None _sbert_model = None def __new__(cls): if cls._instance is None: cls._instance = super(ATSEvaluator, cls).__new__(cls) cls._instance._initialize_gemini() return cls._instance def _initialize_gemini(self): """Setup Gemini configuration once.""" api_key = os.environ.get("GOOGLE_API_KEY", "") if api_key: genai.configure(api_key=api_key) generation_config = { "temperature": 0.1, # Lowered for stricter formatting "top_p": 1, "top_k": 32, "max_output_tokens": 1200, "response_mime_type": "application/json" # ENFORCE NATIVE JSON } safety_settings = [ {"category": "HARM_CATEGORY_HARASSMENT", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_HATE_SPEECH", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_SEXUALLY_EXPLICIT", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_DANGEROUS_CONTENT", "threshold": "BLOCK_NONE"}, ] self.llm = genai.GenerativeModel( model_name="gemini-2.5-flash", generation_config=generation_config ) logger.info("ATSEvaluator: Gemini LLM initialized.") def _extract_fallback_skills(self, text, limit=25): """Regex-based fallback if Gemini fails to extract skills.""" tokens = re.findall(r"[A-Za-z][A-Za-z0-9+#./-]{1,}", text.lower()) tokens = [t for t in tokens if len(t) > 2 and t not in STOPWORDS] # Create 2-word combos (e.g., "machine learning", "react js") bigrams = [f"{tokens[i]} {tokens[i + 1]}" for i in range(len(tokens) - 1)] candidates = tokens + bigrams filtered = [c for c in candidates if len(c) <= 48] # Return the most common phrases return [p for p, _ in Counter(filtered).most_common(limit)] def _get_sbert_model(self): """ LAZY LOADING: Only loads the 1.45GB model when a user actually clicks 'Analyze Fit'. """ if self._sbert_model is None: logger.info("Initializing SBERT model into RAM... This may take a moment.") try: # Import dynamically so it doesn't break if files are missing on boot from model.inference import sbert_inference from config import MODEL_CONFIG self._sbert_model = sbert_inference.load_model(MODEL_CONFIG['sbert_path']) self.sbert_inference = sbert_inference logger.info("SBERT model successfully loaded into memory!") except Exception as e: logger.error(f"Failed to load SBERT model: {e}") raise e return self._sbert_model # --- Core Analysis Methods Ported from app.py --- def _normalize_keyword(self, keyword): return re.sub(r"[^a-z0-9#+./-]", "", keyword.lower().strip()) def _parse_json_response(self, content): cleaned = content.strip() if cleaned.startswith("```"): cleaned = cleaned.strip("`") if cleaned.lower().startswith("json"): cleaned = cleaned[4:].strip() start = cleaned.find("{") end = cleaned.rfind("}") if start != -1 and end != -1: cleaned = cleaned[start:end + 1] try: return json.loads(cleaned) except: return {} def _fuzzy_keyword_match(self, jd_keyword, resume_keywords, threshold=0.88): jd_norm = self._normalize_keyword(jd_keyword) if not jd_norm: return None for res_kw in resume_keywords: res_norm = self._normalize_keyword(res_kw) if not res_norm: continue if jd_norm == res_norm or jd_norm in res_norm or res_norm in jd_norm: return res_kw best_match = None best_score = 0.0 for res_kw in resume_keywords: res_norm = self._normalize_keyword(res_kw) ratio = SequenceMatcher(None, jd_norm, res_norm).ratio() if ratio > best_score: best_score = ratio best_match = res_kw if best_score >= threshold: return best_match return None def evaluate_fit(self, resume_text, jd_text): """ The main public function. Runs semantic similarity and skill gap analysis. """ logger.info("Starting ATS Fit Evaluation...") # 1. Trigger the Lazy Load of the Semantic Model sbert_model = self._get_sbert_model() # 2. Calculate Deep Semantic Similarity semantic_similarity = self.sbert_inference.calculate_similarity(sbert_model, resume_text, jd_text) # 3. Extract Skills via Gemini LLM (with Fallback) prompt = f""" Extract only technical skills from the two texts. Return strictly valid JSON using this exact schema: {{ "jd_required_skills": ["skill1", "skill2"], "resume_skills": ["skill3", "skill4"] }} JD Text: {jd_text[:8000]} Resume Text: {resume_text[:8000]} """ try: # --- THE X-RAY LOGS --- logger.info(f"DEBUG: JD Text Length: {len(jd_text)}") logger.info(f"DEBUG: Resume Text Length: {len(resume_text)}") response = self.llm.generate_content(prompt) logger.info(f"DEBUG: Raw Gemini Response: {response.text}") # ---------------------- response = self.llm.generate_content(prompt) data = self._parse_json_response(response.text) jd_skills = data.get("jd_required_skills", []) resume_skills = data.get("resume_skills", []) # If Gemini hallucinates and returns empty arrays, trigger the fallback manually if not jd_skills: raise ValueError("Gemini returned an empty skills array.") except Exception as e: logger.warning(f"Gemini extraction failed, triggering regex fallback: {e}") jd_skills = self._extract_fallback_skills(jd_text, limit=25) resume_skills = self._extract_fallback_skills(resume_text, limit=40) # 4. Map the Gaps matched = [] missing = [] for jd_skill in jd_skills: hit = self._fuzzy_keyword_match(jd_skill, resume_skills) if hit: matched.append(jd_skill) else: missing.append(jd_skill) # Smart formatting for the score sim_float = float(semantic_similarity) if semantic_similarity else 0.0 match_score = round(sim_float) if sim_float > 1.0 else round(sim_float * 100) return { "match_score": match_score, "matched_skills": matched, "missing_skills": missing } """ The main public function. Runs semantic similarity and skill gap analysis. """ logger.info("Starting ATS Fit Evaluation...") # 1. Trigger the Lazy Load of the Semantic Model sbert_model = self._get_sbert_model() # 2. Calculate Deep Semantic Similarity semantic_similarity = self.sbert_inference.calculate_similarity(sbert_model, resume_text, jd_text) # 3. Extract Skills via Gemini LLM prompt = f""" Extract only technical skills from the two texts. Return only valid JSON: {{ "jd_required_skills": ["skill phrase"], "resume_skills": ["skill phrase"] }} Rules: 1 to 4 words per skill. Max 40 skills. No markdown. JD Text: {jd_text[:9000]} Resume Text: {resume_text[:9000]} """ try: response = self.llm.generate_content(prompt) data = self._parse_json_response(response.text) jd_skills = data.get("jd_required_skills", []) resume_skills = data.get("resume_skills", []) except Exception as e: logger.error(f"Gemini extraction failed: {e}") jd_skills = [] resume_skills = [] # 4. Map the Gaps matched = [] missing = [] for jd_skill in jd_skills: hit = self._fuzzy_keyword_match(jd_skill, resume_skills) if hit: matched.append(jd_skill) else: missing.append(jd_skill) # Smart formatting: if the model already returns a percentage (e.g., 56.16), don't multiply. sim_float = float(semantic_similarity) if semantic_similarity else 0.0 match_score = round(sim_float) if sim_float > 1.0 else round(sim_float * 100) return { "match_score": match_score, "matched_skills": matched, "missing_skills": missing }