TalentMatch-AI / matcher.py
Zulkefl07's picture
Upload 6 files
96a2583 verified
Raw
History Blame Contribute Delete
3.83 kB
# ============================================================
# matcher.py
# This file is the HEART of the project.
# It calculates how similar a resume is to a job description
# using TF-IDF and Cosine Similarity.
# ============================================================
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
import pandas as pd
def build_tfidf_vectorizer():
"""
Creates a TF-IDF Vectorizer.
TF-IDF stands for Term Frequency-Inverse Document Frequency.
It converts text into numbers.
Words that appear a lot in one resume but rarely in others
get a HIGH score (they are more important/unique).
Common words like 'the', 'is', 'a' get a LOW score.
ngram_range=(1,2) means we look at:
- single words: 'python', 'machine'
- word pairs: 'machine learning', 'deep learning'
max_features=5000 means we only keep the top 5000 words.
"""
vectorizer = TfidfVectorizer(
ngram_range=(1, 2), # use single words AND word pairs
max_features=5000, # keep only top 5000 important words
min_df=2, # ignore words that appear in less than 2 documents
max_df=0.95, # ignore words that appear in 95%+ of all documents
stop_words='english' # auto-remove common English words
)
return vectorizer
def calculate_similarity(resume_text, job_text):
"""
Calculates how similar ONE resume is to ONE job description.
Steps:
1. Combine both texts
2. Convert them to TF-IDF vectors (numbers)
3. Use cosine similarity to compare the two vectors
Cosine Similarity returns a value between 0 and 1:
- 1.0 = perfect match
- 0.0 = completely different
Think of it like measuring the angle between two arrows —
if they point in the same direction, they are very similar!
"""
vectorizer = build_tfidf_vectorizer()
# Fit and transform both texts together
tfidf_matrix = vectorizer.fit_transform([resume_text, job_text])
# Calculate cosine similarity between resume (row 0) and job (row 1)
score = cosine_similarity(tfidf_matrix[0:1], tfidf_matrix[1:2])
return round(float(score[0][0]), 4) # return as a clean number
def rank_candidates(df, job_description, resume_col='cleaned_resume', top_n=10):
"""
Ranks ALL candidates in the dataset against a given job description.
Steps:
1. Vectorize all resumes + job description together
2. Calculate similarity of EACH resume vs. job description
3. Sort by similarity score (highest first)
4. Return the top N candidates
Parameters:
df - the DataFrame with all resumes
job_description - the job description text (already cleaned)
resume_col - which column has the cleaned resume text
top_n - how many top candidates to return (default: 10)
"""
# Combine all resumes + job description into one list
all_texts = list(df[resume_col].values) + [job_description]
# Build TF-IDF vectors for everything
vectorizer = build_tfidf_vectorizer()
tfidf_matrix = vectorizer.fit_transform(all_texts)
# The job description vector is the LAST item
job_vector = tfidf_matrix[-1]
# Calculate similarity of each resume vs. job description
similarities = cosine_similarity(tfidf_matrix[:-1], job_vector)
# Add the scores back to the DataFrame (copy to avoid warnings)
ranked_df = df.copy()
ranked_df['computed_similarity'] = similarities.flatten()
# Sort by similarity score — best match first
ranked_df = ranked_df.sort_values(
by='computed_similarity', ascending=False
).reset_index(drop=True)
# Return only the top N results
return ranked_df.head(top_n)