File size: 3,825 Bytes
96a2583
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
# ============================================================
# matcher.py
# This file is the HEART of the project.
# It calculates how similar a resume is to a job description
# using TF-IDF and Cosine Similarity.
# ============================================================

from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
import pandas as pd


def build_tfidf_vectorizer():
    """
    Creates a TF-IDF Vectorizer.

    TF-IDF stands for Term Frequency-Inverse Document Frequency.
    It converts text into numbers.
    Words that appear a lot in one resume but rarely in others
    get a HIGH score (they are more important/unique).
    Common words like 'the', 'is', 'a' get a LOW score.

    ngram_range=(1,2) means we look at:
      - single words: 'python', 'machine'
      - word pairs: 'machine learning', 'deep learning'
    max_features=5000 means we only keep the top 5000 words.
    """
    vectorizer = TfidfVectorizer(
        ngram_range=(1, 2),     # use single words AND word pairs
        max_features=5000,      # keep only top 5000 important words
        min_df=2,               # ignore words that appear in less than 2 documents
        max_df=0.95,            # ignore words that appear in 95%+ of all documents
        stop_words='english'    # auto-remove common English words
    )
    return vectorizer


def calculate_similarity(resume_text, job_text):
    """
    Calculates how similar ONE resume is to ONE job description.

    Steps:
    1. Combine both texts
    2. Convert them to TF-IDF vectors (numbers)
    3. Use cosine similarity to compare the two vectors

    Cosine Similarity returns a value between 0 and 1:
    - 1.0 = perfect match
    - 0.0 = completely different

    Think of it like measuring the angle between two arrows —
    if they point in the same direction, they are very similar!
    """
    vectorizer = build_tfidf_vectorizer()

    # Fit and transform both texts together
    tfidf_matrix = vectorizer.fit_transform([resume_text, job_text])

    # Calculate cosine similarity between resume (row 0) and job (row 1)
    score = cosine_similarity(tfidf_matrix[0:1], tfidf_matrix[1:2])

    return round(float(score[0][0]), 4)  # return as a clean number


def rank_candidates(df, job_description, resume_col='cleaned_resume', top_n=10):
    """
    Ranks ALL candidates in the dataset against a given job description.

    Steps:
    1. Vectorize all resumes + job description together
    2. Calculate similarity of EACH resume vs. job description
    3. Sort by similarity score (highest first)
    4. Return the top N candidates

    Parameters:
        df           - the DataFrame with all resumes
        job_description - the job description text (already cleaned)
        resume_col   - which column has the cleaned resume text
        top_n        - how many top candidates to return (default: 10)
    """
    # Combine all resumes + job description into one list
    all_texts = list(df[resume_col].values) + [job_description]

    # Build TF-IDF vectors for everything
    vectorizer = build_tfidf_vectorizer()
    tfidf_matrix = vectorizer.fit_transform(all_texts)

    # The job description vector is the LAST item
    job_vector = tfidf_matrix[-1]

    # Calculate similarity of each resume vs. job description
    similarities = cosine_similarity(tfidf_matrix[:-1], job_vector)

    # Add the scores back to the DataFrame (copy to avoid warnings)
    ranked_df = df.copy()
    ranked_df['computed_similarity'] = similarities.flatten()

    # Sort by similarity score — best match first
    ranked_df = ranked_df.sort_values(
        by='computed_similarity', ascending=False
    ).reset_index(drop=True)

    # Return only the top N results
    return ranked_df.head(top_n)