| import gradio as gr |
| import pandas as pd |
| import requests |
| import re |
| import tempfile |
| import shutil |
| import os |
| from difflib import SequenceMatcher |
| import json |
| from urllib.parse import quote_plus |
| import zipfile |
| from datetime import datetime |
|
|
|
|
| |
| def zip_and_prepare_download(file_bytes, inner_filename, zip_prefix="Download"): |
| zip_filename = f"{zip_prefix}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.zip" |
| zip_file_path = tempfile.NamedTemporaryFile(delete=False, suffix=".zip").name |
|
|
| with zipfile.ZipFile(zip_file_path, 'w') as zipf: |
| zipf.writestr(inner_filename, file_bytes) |
|
|
| print(f"[DEBUG] Created ZIP at {zip_file_path} with filename {zip_filename}") |
| return zip_file_path |
|
|
|
|
| |
| def construct_query(row): |
| query = str(row['Applicant Name']) |
| optional_fields = ['Job Title', 'State', 'City', 'Skills'] |
|
|
| for field in optional_fields: |
| if field in row and pd.notna(row[field]): |
| value = row[field] |
| query += f" {str(value).strip()}" if str(value).strip() else "" |
|
|
| query += " linkedin" |
| print(f"[DEBUG] Search Query: {query}") |
| return query |
|
|
| def get_name_from_url(link): |
| match = re.search(r'linkedin\.com/in/([a-zA-Z0-9-]+)', link) |
| if match: |
| profile_name = match.group(1).replace('-', ' ') |
| print(f"[DEBUG] Extracted profile name from URL: {profile_name}") |
| return profile_name |
| return None |
|
|
| def calculate_similarity(name1, name2): |
| similarity = SequenceMatcher(None, name1.lower().strip(), name2.lower().strip()).ratio() |
| print(f"[DEBUG] Similarity between '{name1}' and '{name2}' = {similarity}") |
| return similarity |
|
|
| def fetch_linkedin_links(query, api_key, applicant_name): |
| try: |
| print(f"[DEBUG] Sending request to BrightData for query: {query}") |
| url = "https://api.brightdata.com/request" |
| google_url = f"https://www.google.com/search?q={quote_plus(query)}" |
|
|
| payload = { |
| "zone": "serp_api2", |
| "url": google_url, |
| "method": "GET", |
| "country": "us", |
| "format": "raw", |
| "data_format": "html" |
| } |
|
|
| headers = { |
| "Authorization": f"Bearer {api_key}", |
| "Content-Type": "application/json" |
| } |
|
|
| response = requests.post(url, headers=headers, json=payload) |
| response.raise_for_status() |
| html = response.text |
|
|
| linkedin_regex = r'https://(?:[a-z]{2,3}\.)?linkedin\.com/in/[a-zA-Z0-9\-_/]+' |
| matches = re.findall(linkedin_regex, html) |
| print(f"[DEBUG] Found {len(matches)} LinkedIn link(s) in search result") |
|
|
| for link in matches: |
| profile_name = get_name_from_url(link) |
| if profile_name: |
| similarity = calculate_similarity(applicant_name, profile_name) |
| if similarity >= 0.5: |
| print(f"[DEBUG] Match found: {link}") |
| return link |
| print(f"[DEBUG] No matching LinkedIn profile found for: {applicant_name}") |
| return None |
|
|
| except Exception as e: |
| print(f"[ERROR] Error fetching LinkedIn link for query '{query}': {e}") |
| return None |
|
|
|
|
| |
| def process_file_gradio(file_obj, api_key): |
| try: |
| df = pd.read_excel(file_obj.name) |
| print(f"[DEBUG] Input file read successfully. Rows: {len(df)}") |
|
|
| if 'Applicant Name' not in df.columns: |
| return None, "β Missing required column: 'Applicant Name'" |
|
|
| df = df[df['Applicant Name'].notna()] |
| df = df[df['Applicant Name'].str.strip() != ''] |
| print(f"[DEBUG] Valid applicant rows after filtering: {len(df)}") |
|
|
| df['Search Query'] = df.apply(construct_query, axis=1) |
| df['LinkedIn Link'] = df.apply( |
| lambda row: fetch_linkedin_links(row['Search Query'], api_key, row['Applicant Name']), |
| axis=1 |
| ) |
|
|
| temp_dir = tempfile.mkdtemp() |
| output_file = os.path.join(temp_dir, "updated_with_linkedin_links.csv") |
| df.to_csv(output_file, index=False) |
| print(f"[DEBUG] Output written to: {output_file}") |
|
|
| with open(output_file, "rb") as f: |
| csv_bytes = f.read() |
|
|
| zip_path = zip_and_prepare_download(csv_bytes, "updated_with_linkedin_links.csv", "LinkedIn_Links") |
|
|
| shutil.rmtree(temp_dir) |
| return zip_path, "β
Success! Download your file below." |
|
|
| except Exception as e: |
| print(f"[ERROR] Error processing file: {e}") |
| return None, f"β Error: {str(e)}" |
|
|
|
|
| |
| with gr.Blocks(title="LinkedIn Scraper") as demo: |
| gr.Markdown("## π LinkedIn Profile Scraper") |
| gr.Markdown("Upload an Excel file with applicant details to fetch best-matching LinkedIn profile links (via Google Search using BrightData API).") |
|
|
| with gr.Row(): |
| api_key_input = gr.Textbox(label="π BrightData API Key", type="password", placeholder="Enter your BrightData SERP API Key") |
| file_input = gr.File(label="π€ Upload Excel File (.xlsx)", file_types=[".xlsx"]) |
|
|
| process_btn = gr.Button("π Start Processing") |
| status_output = gr.Textbox(label="π’ Status") |
| download_btn = gr.File(label="π₯ Download ZIP") |
|
|
| def run_pipeline(api_key, file): |
| if not api_key: |
| return None, "β Please enter your API key" |
| if not file: |
| return None, "β Please upload a valid Excel file" |
| return process_file_gradio(file, api_key) |
|
|
| process_btn.click( |
| fn=run_pipeline, |
| inputs=[api_key_input, file_input], |
| outputs=[download_btn, status_output] |
| ) |
|
|
| |
| if __name__ == "__main__": |
| demo.launch() |