import gradio as gr import pandas as pd import requests import re import tempfile import shutil import os from difflib import SequenceMatcher import json from urllib.parse import quote_plus import zipfile from datetime import datetime # -----------zip_and_prepare_download-------------- def zip_and_prepare_download(file_bytes, inner_filename, zip_prefix="Download"): zip_filename = f"{zip_prefix}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.zip" zip_file_path = tempfile.NamedTemporaryFile(delete=False, suffix=".zip").name with zipfile.ZipFile(zip_file_path, 'w') as zipf: zipf.writestr(inner_filename, file_bytes) print(f"[DEBUG] Created ZIP at {zip_file_path} with filename {zip_filename}") return zip_file_path # -----------Utilities-------------- def construct_query(row): query = str(row['Applicant Name']) optional_fields = ['Job Title', 'State', 'City', 'Skills'] for field in optional_fields: if field in row and pd.notna(row[field]): value = row[field] query += f" {str(value).strip()}" if str(value).strip() else "" query += " linkedin" print(f"[DEBUG] Search Query: {query}") return query def get_name_from_url(link): match = re.search(r'linkedin\.com/in/([a-zA-Z0-9-]+)', link) if match: profile_name = match.group(1).replace('-', ' ') print(f"[DEBUG] Extracted profile name from URL: {profile_name}") return profile_name return None def calculate_similarity(name1, name2): similarity = SequenceMatcher(None, name1.lower().strip(), name2.lower().strip()).ratio() print(f"[DEBUG] Similarity between '{name1}' and '{name2}' = {similarity}") return similarity def fetch_linkedin_links(query, api_key, applicant_name): try: print(f"[DEBUG] Sending request to BrightData for query: {query}") url = "https://api.brightdata.com/request" google_url = f"https://www.google.com/search?q={quote_plus(query)}" payload = { "zone": "serp_api2", "url": google_url, "method": "GET", "country": "us", "format": "raw", "data_format": "html" } headers = { "Authorization": f"Bearer {api_key}", "Content-Type": "application/json" } response = requests.post(url, headers=headers, json=payload) response.raise_for_status() html = response.text linkedin_regex = r'https://(?:[a-z]{2,3}\.)?linkedin\.com/in/[a-zA-Z0-9\-_/]+' matches = re.findall(linkedin_regex, html) print(f"[DEBUG] Found {len(matches)} LinkedIn link(s) in search result") for link in matches: profile_name = get_name_from_url(link) if profile_name: similarity = calculate_similarity(applicant_name, profile_name) if similarity >= 0.5: print(f"[DEBUG] Match found: {link}") return link print(f"[DEBUG] No matching LinkedIn profile found for: {applicant_name}") return None except Exception as e: print(f"[ERROR] Error fetching LinkedIn link for query '{query}': {e}") return None # ----------Process Excel--------------- def process_file_gradio(file_obj, api_key): try: df = pd.read_excel(file_obj.name) print(f"[DEBUG] Input file read successfully. Rows: {len(df)}") if 'Applicant Name' not in df.columns: return None, "❌ Missing required column: 'Applicant Name'" df = df[df['Applicant Name'].notna()] df = df[df['Applicant Name'].str.strip() != ''] print(f"[DEBUG] Valid applicant rows after filtering: {len(df)}") df['Search Query'] = df.apply(construct_query, axis=1) df['LinkedIn Link'] = df.apply( lambda row: fetch_linkedin_links(row['Search Query'], api_key, row['Applicant Name']), axis=1 ) temp_dir = tempfile.mkdtemp() output_file = os.path.join(temp_dir, "updated_with_linkedin_links.csv") df.to_csv(output_file, index=False) print(f"[DEBUG] Output written to: {output_file}") with open(output_file, "rb") as f: csv_bytes = f.read() zip_path = zip_and_prepare_download(csv_bytes, "updated_with_linkedin_links.csv", "LinkedIn_Links") shutil.rmtree(temp_dir) return zip_path, "✅ Success! Download your file below." except Exception as e: print(f"[ERROR] Error processing file: {e}") return None, f"❌ Error: {str(e)}" # ----------Gradio UI--------------- with gr.Blocks(title="LinkedIn Scraper") as demo: gr.Markdown("## 🔗 LinkedIn Profile Scraper") gr.Markdown("Upload an Excel file with applicant details to fetch best-matching LinkedIn profile links (via Google Search using BrightData API).") with gr.Row(): api_key_input = gr.Textbox(label="🔐 BrightData API Key", type="password", placeholder="Enter your BrightData SERP API Key") file_input = gr.File(label="📤 Upload Excel File (.xlsx)", file_types=[".xlsx"]) process_btn = gr.Button("🚀 Start Processing") status_output = gr.Textbox(label="📢 Status") download_btn = gr.File(label="📥 Download ZIP") def run_pipeline(api_key, file): if not api_key: return None, "❗ Please enter your API key" if not file: return None, "❗ Please upload a valid Excel file" return process_file_gradio(file, api_key) process_btn.click( fn=run_pipeline, inputs=[api_key_input, file_input], outputs=[download_btn, status_output] ) # Run app if __name__ == "__main__": demo.launch()