File size: 6,027 Bytes
0102a44
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
import os
import io
import time
from collections import defaultdict
import numpy as np
from PIL import Image
from deepface import DeepFace
from google.oauth2 import service_account
from googleapiclient.discovery import build
from googleapiclient.http import MediaIoBaseDownload, MediaFileUpload
from simple_salesforce import Salesforce

# Configuration settings
SCOPES = ['https://www.googleapis.com/auth/drive']
SERVICE_ACCOUNT_FILE = r'C:\Users\Ajaya\Downloads\vertical-sunset-440402-h4-f3e25263f3a1.json'
SF_USERNAME = 'navyagrand7890@gmail.com'
SF_PASSWORD = 'navya@123'
SF_SECURITY_TOKEN = 'NeAPhh5Puxmc1EsL1IJPcQB5'
AADHAR_FOLDER_ID = '1Qtb5DYzSFE67Mbb5ZgDIqWUtdJaDD2F4'
CPHOTOS_FOLDER_ID = '1DGeRqRbCPcfLDdEgP0h5fyX-MF8EQ8AH'
SUSPECTS_FOLDER_ID = '1N3RMhVD0OygeufLPYod6IYLtqzvlm3Jv'
THRESHOLD = 0.65

# Initialize services
def initialize_services():
    credentials = service_account.Credentials.from_service_account_file(SERVICE_ACCOUNT_FILE, scopes=SCOPES)
    drive_service = build('drive', 'v3', credentials=credentials)
    sf = Salesforce(username=SF_USERNAME, password=SF_PASSWORD, security_token=SF_SECURITY_TOKEN)
    return drive_service, sf

# List files in a folder
def list_files_in_folder(drive_service, folder_id):
    query = f"'{folder_id}' in parents and trashed=false"
    results = drive_service.files().list(q=query, pageSize=1000, fields="files(id, name)").execute()
    return results.get('files', [])

# Download a file from Google Drive
def download_file(drive_service, file_id, file_name):
    request = drive_service.files().get_media(fileId=file_id)
    file_io = io.BytesIO()
    downloader = MediaIoBaseDownload(file_io, request)
    done = False
    while not done:
        _, done = downloader.next_chunk()
    download_path = os.path.join("downloads", file_name)
    os.makedirs("downloads", exist_ok=True)
    with open(download_path, 'wb') as f:
        f.write(file_io.getvalue())
    return download_path

# Group similar images based on DeepFace embeddings
def group_similar_images(image_paths):
    embeddings = []
    grouped_images = defaultdict(list)

    # Get embeddings for each image
    for image_path in image_paths:
        embedding = DeepFace.represent(image_path, model_name='ArcFace', enforce_detection=False)[0]['embedding']
        embeddings.append((image_path, embedding))

    # Cluster images based on similarity
    for i, (image_path_i, embedding_i) in enumerate(embeddings):
        for j, (image_path_j, embedding_j) in enumerate(embeddings):
            if i >= j:
                continue
            similarity = np.dot(embedding_i, embedding_j) / (np.linalg.norm(embedding_i) * np.linalg.norm(embedding_j))
            if similarity > 0.85:  # Threshold for grouping similar images
                grouped_images[i].append(image_path_j)

    # Print the number of images per group
    print(f"[INFO] Grouped similar images into {len(grouped_images)} groups based on similarity.")

    return grouped_images

# Function to pick best match per group
def pick_best_match_and_clean_up(grouped_images, aadhar_images, threshold, suspects_folder_id, drive_service, sf):
    for group_id, image_paths in grouped_images.items():
        match_found = False
        for cphoto_path in image_paths:
            for aadhar_path in aadhar_images:
                is_match, distance = compare_faces_deepface(aadhar_path, cphoto_path, threshold)
                if is_match:
                    match_found = True
                    print(f"[INFO] Match found for group {group_id} with image {cphoto_path}")
                    break
            if match_found:
                break

        # Clean up: only keep best-matched image, delete others
        if match_found:
            for extra_path in image_paths:
                if extra_path != cphoto_path:
                    os.remove(extra_path)
                    print(f"[INFO] Deleted redundant image: {extra_path}")
        else:
            # No match found for the group, mark as suspect
            for cphoto_path in image_paths:
                uploaded_file_id, uploaded_file_url = upload_file(drive_service, cphoto_path, suspects_folder_id)
                create_suspect_record_in_salesforce(sf, uploaded_file_id, os.path.basename(cphoto_path), uploaded_file_url)

# Upload file to Google Drive
def upload_file(drive_service, file_path, folder_id):
    file_metadata = {'name': os.path.basename(file_path), 'parents': [folder_id]}
    media = MediaFileUpload(file_path, resumable=True)
    file = drive_service.files().create(body=file_metadata, media_body=media, fields='id, webViewLink').execute()
    return file.get('id'), file.get('webViewLink')

# Salesforce record creation
def create_suspect_record_in_salesforce(sf, file_id, file_name, file_url):
    record = {'Name': file_name, 'File_ID__c': file_id, 'File_Name__c': file_name, 'File_URL__c': file_url}
    sf.zia__c.create(record)
    print(f"[INFO] Created suspect record in Salesforce for file: {file_name}")

# Compare images with DeepFace
def compare_faces_deepface(img1_path, img2_path, threshold):
    result = DeepFace.verify(img1_path=img1_path, img2_path=img2_path, model_name='ArcFace', enforce_detection=False)
    return result['verified'], result['distance']

def main():
    drive_service, sf = initialize_services()
    aadhar_files = list_files_in_folder(drive_service, AADHAR_FOLDER_ID)
    cphotos_files = list_files_in_folder(drive_service, CPHOTOS_FOLDER_ID)

    # Download Aadhaar images
    aadhar_images = [download_file(drive_service, file['id'], file['name']) for file in aadhar_files]

    # Download CCTV images and group by similarity
    cphotos_paths = [download_file(drive_service, file['id'], file['name']) for file in cphotos_files]
    grouped_images = group_similar_images(cphotos_paths)

    # Process each group to find best match and clean up
    pick_best_match_and_clean_up(grouped_images, aadhar_images, THRESHOLD, SUSPECTS_FOLDER_ID, drive_service, sf)

if __name__ == "__main__":
    main()