pujithapsx's picture
12 th jan updated rules with client rules
5ec46f6
Raw
History Blame Contribute Delete
25.9 kB
import re
from datetime import datetime
from typing import List, Dict
# =========================================================
# NORMALIZATION & PREPROCESSING
# =========================================================
def preprocess_text(text):
"""Remove extra trailing/leading spaces and normalize whitespace"""
if not text:
return ""
return re.sub(r"\s+", " ", text.strip())
def normalize_text(text):
"""Normalize text to lowercase and remove extra spaces"""
return re.sub(r"\s+", " ", text.lower().strip()) if text else ""
# =========================================================
# VALIDATION FUNCTIONS
# =========================================================
def validate_and_normalize_pincode(pincode):
"""
Validate and normalize pincode to exactly 6 digits
Returns normalized pincode or None if invalid
"""
if not pincode:
return None
digits = re.sub(r'\D', '', str(pincode).strip())
if len(digits) == 6:
return digits
return None
def validate_and_normalize_phone(phone):
"""
Validate and normalize phone to exactly 10 digits
Handles formats: +91, 91-, 91, or plain 10 digits
Returns normalized 10-digit phone or None if invalid
"""
if not phone:
return None
phone_str = str(phone).strip()
# Remove common prefixes and separators
phone_str = re.sub(r'^\+91[-\s]?', '', phone_str)
phone_str = re.sub(r'^91[-\s]?', '', phone_str)
phone_str = re.sub(r'^0[-\s]?', '', phone_str)
digits = re.sub(r'\D', '', phone_str)
if len(digits) == 10:
return digits
return None
def validate_and_normalize_email(email):
"""
Validate and normalize email using regex
Returns normalized email or None if invalid
"""
if not email:
return None
email_str = str(email).strip().lower()
email_pattern = r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$'
if re.match(email_pattern, email_str):
return email_str
return None
def normalize_dob(dob_str):
"""Normalize DOB to YYYY-MM-DD format"""
if not dob_str:
return None
formats = [
"%Y-%m-%d", "%Y/%m/%d",
"%d-%m-%Y", "%d/%m/%Y",
]
for fmt in formats:
try:
dt = datetime.strptime(dob_str, fmt)
return dt.strftime("%Y-%m-%d")
except ValueError:
continue
return None
# =========================================================
# STATE & CITY MAPPINGS
# =========================================================
STATE_MAPPING = {
"andhra pradesh": ["andhra pradesh", "ap", "a.p", "a.p.", "andhra", "andhrapradesh"],
"arunachal pradesh": ["arunachal pradesh", "arunachal", "ar", "a.r"],
"assam": ["assam", "as", "a.s"],
"bihar": ["bihar", "br", "b.r"],
"chhattisgarh": ["chhattisgarh", "chattisgarh", "chhatisgarh", "cg", "c.g"],
"goa": ["goa", "ga", "g.a"],
"gujarat": ["gujarat", "gujrat", "gj", "g.j"],
"haryana": ["haryana", "hr", "h.r"],
"himachal pradesh": ["himachal pradesh", "himachal", "hp", "h.p", "h.p."],
"jharkhand": ["jharkhand", "jh", "j.h"],
"karnataka": ["karnataka", "ka", "k.a", "karn", "karnatak"],
"kerala": ["kerala", "kl", "k.l"],
"madhya pradesh": ["madhya pradesh", "mp", "m.p", "m.p.", "madhya", "madhyapradesh"],
"maharashtra": ["maharashtra", "mh", "m.h", "maha", "maharastra"],
"manipur": ["manipur", "mn", "m.n"],
"meghalaya": ["meghalaya", "ml", "m.l"],
"mizoram": ["mizoram", "mz", "m.z"],
"nagaland": ["nagaland", "nl", "n.l"],
"odisha": ["odisha", "orissa", "od", "o.d", "or", "o.r"],
"punjab": ["punjab", "pb", "p.b"],
"rajasthan": ["rajasthan", "rj", "r.j", "raj"],
"sikkim": ["sikkim", "sk", "s.k"],
"tamil nadu": ["tamil nadu", "tamilnadu", "tn", "t.n", "t.n.", "tamil"],
"telangana": ["telangana", "tg", "t.g", "telengana", "ts", "t.s"],
"tripura": ["tripura", "tr", "t.r"],
"uttar pradesh": ["uttar pradesh", "up", "u.p", "u.p.", "uttar", "uttarpradesh"],
"uttarakhand": ["uttarakhand", "uttaranchal", "uk", "u.k", "ua"],
"west bengal": ["west bengal", "westbengal", "wb", "w.b", "w.b.", "bengal"],
"andaman and nicobar islands": ["andaman and nicobar islands", "andaman", "nicobar", "an", "a.n"],
"chandigarh": ["chandigarh", "ch", "c.h"],
"dadra and nagar haveli and daman and diu": ["dadra and nagar haveli and daman and diu", "dadra", "daman", "diu", "dn", "d.n", "dnh"],
"lakshadweep": ["lakshadweep", "ld", "l.d"],
"delhi": ["delhi", "new delhi", "dl", "d.l"],
"puducherry": ["puducherry", "pondicherry", "py", "p.y"],
"ladakh": ["ladakh", "la", "l.a"],
"jammu and kashmir": ["jammu and kashmir", "jammu", "kashmir", "jk", "j.k", "j&k"]
}
CITY_MAPPING = {
"mumbai": ["mumbai", "bombay", "bom"],
"delhi": ["delhi", "dl", "new delhi", "newdelhi"],
"bengaluru": ["bengaluru", "bangalore", "blr"],
"hyderabad": ["hyderabad", "hyd", "secunderabad"],
"ahmedabad": ["ahmedabad", "amdavad", "amd"],
"chennai": ["chennai", "madras", "maa"],
"kolkata": ["kolkata", "calcutta", "cal"],
"pune": ["pune", "poona"],
"jaipur": ["jaipur", "jai"],
"surat": ["surat", "sur"],
"lucknow": ["lucknow", "lko"],
"kanpur": ["kanpur", "cawnpore"],
"nagpur": ["nagpur", "nag"],
"indore": ["indore", "ind"],
"thane": ["thane", "tha"],
"bhopal": ["bhopal", "bho"],
"visakhapatnam": ["visakhapatnam", "vizag", "vishakhapatnam", "vsp"],
"pimpri-chinchwad": ["pimpri-chinchwad", "pimpri", "chinchwad"],
"patna": ["patna", "pat"],
"vadodara": ["vadodara", "baroda", "vad"],
"ghaziabad": ["ghaziabad", "ghz"],
"ludhiana": ["ludhiana", "ldh"],
"agra": ["agra", "agr"],
"nashik": ["nashik", "nasik"],
"faridabad": ["faridabad", "fbd"],
"meerut": ["meerut", "mer"],
"rajkot": ["rajkot", "raj"],
"kalyan-dombivli": ["kalyan-dombivli", "kalyan", "dombivli"],
"vasai-virar": ["vasai-virar", "vasai", "virar"],
"varanasi": ["varanasi", "banaras", "kashi"],
"srinagar": ["srinagar", "sri"],
"aurangabad": ["aurangabad", "aur"],
"dhanbad": ["dhanbad", "dhn"],
"amritsar": ["amritsar", "asr"],
"navi mumbai": ["navi mumbai", "navimumbai", "new mumbai"],
"allahabad": ["allahabad", "prayagraj", "pra"],
"ranchi": ["ranchi", "ran"],
"howrah": ["howrah", "how"],
"coimbatore": ["coimbatore", "coi"],
"jabalpur": ["jabalpur", "jab"],
"gwalior": ["gwalior", "gwa"],
"vijayawada": ["vijayawada", "vij"],
"jodhpur": ["jodhpur", "jod"],
"madurai": ["madurai", "mad"],
"raipur": ["raipur", "rai"],
"kota": ["kota", "kot"],
"chandigarh": ["chandigarh", "chd"],
"guwahati": ["guwahati", "gauhati", "guw"],
"thiruvananthapuram": ["thiruvananthapuram", "trivandrum", "trv"],
"mysore": ["mysore", "mysuru", "mys"],
"dehradun": ["dehradun", "ddn"],
"bhubaneswar": ["bhubaneswar", "bhubaneshwar", "bbs"],
"kochi": ["kochi", "cochin", "cok"],
"shimla": ["shimla", "simla"],
"tiruchirappalli": ["tiruchirappalli", "trichy", "trc"],
"tirupati": ["tirupati", "tpt"],
"mangalore": ["mangalore", "mangaluru", "man"],
"hubli": ["hubli", "hubli-dharwad", "dharwad"]
}
def standardize_state(text):
"""Standardize state names to canonical form"""
if not text:
return None
tokens = re.split(r"[,\s]+", normalize_text(text))
for token in tokens:
for standard, variants in STATE_MAPPING.items():
if token in variants:
return standard
return normalize_text(text)
def standardize_city(text):
"""Standardize city names to canonical form"""
if not text:
return None
tokens = re.split(r"[,\s]+", normalize_text(text))
for token in tokens:
for standard, variants in CITY_MAPPING.items():
if token in variants:
return standard
return normalize_text(text)
# =========================================================
# 1:N MATCHING LOGIC
# =========================================================
# def compare_any_match(list1: List, list2: List, field_type="text") -> Dict:
# """
# Check if any value in list1 matches any value in list2
# Returns: {"Result": "Match"/"No Match"/"missing value", "Confidence Score": 100, "Overall Similarity": 100/0}
# """
# if not list1 or not list2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# # Apply appropriate normalization based on field type
# if field_type == "pincode":
# vals1 = [validate_and_normalize_pincode(v) for v in list1 if v]
# vals2 = [validate_and_normalize_pincode(v) for v in list2 if v]
# elif field_type == "state":
# vals1 = [standardize_state(v) for v in list1 if v]
# vals2 = [standardize_state(v) for v in list2 if v]
# elif field_type == "city":
# vals1 = [standardize_city(v) for v in list1 if v]
# vals2 = [standardize_city(v) for v in list2 if v]
# else:
# vals1 = [normalize_text(v) for v in list1 if v]
# vals2 = [normalize_text(v) for v in list2 if v]
# # Filter out None values
# vals1 = [v for v in vals1 if v]
# vals2 = [v for v in vals2 if v]
# if not vals1 or not vals2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# # Check if any value matches
# for v1 in vals1:
# for v2 in vals2:
# if v1 == v2:
# return {"Result": "Match", "Confidence Score": 100, "Overall Similarity": 100}
# return {"Result": "No Match", "Confidence Score": 100, "Overall Similarity": 0}
# def compare_phone_any_match(list1: List, list2: List) -> Dict:
# """
# Check if any phone in list1 matches any phone in list2
# Returns: {"Result": "Match"/"No Match"/"missing value", "Confidence Score": 100, "Overall Similarity": 100/0}
# """
# if not list1 or not list2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# phones1 = [validate_and_normalize_phone(p) for p in list1 if p]
# phones2 = [validate_and_normalize_phone(p) for p in list2 if p]
# phones1 = [p for p in phones1 if p]
# phones2 = [p for p in phones2 if p]
# if not phones1 or not phones2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# for p1 in phones1:
# for p2 in phones2:
# if p1 == p2:
# return {"Result": "Match", "Confidence Score": 100, "Overall Similarity": 100}
# return {"Result": "No Match", "Confidence Score": 100, "Overall Similarity": 0}
# def compare_email_any_match(list1: List, list2: List) -> Dict:
# """
# Check if any email in list1 matches any email in list2
# Returns: {"Result": "Match"/"No Match"/"missing value", "Confidence Score": 100, "Overall Similarity": 100/0}
# """
# if not list1 or not list2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# emails1 = [validate_and_normalize_email(e) for e in list1 if e]
# emails2 = [validate_and_normalize_email(e) for e in list2 if e]
# emails1 = [e for e in emails1 if e]
# emails2 = [e for e in emails2 if e]
# if not emails1 or not emails2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# for e1 in emails1:
# for e2 in emails2:
# if e1 == e2:
# return {"Result": "Match", "Confidence Score": 100, "Overall Similarity": 100}
# return {"Result": "No Match", "Confidence Score": 100, "Overall Similarity": 0}
# # =========================================================
# # DIRECT COMPARISON (Exact Match Fields)
# # =========================================================
# def compare_exact(v1, v2) -> Dict:
# """
# Direct exact comparison for fields like GENDER, TAXID, LICENSEID, PASSPORTID, BIRTHDATE
# Returns: {"Result": "Match"/"No Match"/"missing value", "Confidence Score": 100, "Overall Similarity": 100/0}
# """
# if not v1 or not v2:
# return {"Result": "missing value", "Confidence Score": 100, "Overall Similarity": 0}
# v1_norm = normalize_text(v1)
# v2_norm = normalize_text(v2)
# if v1_norm == v2_norm:
# return {"Result": "Match", "Confidence Score": 100, "Overall Similarity": 100}
# else:
# return {"Result": "No Match", "Confidence Score": 100, "Overall Similarity": 0}
# # =========================================================
# # MATCHING RULES EVALUATION
# # =========================================================
# def evaluate_matching_rules(field_results: Dict) -> tuple:
# """Evaluate matching rules based on field results"""
# def rule_matched(*fields):
# return all(field_results.get(f, {}).get("Result") == "Match" for f in fields)
# RULES = [
# (("NAME", "PHONE", "EMAIL", "BIRTHDATE"),
# "Matched on Name, Phone Number, Email ID and DOB"),
# (("NAME", "CITY", "PHONE", "EMAIL", "ZIPCODE"),
# "Matched on Name, City, Phone Number, Email ID and Pincode"),
# (("NAME", "STATE", "PHONE", "EMAIL", "BIRTHDATE"),
# "Matched on Name, State, Phone Number, Email ID and DOB"),
# (("NAME", "PHONE", "ZIPCODE", "BIRTHDATE", "ADDRESSLINE"),
# "Matched on Name, Phone Number, Pincode, DOB and Address"),
# (("NAME", "CITY", "BIRTHDATE", "STATE", "ZIPCODE"),
# "Matched on Name, City, DOB, State and Pincode"),
# (("NAME", "CITY", "ZIPCODE", "BIRTHDATE"),
# "Matched on Name, City, Pincode and DOB"),
# (("NAME", "ADDRESSLINE", "EMAIL"),
# "Matched on Name, Address and Email ID"),
# (("NAME", "PHONE", "ZIPCODE", "BIRTHDATE"),
# "Matched on Name, Phone Number, Pincode and DOB"),
# (("NAME", "EMAIL", "BIRTHDATE"),
# "Matched on Name, Email ID and DOB"),
# (("NAME", "BIRTHDATE", "ADDRESSLINE", "PHONE"),
# "Matched on Name, DOB, Address and Phone Number"),
# (("NAME", "CITY", "ZIPCODE", "EMAIL"),
# "Matched on Name, City, Pincode and Email ID"),
# (("NAME", "ADDRESSLINE", "PHONE", "ZIPCODE"),
# "Matched on Name, Address, Phone Number and Pincode"),
# ]
# for fields, reason in RULES:
# if rule_matched(*fields):
# return "Match", reason
# return "No Match", "None of the defined matching rules were satisfied"
# =========================================================
# 1:N MATCHING LOGIC - Modified to return similarity scores
# =========================================================
def compare_any_match(list1: List, list2: List, field_type="text") -> float:
"""
Check if any value in list1 matches any value in list2
Returns: similarity score (0 or 100) or "missing value"
"""
if not list1 or not list2:
return 0
# Apply appropriate normalization based on field type
if field_type == "pincode":
vals1 = [validate_and_normalize_pincode(v) for v in list1 if v]
vals2 = [validate_and_normalize_pincode(v) for v in list2 if v]
elif field_type == "state":
vals1 = [standardize_state(v) for v in list1 if v]
vals2 = [standardize_state(v) for v in list2 if v]
elif field_type == "city":
vals1 = [standardize_city(v) for v in list1 if v]
vals2 = [standardize_city(v) for v in list2 if v]
else:
vals1 = [normalize_text(v) for v in list1 if v]
vals2 = [normalize_text(v) for v in list2 if v]
# Filter out None values
vals1 = [v for v in vals1 if v]
vals2 = [v for v in vals2 if v]
if not vals1 or not vals2:
return 0
# Check if any value matches
for v1 in vals1:
for v2 in vals2:
if v1 == v2:
return 100
return 0
def compare_phone_any_match(list1: List, list2: List) -> float:
"""
Check if any phone in list1 matches any phone in list2
Returns: similarity score (0 or 100) or "missing value"
"""
if not list1 or not list2:
return 0
phones1 = [validate_and_normalize_phone(p) for p in list1 if p]
phones2 = [validate_and_normalize_phone(p) for p in list2 if p]
phones1 = [p for p in phones1 if p]
phones2 = [p for p in phones2 if p]
if not phones1 or not phones2:
return 0
for p1 in phones1:
for p2 in phones2:
if p1 == p2:
return 100
return 0
def compare_email_any_match(list1: List, list2: List) -> float:
"""
Check if any email in list1 matches any email in list2
Returns: similarity score (0 or 100) or "missing value"
"""
if not list1 or not list2:
return 0
emails1 = [validate_and_normalize_email(e) for e in list1 if e]
emails2 = [validate_and_normalize_email(e) for e in list2 if e]
emails1 = [e for e in emails1 if e]
emails2 = [e for e in emails2 if e]
if not emails1 or not emails2:
return 0
for e1 in emails1:
for e2 in emails2:
if e1 == e2:
return 100
return 0
# =========================================================
# DIRECT COMPARISON - Modified to return similarity scores
# =========================================================
def compare_exact(v1, v2) -> float:
"""
Direct exact comparison for fields like GENDER, TAXID, LICENSEID, PASSPORTID, BIRTHDATE
Returns: similarity score (0 or 100) or "missing value"
"""
if not v1 or not v2:
return 0
v1_norm = normalize_text(v1)
v2_norm = normalize_text(v2)
if v1_norm == v2_norm:
return 100
else:
return 0
# =========================================================
# MATCHING RULES EVALUATION - Modified for new rule logic
# =========================================================
def evaluate_matching_rules(field_results: Dict) -> tuple:
"""Evaluate matching rules based on field results with similarity scores"""
def get_score(field_name):
"""Get numeric score from field result, handle missing values"""
result = field_results.get(field_name)
if result == "missing value" or result is None:
return 0
return round(float(result),2)
def rule_satisfied(conditions):
"""Check if all conditions in a rule are satisfied"""
for field, threshold in conditions:
if get_score(field) < threshold:
return False
return True
# Define all matching rules
RULES = [
([("CITY", 100), ("ADDRESSLINE", 85), ("NAME", 60), ("BIRTHDATE", 100)],
"CITY >= 100 and ADDRESS >= 85 and NAME >= 60 and BIRTHDATE >= 100"),
([("CITY", 100), ("ADDRESSLINE", 75), ("NAME", 60), ("BIRTHDATE", 100)],
"CITY >= 100 and ADDRESS >= 75 and NAME >= 60 and BIRTHDATE >= 100"),
([("ADDRESSLINE", 85), ("CITY", 100), ("NAME", 85), ("BIRTHDATE", 100)],
"ADDRESS >= 85 and CITY >= 100 and NAME >= 85 and BIRTHDATE >= 100"),
([("ADDRESSLINE", 85), ("NAME", 85), ("BIRTHDATE", 100), ("ZIPCODE", 100)],
"ADDRESS >= 85 and NAME >= 85 and BIRTHDATE >= 100 and ZIPCODE >= 100"),
([("ADDRESSLINE", 65), ("NAME", 60), ("BIRTHDATE", 100), ("ZIPCODE", 100)],
"ADDRESS >= 65 and NAME >= 60 and BIRTHDATE >= 100 and ZIPCODE >= 100"),
([("CITY", 100), ("ADDRESSLINE", 65), ("NAME", 60), ("BIRTHDATE", 100)],
"CITY >= 100 and ADDRESS >= 65 and NAME >= 60 and BIRTHDATE >= 100"),
([("ADDRESSLINE", 70), ("NAME", 60), ("BIRTHDATE", 100), ("COMPANYNAME", 60)],
"ADDRESS >= 70 and NAME >= 60 and BIRTHDATE >= 100 and EMPLOYERNAME >= 60"),
([("ADDRESSLINE", 75), ("CITY", 95), ("NAME", 50), ("ZIPCODE", 100)],
"ADDRESS >= 75 and CITY >= 95 and NAME >= 50 and ZIPCODE >= 100"),
([("ADDRESSLINE", 70), ("NAME", 60), ("PHONE", 100), ("ZIPCODE", 100)],
"ADDRESS >= 70 and NAME >= 60 and PHONE >= 100 and ZIPCODE >= 100"),
([("NAME", 85), ("ZIPCODE", 100), ("ADDRESSLINE", 60), ("CITY", 100)],
"NAME >= 85 AND ZIPCODE >=100 AND ADDRESS >=60 AND CITY >= 100"),
([("NAME", 85), ("CITY", 100), ("ADDRESSLINE", 60)],
"NAME >= 85 AND CITY >= 100 AND ADDRESS >= 60"),
([("NAME", 60), ("BIRTHDATE", 100), ("ZIPCODE", 100)],
"NAME >= 60 and BIRTHDATE >= 100 and ZIPCODE >= 100"),
([("CITY", 100), ("ADDRESSLINE", 75), ("NAME", 85)],
"CITY >= 100 and ADDRESS >= 75 and NAME >= 85"),
([("ADDRESSLINE", 75), ("NAME", 85), ("ZIPCODE", 100)],
"ADDRESS >= 75 and NAME >= 85 and ZIPCODE >= 100"),
([("CITY", 100), ("NAME", 60), ("BIRTHDATE", 100)],
"CITY >= 100 and NAME >= 60 and BIRTHDATE >= 100"),
([("CITY", 100), ("ADDRESSLINE", 70), ("LASTNAME", 85)],
"CITY >= 100 and ADDRESS >= 70 and LASTNAME >= 85"),
([("CITY", 100), ("ADDRESSLINE", 80), ("LASTNAME", 85)],
"CITY >= 100 and ADDRESS >= 80 and LASTNAME >= 85"),
([("CITY", 100), ("ADDRESSLINE", 90), ("LASTNAME", 85)],
"CITY >= 100 and ADDRESS >= 90 and LASTNAME >= 85"),
([("ADDRESSLINE", 80), ("LASTNAME", 85), ("ZIPCODE", 100)],
"ADDRESS >= 80 and LASTNAME >= 85 and ZIPCODE >= 100"),
([("ADDRESSLINE", 70), ("LASTNAME", 85), ("ZIPCODE", 100)],
"ADDRESS >= 70 and LASTNAME >= 85 and ZIPCODE >= 100"),
([("ADDRESSLINE", 90), ("LASTNAME", 85), ("ZIPCODE", 100)],
"ADDRESS >= 90 and LASTNAME >= 85 and ZIPCODE >= 100"),
([("NAME", 80), ("BIRTHDATE", 100), ("ZIPCODE", 100)],
"NAME >= 80 and BIRTHDATE >= 100 and ZIPCODE >= 100"),
([("NAME", 100), ("COMPANYNAME", 75), ("PHONE", 100)],
"NAME >= 100 and EMPLOYERNAME >= 75 and PHONE >= 100"),
([("NAME", 80), ("BIRTHDATE", 100), ("CITY", 100)],
"NAME >= 80 AND BIRTHDATE >= 100 AND CITY >= 100"),
([("NAME", 85), ("ZIPCODE", 100), ("ADDRESSLINE", 60)],
"NAME >= 85 AND ZIPCODE >=100 AND ADDRESS >=60"),
([("LASTNAME", 85), ("CITY", 100), ("ADDRESSLINE", 60)],
"LASTNAME >= 85 AND CITY >=100 AND ADDRESS >= 60"),
([("BIRTHDATE", 100), ("ZIPCODE", 100), ("ADDRESSLINE", 65)],
"BIRTHDATE >= 100 AND ZIPCODE >= 100 AND ADDRESS >= 65"),
([("BIRTHDATE", 100), ("CITY", 100), ("ADDRESSLINE", 65)],
"BIRTHDATE >= 100 AND CITY >= 100 AND ADDRESS >= 65"),
([("LASTNAME", 85), ("ZIPCODE", 100), ("ADDRESSLINE", 60)],
"LASTNAME >= 85 AND ZIPCODE >= 100 AND ADDRESS >= 60"),
([("NAME", 85), ("PHONE", 100)],
"NAME >= 85 AND PHONE >= 100"),
([("BIRTHDATE", 100), ("PHONE", 100)],
"BIRTHDATE >= 100 AND PHONE >= 100"),
([("BIRTHDATE", 100), ("NAME", 85)],
"BIRTHDATE >=100 AND NAME>=85"),
([("ADDRESSLINE", 60), ("TAXID", 100)],
"ADDRESS >= 60 and PAN >= 100"),
([("ADDRESSLINE", 60), ("LICENSEID", 100)],
"ADDRESS >= 60 and DRIVING_LICN_NO >= 100"),
([("BIRTHDATE", 75), ("PHONE", 100)],
"BIRTHDATE >= 75 and PHONE >= 100"),
([("BIRTHDATE", 75), ("TAXID", 100)],
"BIRTHDATE >= 75 and PAN >= 100"),
([("BIRTHDATE", 75), ("LICENSEID", 100)],
"BIRTHDATE >= 75 and DRIVING_LICN_NO >= 100"),
([("BIRTHDATE", 75), ("PASSPORTID", 100)],
"BIRTHDATE >= 75 and PASSPORT_NO >= 100"),
([("NAME", 60), ("PASSPORTID", 100)],
"NAME >= 60 and PASSPORT_NO >= 100"),
([("NAME", 60), ("LICENSEID", 100)],
"NAME >= 60 and DRIVING_LICN_NO >= 100"),
([("NAME", 60), ("TAXID", 100)],
"NAME >= 60 and PAN >= 100"),
([("PHONE", 100)], "PHONE >= 100"),
([("LICENSEID", 100)], "DRIVING_LICN_NO >= 100"),
([("PASSPORTID", 100)], "PASSPORT_NO >= 100"),
([("TAXID", 100)], "PAN >= 100"),
([("EMAIL", 100)], "EMAIL >= 100"),
]
# Check each rule in order
for conditions, reason in RULES:
if rule_satisfied(conditions):
return "Match", reason
return "No Match", "None of the defined matching rules were satisfied"
# =========================================================
# PATTERN-BASED FIELD MATCHING (0 or 100 logic)
# =========================================================
def apply_pattern_matching_logic(field_name: str, score) -> float:
"""
Apply 0 or 100 logic for pattern-based fields
For DOB, PHONE, EMAIL, ZIPCODE, etc.: if match -> 100, else -> 0
For other fields: return the actual similarity score
"""
# Pattern fields that should be 0 or 100
PATTERN_FIELDS = {
"BIRTHDATE", "PHONE", "EMAIL", "ZIPCODE",
"TAXID", "LICENSEID", "PASSPORTID", "GENDER"
}
# If it's a missing value, keep it as is
if score == "missing value":
return 0
# If it's a pattern field, apply 0 or 100 logic
if field_name in PATTERN_FIELDS:
return 100 if score >= 100 else 0
# For non-pattern fields, return the actual score
return score