import streamlit as st import pandas as pd import numpy as np import joblib import os import logging from datetime import datetime from openai import AzureOpenAI from dotenv import load_dotenv import uuid # Configure logging logging.basicConfig(level=logging.ERROR) # Load environment variables load_dotenv() # Initialize Azure OpenAI GPT client azure_api_key = os.getenv("AZURE_OPENAI_API_KEY") azure_endpoint = os.getenv("AZURE_OPENAI_ENDPOINT") client = AzureOpenAI(azure_endpoint=azure_endpoint, api_key=azure_api_key, api_version="2024-02-01") # Set default values for missing columns defaults = { 'policy_number': 176450963679054, } # Function to analyze claim details using few-shot prompting def analyze_claim_with_few_shot_prompt(claim_data): """ Analyze the claim data using few-shot prompting to classify it as Fraudulent or Non-Fraudulent. Parameters: claim_data (dict): Dictionary containing claim details. Returns: tuple: Classification (Fraudulent/Non-Fraudulent) and reasoning as a string. """ examples = """ ### Examples: 1. **Scenario**: The client stated that the accident happened at 3 AM in a remote area, but the vehicle showed no signs of damage. **Outcome**: Fraudulent **Reasoning**: The timing and location are suspicious, and the lack of vehicle damage contradicts the claim. 2. **Scenario**: The client provided consistent details about a minor collision, with a supporting police report and witness statements. **Outcome**: Non-Fraudulent **Reasoning**: All evidence aligns with the claim details. 3. **Scenario**: The client filed a claim for vehicle theft but could not produce a police report and provided vague answers about the incident. **Outcome**: Fraudulent **Reasoning**: Lack of supporting documentation and evasive responses indicate possible fraud. 4. **Scenario**: The client reported a severe collision with clear photographic evidence and a medical report for injuries. **Outcome**: Non-Fraudulent **Reasoning**: Strong documentation supports the claim. 5. **Scenario**: The client reported an accident I was driving at 2 A.M. there was a lot of traffic and few people just came in the way on the highway I was driving at the speed of 90. **Outcome**: Fraudulent **Reasoning**: Suspicious timing, case do not align with common sense. """ prompt = f""" As an auto insurance fraud detection expert working at a large automobile insurance company, your job is to help detect fraud in auto insurance claims filed by customers. ### Task: Your job is to analyze customer and agent call log details and additional claim metadata to classify whether the claim is fraudulent or not. ### Provided details: - **Policy Number**: {claim_data['policy_number']} - **Claim Date**: {claim_data['claim_date']} - **Claim Amount**: {claim_data['claim_amount']} - **Vehicle Info**: {claim_data['vehicle_info']} - **Insurance Purchase Date**: {claim_data['insurance_purchase_date']} - **Insurance Renewal Date**: {claim_data['insurance_renewal_date']} - **Claim Details**: {claim_data['claim_details']} - **Incident Location**: {claim_data['incident_location']} {examples} ### Indicators for Fraud Detection: Look for the following indicators when analyzing the call logs and claim metadata: - Inconsistent or contradictory statements made by customers or agents. - Identify any suspicious behaviors or patterns, such as overly detailed descriptions, repeated claims, or unusual urgency. - Pay attention to any discrepancies between the call logs and the corresponding claim documents. For example, a claim is for a windshield but the incident is of being rear-ended. - Note if there are any red flags, such as customers hesitating or changing their stories during the conversation. - Question cases that do not align with common sense, such as a customer claiming an accident was caused by heavy traffic at midnight. - Suspicious timing or location of the incident - Any past history of dubious claims ### Step-by-Step Instructions: 1.Initial Context Understanding: Read and understand the call log details and the provided metadata. 2.Detecting Patterns: Identify specific phrases, patterns, or inconsistencies that may indicate fraudulent activity. 3.Classification: Classify the claim as “Fraudulent” or “Non-Fraudulent” based on your analysis. Provide a detailed explanation supporting your classification. Highlight the specific parts of the call log and metadata that influenced your decision. ### Call Log and Metadata to Analyze: {claim_data['call_log']} Based on the above details, classify the claim below: """ try: response = client.chat.completions.create( model="gpt-35-turbo", messages=[{"role": "user", "content": prompt}], temperature=0.0 ) response_text = response.choices[0].message.content.strip() if "Classification:" in response_text: # Find the first occurrence of classification classification_index = response_text.find("Classification:") classification_line = response_text[classification_index:].split('\n')[0] if "Non-Fraudulent" in classification_line: classification = "Non-Fraudulent" elif "Fraudulent" in classification_line: classification = "Fraudulent" elif "Needs Expert Opinion" in classification_line or "not possible to classify" in classification_line: classification = "Needs Expert Opinion" reasoning = response_text.split('Reasoning: ')[-1].strip() return classification, reasoning, response_text except Exception as e: logging.error(f"Classification of Claim for Policy {claim_data['policy_number']} Failed due to Error: {e}") return "Error", "Unable to classify claim, Retry!" # Streamlit UI st.set_page_config(page_title="Vehicle Insurance Fraud Detection", layout="wide") st.markdown( """ """, unsafe_allow_html=True) # Custom title in the center and with navy blue color st.markdown('
Vehicle Insurance Fraud Detection
', unsafe_allow_html=True) # Input form for structured data st.header('Claim Details') policy_number= st.text_input('Policy Number',help='Enter the alphanumeric policy number (e.g., ABC12345).') incident_date= st.date_input('Claim Date', value=datetime(2024, 1, 1)) total_claim_amount = st.number_input('Claim Amount', min_value=0, max_value=100000, value=5000) insurance_purchase_date= st.date_input('Insurance Purchase Date', value=datetime(2023, 4, 1)) insurance_renewal_date= st.date_input('Insurance Purchase Date', value=datetime(2025, 4, 1)) incident_location= st.text_input('Incident Location', value ='Mumbai') auto_year = st.number_input('Auto Year', min_value=1990, max_value=2024, value=2015) # Notes Section st.header('Notes') customer_notes = st.text_area('Customer Notes', help='Enter any specific details shared by the customer.') agent_notes = st.text_area('Agent Notes', help='Enter any observations or details from the agent.') # Convert incident_date to a timestamp incident_timestamp = datetime.combine(incident_date, datetime.min.time()).timestamp() # Compile data claim_data = { 'policy_number': policy_number, 'claim_date': str(incident_date), 'claim_amount': total_claim_amount, 'vehicle_info': f'{auto_year}', 'insurance_purchase_date': str(insurance_purchase_date), 'insurance_renewal_date': str(insurance_renewal_date), 'claim_details': customer_notes, 'incident_location': incident_location, 'call_log': customer_notes + "\n" + agent_notes } # Streamlit UI - Button and Classification Styling if st.button('Classify Claim'): with st.spinner('Processing...'): classification, reasoning, response = analyze_claim_with_few_shot_prompt(claim_data) # Conditional Classification Text Color: Red for "Fraudulent", Green otherwise if classification == "Fraudulent": classification_color = 'red' else: classification_color = 'green' # Display Classification with conditional color st.markdown(f"### **Classification:** {classification}", unsafe_allow_html=True) st.markdown("### **Reasoning:**", unsafe_allow_html=True) st.write(reasoning) #st.markdown("### ** Response :: **") #st.write(response)