File size: 7,259 Bytes
fe6855c
f404c1e
 
 
 
 
fe6855c
 
f404c1e
 
fe6855c
 
 
 
f404c1e
 
 
 
 
 
 
fe6855c
 
f404c1e
 
 
fe6855c
 
 
 
f404c1e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
102b81e
f404c1e
 
 
 
102b81e
f404c1e
 
 
 
fe6855c
f404c1e
 
102b81e
 
 
fe6855c
f404c1e
 
 
 
fe6855c
f404c1e
 
 
 
 
 
 
 
 
 
fe6855c
 
 
 
 
 
 
 
 
f404c1e
fe6855c
 
 
 
f404c1e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
122ab6b
f404c1e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155

import json
import os
from pathlib import Path
import re
from typing import List, Optional
from app.services.duplicate_checker import check_duplicate
from app.services.email_reader import parse_email_bytes
from app.services.gemeni_classification import classify_email_gemeni, extract_key_number_with_llm, extract_text_from_attachment
from config import settings


async def process_single_email(file_content: bytes, filename: str) -> Optional[dict]:
    """Processes a single email content."""
    # Load customizable priority rules
    rules = load_priority_rules()  # Expected to return a dict with "priority_rules" key
    priority_config = rules.get("priority_rules", {})
    use_priority = priority_config.get("is_prioritization_extraction", False)
    all_extracted_numbers = []  # This will hold the combined results from all attachments

    # Parse the email into its components
    email_data = parse_email_bytes(file_content, filename)
    if email_data:
        # attachment_text = ""
        # for attachment in email_data["attachments"]:
        #     attachment_text += extract_text_from_attachment(attachment["content"], attachment["filename"])

        email_chain_text = email_data["email_chain_text"]
        email_body_text = email_data["body"]

        # Process attachments: extract text and numerical fields
        attachment_text = ""
        extracted_numbers = []
        for attachment in email_data.get("attachments", []):
            text = extract_text_from_attachment(attachment["content"], attachment["filename"])
            attachment_text += text + "\n"
            # extracted_numbers.extend(extract_key_number_with_llm(text,rules))
            extracted_numbers_json = extract_key_number_with_llm(text, rules)
            if extracted_numbers_json:
                try:
                    parsed_result = json.loads(extracted_numbers_json)  # Convert JSON string to Python object
                    # Check if the parsed result is a list; if so, merge it into our overall list.
                    if isinstance(parsed_result, list):
                        all_extracted_numbers.extend(parsed_result)
                    else:
                        all_extracted_numbers.append(parsed_result)
                except Exception as e:
                    print("Error parsing JSON result from LLM:", e)

        # Choose classification logic based on priority configuration
        if use_priority:
            # -- Priority Based Extraction Logic --
            identification_order = priority_config.get("request_type_identification", {}).get("order", [])
            classification_source = ""
            if "email_content" in identification_order and email_body_text.strip():
                classification_source = email_body_text
            elif "document_content" in identification_order and attachment_text.strip():
                classification_source = attachment_text
            else:
                classification_source = email_body_text  # default fallback
            
            primary_result = classify_email_gemeni(email_data["subject"], classification_source, rules)

            # special condition to check if priority is email content and email has multi thread then we 
            # have to compare confidence score with primary confidence score 
            if "email_content" in identification_order:
                email_chain_result = classify_email_gemeni(email_data["subject"], email_chain_text, rules) if email_chain_text else ("Unknown", "Unknown", "0")
                email_chain_confidence = float(email_chain_result[2])
                primary_email_confidence = float(primary_result[2])
                if email_chain_confidence > primary_email_confidence and email_chain_confidence:
                    primary_result  = email_chain_result

        else:
            # 1. Separate Classification:
            document_result = classify_email_gemeni(email_data["subject"], attachment_text, rules) if attachment_text else ("Unknown", "Unknown", "0")
            email_chain_result = classify_email_gemeni(email_data["subject"], email_chain_text, rules) if email_chain_text else ("Unknown", "Unknown", "0")
            primary_email_result = classify_email_gemeni(email_data["subject"], email_body_text, rules)

            # 2. Confidence Score Comparison:
            document_confidence = float(document_result[2])
            email_chain_confidence = float(email_chain_result[2])
            primary_email_confidence = float(primary_email_result[2])

            primary_result  = primary_email_result  # Default to email body
            if document_confidence > primary_email_confidence and document_confidence > email_chain_confidence:
                primary_result  = document_result
            elif email_chain_confidence > primary_email_confidence and email_chain_confidence > document_confidence:
                primary_result  = email_chain_result

        # Extract classification results
        request_type = primary_result[0]
        sub_request_type = primary_result[1]
        confidence_score = primary_result[2]

        duplicate_flag, duplicate_reason = check_duplicate(email_data["body"])
        email_obj = {
            "sender": email_data["sender"],
            "subject": email_data["subject"],
            "request_type": request_type,
            "sub_request_type": sub_request_type,
            "confidence_score": confidence_score,
            "duplicate_flag": duplicate_flag,
            "extracted_numbers_list": all_extracted_numbers
        }
        return email_obj
    else:
        print(f"Parsing failed for file: {filename}")
        return None
    

def load_priority_rules() -> dict:
    """
    Loads the priority rules from a JSON file.
    The JSON file should be located at 'config/rules.json'.
    If the file does not exist, a default rules dictionary is returned.
    
    Expected JSON structure example:
    {
      "priority_rules": {
        "is_prioritization_extraction": true,
        "request_type_identification": {
          "order": ["email_content", "document_content"],
          "fallback": "document_content"
        },
        "numerical_field_extraction": {
          "preferred_source": ["attachments"],
          "fallback": "email_body"
        }
      }
    }
    """
    # Get the rules directory and filename from environment variables
    RULES_FILENAME =  settings.settings.ALLOWED_PRIORITY_RULES_FILENAME
    RULES_DIR = Path(settings.settings.DATA_DIRECTORY_ATTACHMENTS_PATH)
    # Build the full file path using pathlib
    RULES_FILE_PATH = Path(RULES_DIR) / RULES_FILENAME
    
    if os.path.exists(RULES_FILE_PATH):
        with open(RULES_FILE_PATH, "r") as file:
            return json.load(file)
    
    # Return default rules if file doesn't exist
    return {
        "priority_rules": {
            "is_prioritization_extraction": False,
            "request_type_identification": {
                "order": ["email_content", "document_content"],
                "fallback": "document_content"
            },
            "numerical_field_extraction": {
                "preferred_source": ["attachments"],
                "fallback": "email_body"
            }
        }
    }