File size: 2,464 Bytes
31f24ea
 
 
 
 
e648ec3
 
 
31f24ea
e648ec3
31f24ea
e648ec3
31f24ea
e648ec3
31f24ea
 
 
 
e648ec3
31f24ea
e648ec3
31f24ea
e648ec3
31f24ea
 
 
 
 
 
 
 
 
 
 
 
 
 
e648ec3
31f24ea
 
 
 
 
 
 
 
 
 
e648ec3
 
 
31f24ea
 
 
 
 
 
e648ec3
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
import re

def extract_bkt_items(pdf_lines):
    parsed_items = []
    seen_identifiers = set()
    port_destination, country_destination = "", ""
    is_original_page = True
    current_hs_code, current_license_no, current_license_date = "", "", ""
    
    for line in pdf_lines:
        line_str = line.strip()
        if not line_str: continue
        lower_line = line_str.lower()
        if "duplicate for" in lower_line or "triplicate for" in lower_line or "extra copy" in lower_line:
            if "original for recipient" not in lower_line:
                is_original_page = False
        if "original for recipient" in lower_line:
            is_original_page = True
        if not is_original_page: continue

        if "final destination" in lower_line:
            parts_dest = line_str.split(":")
            port_destination = parts_dest[-1].strip().upper() if len(parts_dest) > 1 else ""

        hs_match = re.search(r'\b(401[1236]\d{4}|843[123]\d{4})\b', line_str)
        if hs_match:
            current_hs_code = hs_match.group(1)
            continue

        if "sub-total" in lower_line or "sub total" in lower_line:
            nums = re.findall(r'[\d,]+\.\d{2,3}|\b\d+\b', line_str)
            if nums and current_hs_code:
                qty = nums[0] if len(nums) > 0 else ""
                taxable_val = nums[-3] if len(nums) >= 3 else (nums[1] if len(nums) > 1 else "")
                igst_amt = nums[-2] if len(nums) >= 2 else ""
                
                unique_key = f"{current_hs_code}_{qty}_{taxable_val}"
                if unique_key in seen_identifiers: continue
                seen_identifiers.add(unique_key)
                
                item_dict = {
                    "raw_parts": line_str.split(),
                    "line_text": line_str.upper(),
                    "hs_code": current_hs_code,
                    "license_no": current_license_no,
                    "license_date": current_license_date,
                    "quantity": qty,
                    "value": taxable_val,
                    "taxable_value": taxable_val,
                    "igst_rate": "18.00",
                    "igst_amt": igst_amt,
                    "nums": nums,
                    "port_destination": port_destination,
                    "country_destination": country_destination
                }
                parsed_items.append(item_dict)
                current_hs_code = ""
    return parsed_items