Spaces:
Running on Zero
Running on Zero
File size: 6,828 Bytes
31f24ea bd75d18 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 | import re
import io
import pdfplumber
def apply_value_replacement(extracted_text, mapping_str):
if not extracted_text or not mapping_str or "=" not in mapping_str:
return extracted_text
text_clean = str(extracted_text).strip()
pairs = [p.strip() for p in mapping_str.split(",") if "=" in p]
for pair in pairs:
parts = pair.split("=")
if len(parts) == 2:
find_kw = parts[0].strip()
replace_kw = parts[1].strip()
if text_clean.lower() == find_kw.lower():
return replace_kw
elif find_kw.lower() in text_clean.lower():
pattern = re.compile(re.escape(find_kw), re.IGNORECASE)
return pattern.sub(replace_kw, text_clean)
return text_clean
def apply_rule_filter(raw_text, mode, stop_kw, flt, keyword=""):
if flt == "Exact Keyword Paste (If Found)":
target_check = stop_kw.strip() if stop_kw and str(stop_kw).strip() else keyword.strip()
if target_check and target_check.lower() in str(raw_text).lower():
return target_check
return target_check if target_check else ""
if not raw_text: return ""
text = raw_text.strip()
if text.startswith(":"): text = text[1:].strip()
if keyword and ("consignee" in keyword.lower() or "buyer" in keyword.lower()):
return text
if mode == "Word Position" or mode.startswith("Word "):
w_num = int(stop_kw.strip()) if stop_kw and str(stop_kw).strip().isdigit() else 1
parts = text.split()
text = parts[w_num - 1].strip() if len(parts) >= w_num else ""
elif mode == "After Word" and stop_kw:
if "=" not in stop_kw and stop_kw.lower() in text.lower():
start_idx = text.lower().find(stop_kw.lower()) + len(stop_kw)
text = text[start_idx:].strip()
if text.startswith(":"): text = text[1:].strip()
elif mode == "Between Keywords" and stop_kw:
if "=" not in stop_kw and stop_kw.lower() in text.lower():
text = text.lower().split(stop_kw.lower())[0].strip()
elif mode == "Exact Word":
parts = text.split()
text = parts[0].strip() if parts else ""
elif mode == "Full Line":
text = text.split("\n")[0].strip()
if flt in ["Text Inside Parentheses ()", "Inside Parentheses ()"]:
bracket_match = re.search(r'\((.*?)\)', text)
text = bracket_match.group(1).strip() if bracket_match else text.strip()
elif flt == "Container Number (ISO Format)":
cntr_match = re.search(r'\b[A-Za-z]{4}\s*\d{7}\b', text)
text = cntr_match.group(0).replace(" ", "") if cntr_match else text.strip()
elif flt == "Remove All Spaces":
text = text.replace(" ", "").strip()
elif flt == "Numbers Only":
nums = re.findall(r'[\d,.]+', text)
text = nums[0].strip() if nums else ""
elif flt == "Letters Only":
text = re.sub(r'[^A-Za-z\s]', '', text).strip()
elif flt == "Clean Date (DD/MM/YYYY)":
d_match = re.search(r'\b\d{2}[./-]\d{2}[./-]\d{4}\b', text)
text = d_match.group(0).replace(".", "/").replace("-", "/") if d_match else text.strip()
if stop_kw and "=" in stop_kw: text = apply_value_replacement(text, stop_kw)
if flt and "=" in flt: text = apply_value_replacement(text, flt)
return text.strip()
def extract_header_value(pdf_lines, pdf_text, keyword, position, mode, stop_kw, filter_type, field_label="", pdf_bytes=None):
raw_t = ""
if keyword and pdf_bytes and ("Box" in str(position) or "डब्बा" in str(position)):
try:
with pdfplumber.open(io.BytesIO(pdf_bytes)) as pdf:
page = pdf.pages[0]
words = page.extract_words()
kw_word = None
for w in words:
if keyword.lower() in w['text'].lower():
kw_word = w
break
if kw_word:
kw_x0 = kw_word['x0']
kw_y0 = kw_word['top']
box_x0 = kw_x0 - 5
box_x1 = kw_x0 + 300
box_y0 = kw_y0 - 2
box_y1 = kw_y0 + 100
block_words = []
for w in words:
if box_x0 <= w['x0'] <= box_x1 and box_y0 <= w['top'] <= box_y1:
if keyword.lower() not in w['text'].lower():
block_words.append(w)
if block_words:
sorted_words = sorted(block_words, key=lambda x: (round(x['top']/6)*6, x['x0']))
extracted_phrase = " ".join([w['text'] for w in sorted_words]).strip()
if extracted_phrase:
extracted_phrase = re.sub(r'^[:\-\s]+', '', extracted_phrase)
return apply_rule_filter(extracted_phrase, mode, stop_kw, filter_type, keyword)
except Exception:
pass
if filter_type == "Exact Keyword Paste (If Found)":
raw_t = pdf_text
elif keyword:
for line_i, line in enumerate(pdf_lines):
if keyword.lower() in line.lower():
if "Right" in str(position) or position == "Right (आगे)":
start_idx = line.lower().find(keyword.lower()) + len(keyword)
raw_t = line[start_idx:].strip()
if raw_t.startswith(":"): raw_t = raw_t[1:].strip()
if raw_t: break
elif "Below" in str(position) or position == "Below (नीचे)":
if line_i + 1 < len(pdf_lines):
raw_t = pdf_lines[line_i + 1].strip()
if raw_t: break
elif position == "2 Lines Below":
if line_i + 2 < len(pdf_lines):
raw_t = pdf_lines[line_i + 2].strip()
if raw_t: break
else:
raw_t = pdf_text
if position == "📦 Extract Inside Box (डब्बे के अंदर का टेक्स्ट)":
return raw_t.strip()
return apply_rule_filter(raw_t, mode, stop_kw, filter_type, keyword)
def detect_igst_status(pdf_text, lut_keywords="", paid_keywords=""):
if not pdf_text: return "UNKNOWN"
text_lower = pdf_text.lower()
custom_lut_kws = [k.strip().lower() for k in lut_keywords.split(",") if k.strip()]
for kw in custom_lut_kws:
if kw in text_lower: return "LUT"
custom_paid_kws = [k.strip().lower() for k in paid_keywords.split(",") if k.strip()]
for kw in custom_paid_kws:
if kw in text_lower: return "P"
return "UNKNOWN" |