GGUF
conversational
Sanjay1905 commited on
Commit
d18bebb
Β·
verified Β·
1 Parent(s): e7e3654

Upload 3 files

Browse files
Files changed (3) hide show
  1. finalsite.py +1254 -0
  2. requirements.txt +10 -0
  3. unsloth.F16.gguf +3 -0
finalsite.py ADDED
@@ -0,0 +1,1254 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import cv2
3
+ import numpy as np
4
+ import gradio as gr
5
+ from ultralytics import YOLO
6
+ from pdf2image import convert_from_path
7
+ from PIL import Image
8
+ import easyocr
9
+ import uuid
10
+ import re
11
+ import difflib
12
+ import math
13
+ from faker import Faker
14
+ import datetime
15
+ import random
16
+ import csv
17
+ import tempfile
18
+ from io import StringIO
19
+
20
+ # Initialize Faker
21
+ fake = Faker()
22
+
23
+ # Conditional import for LLM
24
+ try:
25
+ from llama_cpp import Llama
26
+ LLAMA_AVAILABLE = True
27
+ except ImportError:
28
+ print("Warning: llama_cpp not available. LLM functionality will be disabled.")
29
+ LLAMA_AVAILABLE = False
30
+
31
+ # --- Configuration ---
32
+ # Folders for temporary files and results
33
+ UPLOAD_FOLDER = 'static/uploads/'
34
+ RESULTS_FOLDER = 'static/results/'
35
+
36
+ # --- Model Paths (Update these paths if necessary) ---
37
+ CUSTOM_MODEL_PATH = 'best.pt'
38
+ PRETRAINED_MODEL_PATH = 'yolov10s.pt'
39
+ SIGNATURE_MODEL_PATH = 'yolov8s.pt'
40
+ LLAMA_MODEL_PATH = "unsloth.F16.gguf"
41
+
42
+ # Detection Parameters
43
+ YOLO_CONFIDENCE_THRESHOLD = 0.5
44
+ OCR_CONFIDENCE_THRESHOLD = 0.5
45
+
46
+ # Create directories if they don't exist
47
+ os.makedirs(UPLOAD_FOLDER, exist_ok=True)
48
+ os.makedirs(RESULTS_FOLDER, exist_ok=True)
49
+
50
+ # --- Global Model Placeholders ---
51
+ custom_model, pretrained_model, signature_model, reader, llama_model = None, None, None, None, None
52
+
53
+ def load_models():
54
+ """
55
+ Loads all AI models into the global scope. This function is called on the first
56
+ analysis request to avoid startup conflicts. It ensures models are only loaded once.
57
+ """
58
+ global custom_model, pretrained_model, signature_model, reader, llama_model
59
+ # If models are already loaded, do nothing.
60
+ if reader is not None and (llama_model is not None or not LLAMA_AVAILABLE):
61
+ print("Models already loaded.")
62
+ return
63
+ print("=== Loading Models (this may take a moment) ===")
64
+
65
+ # Helper function to check for model files
66
+ def check_model_path(path, name):
67
+ if not os.path.exists(path):
68
+ print(f"βœ— WARNING: {name} model not found at '{path}'. The application may not function correctly.")
69
+ return False
70
+ return True
71
+
72
+ # YOLO Models
73
+ if check_model_path(CUSTOM_MODEL_PATH, "Custom YOLO"):
74
+ try:
75
+ custom_model = YOLO(CUSTOM_MODEL_PATH)
76
+ print("βœ“ Custom YOLO model loaded.")
77
+ except Exception as e:
78
+ print(f"βœ— Error loading custom model: {e}")
79
+ if check_model_path(PRETRAINED_MODEL_PATH, "Pre-trained YOLO"):
80
+ try:
81
+ pretrained_model = YOLO(PRETRAINED_MODEL_PATH)
82
+ print("βœ“ Pre-trained YOLO model loaded.")
83
+ except Exception as e:
84
+ print(f"βœ— Error loading pre-trained model: {e}")
85
+ if check_model_path(SIGNATURE_MODEL_PATH, "Signature YOLO"):
86
+ try:
87
+ signature_model = YOLO(SIGNATURE_MODEL_PATH)
88
+ print("βœ“ Signature YOLO model loaded.")
89
+ except Exception as e:
90
+ print(f"βœ— Error loading signature model: {e}")
91
+
92
+ # OCR Model
93
+ try:
94
+ reader = easyocr.Reader(['en'], gpu=True)
95
+ print("βœ“ EasyOCR model loaded.")
96
+ except Exception as e:
97
+ print(f"βœ— Error loading EasyOCR: {e}. Text detection will be unavailable.")
98
+
99
+ # LLM Model - Only load if available
100
+ if LLAMA_AVAILABLE and check_model_path(LLAMA_MODEL_PATH, "LLM"):
101
+ try:
102
+ llama_model = Llama(
103
+ model_path=LLAMA_MODEL_PATH,
104
+ n_gpu_layers=-1, n_ctx=4096, chat_format="llama-3", verbose=False
105
+ )
106
+ print("βœ“ LLM model loaded.")
107
+ except Exception as e:
108
+ print(f"βœ— Error loading LLM model: {e}. Text analysis will be unavailable.")
109
+ print("=== All Models Initialized ===")
110
+
111
+ # YOLO Class Mappings
112
+ CUSTOM_CLASS_NAMES = {0: 'face', 1: 'qr', 2: 'signature'}
113
+ PRETRAINED_CLASS_MAP = {0: 'face'}
114
+
115
+ # --- Core Detection & Processing Functions ---
116
+ def detect_visual_pii(image_data):
117
+ """Runs the three-stage YOLO detection on a single image."""
118
+ all_boxes = []
119
+ all_classes = []
120
+ if custom_model is None:
121
+ print("Custom model not available for visual detection")
122
+ return all_boxes, all_classes
123
+
124
+ # Pass 1: Custom Model
125
+ custom_results = custom_model.predict(source=image_data, conf=YOLO_CONFIDENCE_THRESHOLD, verbose=False)[0]
126
+ detected_custom_classes = {CUSTOM_CLASS_NAMES[int(cls)] for cls in custom_results.boxes.cls}
127
+ for box, cls in zip(custom_results.boxes.xyxy.cpu().numpy().astype(int), custom_results.boxes.cls):
128
+ all_boxes.append(box)
129
+ all_classes.append(CUSTOM_CLASS_NAMES[int(cls)])
130
+
131
+ # Pass 2: Pre-trained Model (Face Fallback)
132
+ if 'face' not in detected_custom_classes and pretrained_model is not None:
133
+ print(" Custom model missed 'face'. Trying pre-trained model as fallback.")
134
+ pretrained_results = pretrained_model.predict(source=image_data, conf=YOLO_CONFIDENCE_THRESHOLD, verbose=False)[0]
135
+ for box in pretrained_results.boxes:
136
+ if int(box.cls[0]) in PRETRAINED_CLASS_MAP:
137
+ all_boxes.append(box.xyxy.cpu().numpy().astype(int)[0])
138
+ all_classes.append("face (fallback)")
139
+
140
+ # Pass 3: Specialized Model (Signature Fallback)
141
+ if 'signature' not in detected_custom_classes and signature_model is not None:
142
+ print(" Custom model missed 'signature'. Trying specialized signature model as fallback.")
143
+ signature_results = signature_model.predict(source=image_data, conf=YOLO_CONFIDENCE_THRESHOLD, verbose=False)[0]
144
+ for box in signature_results.boxes:
145
+ all_boxes.append(box.xyxy.cpu().numpy().astype(int)[0])
146
+ all_classes.append("signature (fallback)")
147
+
148
+ return all_boxes, all_classes
149
+
150
+ # --- OCR + LLM Functions ---
151
+ def calculate_distance(bbox1, bbox2):
152
+ """Calculates the Euclidean distance between the centers of two bounding boxes."""
153
+ c1_x = (bbox1[0] + bbox1[2]) / 2
154
+ c1_y = (bbox1[1] + bbox1[3]) / 2
155
+ c2_x = (bbox2[0] + bbox2[2]) / 2
156
+ c2_y = (bbox2[1] + bbox2[3]) / 2
157
+ return math.sqrt((c2_x - c1_x)**2 + (c2_y - c1_y)**2)
158
+
159
+ def refine_pii_flags(ocr_results, isolation_threshold=150):
160
+ """Post-processing step to unmark short, isolated PII detections."""
161
+ pii_indices = [i for i, result in enumerate(ocr_results) if result["is_pii"]]
162
+ if len(pii_indices) <= 1:
163
+ return ocr_results
164
+ indices_to_unmark = []
165
+ for i in pii_indices:
166
+ current_result = ocr_results[i]
167
+ normalized_text = re.sub(r'[^a-zA-Z0-9]', '', current_result["text"])
168
+
169
+ if len(normalized_text) <= 3:
170
+ min_dist_to_neighbor = float('inf')
171
+
172
+ for j in pii_indices:
173
+ if i == j:
174
+ continue
175
+
176
+ other_result = ocr_results[j]
177
+ dist = calculate_distance(current_result["bbox"], other_result["bbox"])
178
+ if dist < min_dist_to_neighbor:
179
+ min_dist_to_neighbor = dist
180
+
181
+ if min_dist_to_neighbor > isolation_threshold:
182
+ print(f" - Refining PII: Unmarking short ('{current_result['text']}') and isolated (min_dist: {min_dist_to_neighbor:.2f}px) PII.")
183
+ indices_to_unmark.append(i)
184
+
185
+ for i in indices_to_unmark:
186
+ ocr_results[i]["is_pii"] = False
187
+
188
+ return ocr_results
189
+
190
+ def parse_pii_output(generated_text):
191
+ """Parse the new curly braces format PII output"""
192
+ pii_list = []
193
+ try:
194
+ match = re.search(r'\{([^}]*)\}', generated_text)
195
+ if match:
196
+ content = match.group(1)
197
+ items = re.findall(r'"([^"]*)"', content)
198
+ pii_list = [item.strip() for item in items if item.strip()]
199
+ except Exception as e:
200
+ print(f"Error parsing PII output: {e}")
201
+ pii_list = []
202
+ return pii_list
203
+
204
+ def normalize_text(text):
205
+ """Comprehensive text normalization for better matching"""
206
+ if not text:
207
+ return ""
208
+ normalized = re.sub(r'[.,;:!?()"\'\-_/\\]', '', text)
209
+ ocr_corrections = {
210
+ '0': 'o', 'O': '0', '1': 'l', 'l': '1', '5': 's', 'S': '5',
211
+ '8': 'b', 'B': '8', 'rn': 'm', 'RN': 'M', 'vv': 'w', 'VV': 'W',
212
+ 'cl': 'd', 'CL': 'D',
213
+ }
214
+ for wrong, correct in ocr_corrections.items():
215
+ normalized = normalized.replace(wrong, correct)
216
+ normalized = ' '.join(normalized.split()).lower()
217
+ return normalized
218
+
219
+ def fuzzy_match_score(text1, text2, threshold=0.8):
220
+ """Calculate fuzzy matching score between two strings"""
221
+ if not text1 or not text2:
222
+ return False
223
+ return difflib.SequenceMatcher(None, text1.lower(), text2.lower()).ratio() >= threshold
224
+
225
+ def levenshtein_distance(s1, s2):
226
+ """Calculate Levenshtein distance between two strings"""
227
+ if len(s1) < len(s2):
228
+ return levenshtein_distance(s2, s1)
229
+ if len(s2) == 0:
230
+ return len(s1)
231
+ previous_row = list(range(len(s2) + 1))
232
+ for i, c1 in enumerate(s1):
233
+ current_row = [i + 1]
234
+ for j, c2 in enumerate(s2):
235
+ insertions = previous_row[j + 1] + 1
236
+ deletions = current_row[j] + 1
237
+ substitutions = previous_row[j] + (c1 != c2)
238
+ current_row.append(min(insertions, deletions, substitutions))
239
+ previous_row = current_row
240
+ return previous_row[-1]
241
+
242
+ def is_similar_by_edit_distance(text1, text2, max_distance=2):
243
+ """Check if two texts are similar within edit distance threshold"""
244
+ if not text1 or not text2:
245
+ return False
246
+ distance = levenshtein_distance(text1.lower(), text2.lower())
247
+ max_len = max(len(text1), len(text2))
248
+ if max_len <= 3:
249
+ threshold = 1
250
+ elif max_len <= 6:
251
+ threshold = 2
252
+ else:
253
+ threshold = min(max_distance, max_len // 3)
254
+ return distance <= threshold
255
+
256
+ def extract_sentence_text(ocr_results):
257
+ """Extract sentence-based text for LLM input"""
258
+ paragraph_text = ""
259
+ for item in ocr_results:
260
+ if len(item) == 3:
261
+ _, text, _ = item
262
+ elif len(item) == 2:
263
+ _, text = item
264
+ else:
265
+ print(f"Unexpected OCR result format: {item}")
266
+ continue
267
+ if text.strip():
268
+ paragraph_text += text + " "
269
+ return paragraph_text.strip()
270
+
271
+ def extract_word_bboxes_improved(ocr_results):
272
+ """Improved word extraction with better handling of punctuation and spacing"""
273
+ word_bbox_map = []
274
+ for item in ocr_results:
275
+ if len(item) == 3:
276
+ bbox, text, confidence = item
277
+ elif len(item) == 2:
278
+ bbox, text = item
279
+ confidence = 1.0
280
+ else:
281
+ print(f"Unexpected OCR result format: {item}")
282
+ continue
283
+
284
+ original_text = text.strip()
285
+ if not original_text:
286
+ continue
287
+
288
+ if isinstance(bbox[0], (list, tuple)):
289
+ x_coords = [point[0] for point in bbox]
290
+ y_coords = [point[1] for point in bbox]
291
+ line_x1, line_y1 = min(x_coords), min(y_coords)
292
+ line_x2, line_y2 = max(x_coords), max(y_coords)
293
+ else:
294
+ line_x1, line_y1, line_x2, line_y2 = bbox
295
+
296
+ tokens = re.findall(r'\S+', original_text)
297
+ if len(tokens) <= 1:
298
+ padding = 1
299
+ word_bbox_map.append({
300
+ "word": original_text,
301
+ "bbox": [
302
+ max(0, int(line_x1 - padding)),
303
+ max(0, int(line_y1 - padding)),
304
+ int(line_x2 + padding),
305
+ int(line_y2 + padding)
306
+ ],
307
+ "confidence": confidence,
308
+ "original_line": original_text
309
+ })
310
+ continue
311
+
312
+ full_width = line_x2 - line_x1
313
+ text_without_spaces = original_text.replace(' ', '')
314
+ total_chars = len(text_without_spaces)
315
+
316
+ char_position = 0
317
+ for i, token in enumerate(tokens):
318
+ token_start_ratio = char_position / total_chars if total_chars > 0 else 0
319
+ char_position += len(token)
320
+ token_end_ratio = char_position / total_chars if total_chars > 0 else 1
321
+
322
+ token_x1 = line_x1 + (full_width * token_start_ratio)
323
+ token_x2 = line_x1 + (full_width * token_end_ratio)
324
+
325
+ padding = 1
326
+ word_bbox = [
327
+ max(0, int(token_x1 - padding)),
328
+ max(0, int(line_y1 - padding)),
329
+ int(min(token_x2 + padding, line_x2)),
330
+ int(line_y2 + padding)
331
+ ]
332
+
333
+ word_bbox_map.append({
334
+ "word": token,
335
+ "bbox": word_bbox,
336
+ "confidence": confidence,
337
+ "original_line": original_text
338
+ })
339
+ return sorted(word_bbox_map, key=lambda x: (x['bbox'][1], x['bbox'][0]))
340
+
341
+ def advanced_match_pii_to_words(pii_list, word_bbox_map):
342
+ """Advanced multi-strategy PII matching with comprehensive fallbacks"""
343
+ ocr_results_for_template = []
344
+ words = [info['word'] for info in word_bbox_map]
345
+ bboxes = [info['bbox'] for info in word_bbox_map]
346
+ is_pii_flags = [False] * len(words)
347
+ # Pre-process all words with different normalization strategies
348
+ normalized_words = [normalize_text(word) for word in words]
349
+ print(f"Processing {len(pii_list)} PII items against {len(words)} OCR words")
350
+ for pii_idx, pii_item in enumerate(pii_list):
351
+ if not pii_item.strip():
352
+ continue
353
+
354
+ print(f"Processing PII item {pii_idx + 1}: '{pii_item}'")
355
+
356
+ # Normalize the PII item
357
+ normalized_pii = normalize_text(pii_item)
358
+ pii_words = normalized_pii.split()
359
+
360
+ if not pii_words:
361
+ continue
362
+
363
+ matched = False
364
+
365
+ # Strategy 1: Exact matching after normalization
366
+ if len(pii_words) == 1:
367
+ pii_word = pii_words[0]
368
+ for idx, norm_word in enumerate(normalized_words):
369
+ if norm_word == pii_word and not is_pii_flags[idx]:
370
+ is_pii_flags[idx] = True
371
+ matched = True
372
+ print(f" βœ“ Exact match: '{words[idx]}' -> '{pii_item}'")
373
+ else:
374
+ # Multi-word exact matching
375
+ pii_len = len(pii_words)
376
+ start_idx = 0
377
+ while start_idx < len(normalized_words) - pii_len + 1:
378
+ exact_match = True
379
+ for j in range(pii_len):
380
+ if normalized_words[start_idx + j] != pii_words[j]:
381
+ exact_match = False
382
+ break
383
+
384
+ if exact_match:
385
+ # Check spatial proximity
386
+ spatial_ok = True
387
+ for j in range(1, pii_len):
388
+ prev_bbox = bboxes[start_idx + j - 1]
389
+ curr_bbox = bboxes[start_idx + j]
390
+
391
+ horizontal_distance = curr_bbox[0] - prev_bbox[2]
392
+ vertical_alignment = (abs(prev_bbox[1] - curr_bbox[1]) < 30 and
393
+ abs(prev_bbox[3] - curr_bbox[3]) < 30)
394
+
395
+ if not (vertical_alignment and horizontal_distance <= 150):
396
+ spatial_ok = False
397
+ break
398
+
399
+ if spatial_ok:
400
+ for j in range(pii_len):
401
+ if not is_pii_flags[start_idx + j]:
402
+ is_pii_flags[start_idx + j] = True
403
+ matched = True
404
+ matched_text = ' '.join(words[start_idx:start_idx + pii_len])
405
+ print(f" βœ“ Multi-word exact: '{matched_text}' -> '{pii_item}'")
406
+ start_idx += pii_len
407
+ continue
408
+ start_idx += 1
409
+
410
+ # Strategy 2: Fuzzy matching if exact matching failed
411
+ if not matched:
412
+ if len(pii_words) == 1:
413
+ pii_word = pii_words[0]
414
+ for idx, norm_word in enumerate(normalized_words):
415
+ if (not is_pii_flags[idx] and
416
+ (fuzzy_match_score(norm_word, pii_word, 0.9) or
417
+ is_similar_by_edit_distance(norm_word, pii_word, 2))):
418
+ is_pii_flags[idx] = True
419
+ matched = True
420
+ print(f" βœ“ Fuzzy match: '{words[idx]}' -> '{pii_item}'")
421
+ else:
422
+ # Multi-word fuzzy matching
423
+ pii_len = len(pii_words)
424
+ start_idx = 0
425
+ while start_idx < len(normalized_words) - pii_len + 1:
426
+ fuzzy_match = True
427
+ for j in range(pii_len):
428
+ if not (fuzzy_match_score(normalized_words[start_idx + j], pii_words[j], 0.85) or
429
+ is_similar_by_edit_distance(normalized_words[start_idx + j], pii_words[j], 2)):
430
+ fuzzy_match = False
431
+ break
432
+
433
+ if fuzzy_match:
434
+ # Check spatial proximity
435
+ spatial_ok = True
436
+ for j in range(1, pii_len):
437
+ prev_bbox = bboxes[start_idx + j - 1]
438
+ curr_bbox = bboxes[start_idx + j]
439
+
440
+ horizontal_distance = curr_bbox[0] - prev_bbox[2]
441
+ vertical_alignment = (abs(prev_bbox[1] - curr_bbox[1]) < 30 and
442
+ abs(prev_bbox[3] - curr_bbox[3]) < 30)
443
+
444
+ if not (vertical_alignment and horizontal_distance <= 150):
445
+ spatial_ok = False
446
+ break
447
+
448
+ if spatial_ok:
449
+ for j in range(pii_len):
450
+ if not is_pii_flags[start_idx + j]:
451
+ is_pii_flags[start_idx + j] = True
452
+ matched = True
453
+ matched_text = ' '.join(words[start_idx:start_idx + pii_len])
454
+ print(f" βœ“ Multi-word fuzzy: '{matched_text}' -> '{pii_item}'")
455
+ start_idx += pii_len
456
+ continue
457
+ start_idx += 1
458
+
459
+ # Strategy 3: Substring and partial matching
460
+ if not matched:
461
+ full_normalized_text = ' '.join(normalized_words)
462
+
463
+ pos = 0
464
+ while True:
465
+ start_pos = full_normalized_text.find(normalized_pii, pos)
466
+ if start_pos == -1:
467
+ break
468
+ end_pos = start_pos + len(normalized_pii)
469
+
470
+ char_count = 0
471
+ start_word_idx = None
472
+ end_word_idx = None
473
+
474
+ for idx, norm_word in enumerate(normalized_words):
475
+ word_start = char_count
476
+ word_end = char_count + len(norm_word)
477
+
478
+ if start_word_idx is None and word_end > start_pos:
479
+ start_word_idx = idx
480
+
481
+ if word_start < end_pos:
482
+ end_word_idx = idx
483
+
484
+ char_count += len(norm_word) + 1
485
+
486
+ if start_word_idx is not None and end_word_idx is not None and end_word_idx - start_word_idx + 1 >= len(pii_words):
487
+ spatial_ok = True
488
+ for j in range(start_word_idx, end_word_idx):
489
+ if j + 1 <= end_word_idx:
490
+ prev_bbox = bboxes[j]
491
+ next_bbox = bboxes[j + 1]
492
+
493
+ horizontal_distance = next_bbox[0] - prev_bbox[2]
494
+ vertical_alignment = (abs(prev_bbox[1] - next_bbox[1]) < 30 and
495
+ abs(prev_bbox[3] - next_bbox[3]) < 30)
496
+
497
+ if not (vertical_alignment and horizontal_distance <= 200):
498
+ spatial_ok = False
499
+ break
500
+
501
+ if spatial_ok:
502
+ for j in range(start_word_idx, end_word_idx + 1):
503
+ if not is_pii_flags[j]:
504
+ is_pii_flags[j] = True
505
+ matched = True
506
+ matched_text = ' '.join(words[start_word_idx:end_word_idx + 1])
507
+ print(f" βœ“ Substring match: '{matched_text}' -> '{pii_item}'")
508
+ pos = end_pos
509
+
510
+ # Strategy 4: Individual word matching with relaxed criteria
511
+ if not matched:
512
+ for pii_word in pii_words:
513
+ if len(pii_word) < 3:
514
+ continue
515
+
516
+ for idx, norm_word in enumerate(normalized_words):
517
+ if not is_pii_flags[idx]:
518
+ if (norm_word == pii_word or
519
+ fuzzy_match_score(norm_word, pii_word, 0.8) or
520
+ is_similar_by_edit_distance(norm_word, pii_word, 2) or
521
+ (len(pii_word) > 5 and (pii_word in norm_word or norm_word in pii_word))):
522
+ is_pii_flags[idx] = True
523
+ print(f" βœ“ Individual word match: '{words[idx]}' -> '{pii_word}' from '{pii_item}'")
524
+
525
+ if not matched:
526
+ print(f" βœ— No match found for: '{pii_item}'")
527
+ for idx, word_info in enumerate(word_bbox_map):
528
+ ocr_results_for_template.append({
529
+ "text": word_info["word"],
530
+ "bbox": word_info["bbox"],
531
+ "is_pii": is_pii_flags[idx],
532
+ "confidence": word_info.get("confidence", 1.0)
533
+ })
534
+ ocr_results_for_template = merge_horizontal_pii_boxes_improved(ocr_results_for_template)
535
+ return ocr_results_for_template
536
+
537
+ def merge_horizontal_pii_boxes_improved(ocr_results, merge_distance=50):
538
+ """Improved merging with better spatial awareness and tighter boxes"""
539
+ if not ocr_results:
540
+ return ocr_results
541
+ merged_results = []
542
+ i = 0
543
+ while i < len(ocr_results):
544
+ current_word = ocr_results[i]
545
+
546
+ if not current_word["is_pii"]:
547
+ merged_results.append(current_word)
548
+ i += 1
549
+ continue
550
+
551
+ merge_group = [current_word]
552
+ j = i + 1
553
+
554
+ while j < len(ocr_results):
555
+ next_word = ocr_results[j]
556
+
557
+ if not next_word["is_pii"]:
558
+ break
559
+
560
+ current_bbox = merge_group[-1]["bbox"]
561
+ next_bbox = next_word["bbox"]
562
+
563
+ y_center_current = (current_bbox[1] + current_bbox[3]) / 2
564
+ y_center_next = (next_bbox[1] + next_bbox[3]) / 2
565
+ y_overlap = abs(y_center_current - y_center_next) < 20
566
+
567
+ horizontal_distance = next_bbox[0] - current_bbox[2]
568
+
569
+ if y_overlap and horizontal_distance <= merge_distance:
570
+ merge_group.append(next_word)
571
+ j += 1
572
+ else:
573
+ break
574
+
575
+ if len(merge_group) > 1:
576
+ min_x = min(word["bbox"][0] for word in merge_group)
577
+ min_y = min(word["bbox"][1] for word in merge_group)
578
+ max_x = max(word["bbox"][2] for word in merge_group)
579
+ max_y = max(word["bbox"][3] for word in merge_group)
580
+
581
+ merged_text = " ".join(word["text"] for word in merge_group)
582
+
583
+ merged_word = {
584
+ "text": merged_text,
585
+ "bbox": [min_x, min_y, max_x, max_y],
586
+ "is_pii": True,
587
+ "confidence": max(word.get("confidence", 1.0) for word in merge_group)
588
+ }
589
+ merged_results.append(merged_word)
590
+ print(f" βœ“ Merged PII box: '{merged_text}' at [{min_x},{min_y},{max_x},{max_y}]")
591
+ else:
592
+ merged_results.append(current_word)
593
+
594
+ i = j
595
+ return merged_results
596
+
597
+ def post_process_pii_detection(ocr_results_for_template, pii_list):
598
+ """Post-process to catch any missed PII using relaxed matching"""
599
+ words = [result["text"] for result in ocr_results_for_template]
600
+ for pii_item in pii_list:
601
+ normalized_pii = normalize_text(pii_item)
602
+ pii_words = normalized_pii.split()
603
+
604
+ if not pii_words:
605
+ continue
606
+
607
+ pii_detected = False
608
+ for result in ocr_results_for_template:
609
+ if result["is_pii"]:
610
+ result_normalized = normalize_text(result["text"])
611
+ if (normalized_pii in result_normalized or
612
+ result_normalized in normalized_pii or
613
+ fuzzy_match_score(result_normalized, normalized_pii, 0.7)):
614
+ pii_detected = True
615
+ break
616
+
617
+ if not pii_detected:
618
+ print(f" ⚠ PII not detected, trying fallback matching: '{pii_item}'")
619
+
620
+ for idx, result in enumerate(ocr_results_for_template):
621
+ if result["is_pii"]:
622
+ continue
623
+
624
+ word_normalized = normalize_text(result["text"])
625
+
626
+ for pii_word in pii_words:
627
+ if (len(pii_word) > 3 and
628
+ (pii_word in word_normalized or
629
+ word_normalized in pii_word or
630
+ fuzzy_match_score(word_normalized, pii_word, 0.6) or
631
+ is_similar_by_edit_distance(word_normalized, pii_word, 3))):
632
+
633
+ ocr_results_for_template[idx]["is_pii"] = True
634
+ print(f" βœ“ Fallback match: '{result['text']}' -> '{pii_word}' from '{pii_item}'")
635
+ break
636
+ return ocr_results_for_template
637
+
638
+ def detect_pii_from_combined_text(combined_text):
639
+ """Detect PII from combined multi-page text using LLM"""
640
+ if llama_model is None:
641
+ print("LLM model not available for PII detection")
642
+ return [], "LLM model not available"
643
+ instruction = (
644
+ "Extract all Personally Identifiable Information (PII) of the main subject from the given text. "
645
+ "Include data like Name, Date of Birth, Gender, Address, Phone Number, Email, Social Security Number (SSN), Member ID, Group Number, or any other PII data available. "
646
+ "Ignore any information about doctors, staff, providers, colleagues, organizations, companies, hospitals, educational institutes, or facilities. "
647
+ "Return the results strictly as a flat set of strings enclosed in { } without labels."
648
+ )
649
+ prompt_content = f"{instruction}\n{combined_text}"
650
+ pii_list = []
651
+ llama_raw_output = ""
652
+ try:
653
+ messages = [{"role": "user", "content": prompt_content}]
654
+ response = llama_model.create_chat_completion(
655
+ messages=messages,
656
+ max_tokens=512,
657
+ temperature=0.1,
658
+ )
659
+ llama_raw_output = response['choices'][0]['message']['content']
660
+ pii_list = parse_pii_output(llama_raw_output)
661
+ print(f"LLM detected {len(pii_list)} PII items from combined text: {pii_list}")
662
+ except Exception as e:
663
+ print(f"Error during Llama PII detection: {e}")
664
+ llama_raw_output = f"Error: {str(e)}"
665
+ pii_list = []
666
+ return pii_list, llama_raw_output
667
+
668
+ # --- Combined Processing Function ---
669
+ def process_page_combined(img_cv, global_pii_list):
670
+ """Process a single page with both YOLO and OCR+LLM detection"""
671
+ all_detections = []
672
+ # Step 1: YOLO Visual Detection
673
+ print(" Running YOLO visual detection...")
674
+ visual_boxes, visual_classes = detect_visual_pii(img_cv)
675
+ for box, cls in zip(visual_boxes, visual_classes):
676
+ all_detections.append({
677
+ "text": cls,
678
+ "bbox": box.tolist() if hasattr(box, 'tolist') else box,
679
+ "is_pii": True,
680
+ "confidence": 1.0,
681
+ "detection_type": "visual"
682
+ })
683
+ print(f" YOLO detected {len(visual_boxes)} visual elements")
684
+ # Step 2: OCR + LLM Text Detection
685
+ if reader is not None:
686
+ print(" Running OCR text extraction...")
687
+ word_ocr_results = reader.readtext(img_cv, paragraph=False, width_ths=0.7, height_ths=0.7)
688
+
689
+ if word_ocr_results:
690
+ word_bbox_map = extract_word_bboxes_improved(word_ocr_results)
691
+
692
+ ocr_results_for_template = advanced_match_pii_to_words(global_pii_list, word_bbox_map)
693
+ ocr_results_for_template = post_process_pii_detection(ocr_results_for_template, global_pii_list)
694
+ ocr_results_for_template = refine_pii_flags(ocr_results_for_template)
695
+ ocr_results_for_template = merge_horizontal_pii_boxes_improved(ocr_results_for_template)
696
+
697
+ for result in ocr_results_for_template:
698
+ if result["is_pii"]:
699
+ result["detection_type"] = "text"
700
+ all_detections.append(result)
701
+
702
+ print(f" OCR detected {sum(1 for r in ocr_results_for_template if r['is_pii'])} text PII elements")
703
+ return all_detections
704
+
705
+ def classify_pii(text):
706
+ text = text.strip()
707
+ clean_text = re.sub(r'\s+', '', text)
708
+ if re.match(r'^\d{3}-\d{2}-\d{4}$', text) or re.match(r'^\d{3}-\d{2}-\d{4}$', clean_text):
709
+ return 'ssn'
710
+ elif re.match(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b', text) or re.match(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b', clean_text) or '@' in text:
711
+ return 'email'
712
+ elif re.match(r'^\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}$', text) or re.match(r'^\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}$', clean_text):
713
+ return 'phone'
714
+ elif re.match(r'^\d{1,2}/\d{1,2}/\d{4}$', text) or re.match(r'^\d{4}-\d{2}-\d{2}$', text) or re.match(r'^\d{1,2}/\d{1,2}/\d{4}$', clean_text) or re.match(r'^\d{1,2}-\d{1,2}-\d{4}$', text) or re.match(r'^\d{1,2}-\d{1,2}-\d{2}$', text) or re.match(r'^\d{2}/\d{2}/\d{4}$', text) or re.match(r'^\d{2}-\d{2}-\d{4}$', text) or re.match(r'^(January|February|March|April|May|June|July|August|September|October|November|December)\s+\d{1,2},\s+\d{4}$', text, re.IGNORECASE) or re.match(r'^(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{1,2},\s+\d{4}$', text, re.IGNORECASE) or re.match(r'^\d{1,2}\s+(January|February|March|April|May|June|July|August|September|October|November|December)\s+\d{4}$', text, re.IGNORECASE) or re.match(r'^\d{1,2}\s+(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s+\d{4}$', text, re.IGNORECASE):
715
+ return 'dob'
716
+ elif re.match(r'^(?=.*\d)[A-Za-z0-9]+$', text) and len(text) > 5:
717
+ return 'id'
718
+ elif ',' in text or 'St' in text or 'Ave' in text or re.search(r'\d{5}', text):
719
+ return 'address'
720
+ elif re.match(r'^(male|female|m|f|transgender|nonbinary|non-binary|other|unknown|u|o)$', text.lower()):
721
+ return 'gender'
722
+ else:
723
+ return 'name'
724
+
725
+ def format_gender(base_gender, original_text):
726
+ orig = original_text.strip()
727
+ orig_lower = orig.lower()
728
+ if orig_lower not in ['male', 'female', 'm', 'f', 'transgender', 'nonbinary', 'non-binary', 'other', 'unknown', 'u', 'o']:
729
+ return base_gender.capitalize()
730
+ if len(orig) == 1:
731
+ char = 'M' if base_gender == 'male' else 'F'
732
+ return char.lower() if orig.islower() else char
733
+ else:
734
+ if orig.isupper():
735
+ return base_gender.upper()
736
+ elif orig.islower():
737
+ return base_gender.lower()
738
+ else:
739
+ return base_gender.capitalize()
740
+
741
+ def generate_fake(pii_type, length, original=None):
742
+ max_attempts = 100
743
+ if pii_type == 'dob' and original:
744
+ formats = [
745
+ '%m/%d/%Y', '%m/%d/%y', '%d/%m/%Y', '%d/%m/%y', '%Y-%m-%d', '%y-%m-%d', '%m-%d-%Y', '%m-%d-%y',
746
+ '%d-%m-%Y', '%d-%m-%y', '%Y/%m/%d', '%y/%m/%d', '%d.%m.%Y', '%m.%d.%Y', '%B %d, %Y', '%b %d, %Y',
747
+ '%d %B %Y', '%d %b %Y', '%B %d %Y', '%b %d %Y'
748
+ ]
749
+ for fmt in formats:
750
+ try:
751
+ datetime.datetime.strptime(original.strip(), fmt)
752
+ fake_dt = fake.date_object()
753
+ return fake_dt.strftime(fmt)
754
+ except ValueError:
755
+ pass
756
+ return fake.date(pattern='%m/%d/%Y')
757
+ elif pii_type == 'gender' and original:
758
+ return format_gender(random.choice(['male', 'female']), original)
759
+ elif pii_type in ['name', 'email', 'phone', 'address']:
760
+ for _ in range(max_attempts):
761
+ if pii_type == 'name': f = fake.name()
762
+ elif pii_type == 'email': f = fake.email()
763
+ elif pii_type == 'phone': f = fake.phone_number()
764
+ elif pii_type == 'address': f = fake.address().replace('\n', ', ')
765
+ if len(f) == length:
766
+ return f
767
+
768
+ closest = None
769
+ min_diff = float('inf')
770
+ for _ in range(50):
771
+ if pii_type == 'name': f = fake.name()
772
+ elif pii_type == 'email': f = fake.email()
773
+ elif pii_type == 'phone': f = fake.phone_number()
774
+ elif pii_type == 'address': f = fake.address().replace('\n', ', ')
775
+ diff = abs(len(f) - length)
776
+ if diff < min_diff:
777
+ min_diff, closest = diff, f
778
+ return closest
779
+ elif pii_type == 'ssn':
780
+ return fake.ssn()
781
+ else: # id and others
782
+ return fake.lexify(text='?' * length)
783
+
784
+ def detect_format(text):
785
+ text = text.strip().strip('"')
786
+ if text.startswith('ISA*'):
787
+ return 'edi_x12'
788
+ else:
789
+ return 'plain'
790
+
791
+ def convert_to_readable(text, format_type):
792
+ if format_type == 'edi_x12':
793
+ parsed_transactions = parse_edi_fallback(text)
794
+ output = StringIO()
795
+ format_output(parsed_transactions, file=output)
796
+ return output.getvalue()
797
+ else:
798
+ return text
799
+
800
+ def redact_text(original_text, pii_list):
801
+ redacted = original_text
802
+ for pii in pii_list:
803
+ redacted = re.sub(re.escape(pii), '[REDACTED]', redacted, flags=re.IGNORECASE)
804
+ return redacted
805
+
806
+ def anonymize_text(original_text, pii_list):
807
+ anonymized = original_text
808
+ pii_map = {}
809
+ for pii in pii_list:
810
+ normalized = normalize_text(pii)
811
+ pii_type = classify_pii(pii)
812
+ if normalized not in pii_map:
813
+ if pii_type == 'gender':
814
+ base_gender = random.choice(['male', 'female'])
815
+ fake_val = format_gender(base_gender, pii)
816
+ pii_map[normalized] = fake_val
817
+ else:
818
+ fake_val = generate_fake(pii_type, len(pii), pii)
819
+ pii_map[normalized] = fake_val
820
+ else:
821
+ fake_val = pii_map[normalized]
822
+ anonymized = re.sub(re.escape(pii), fake_val, anonymized, flags=re.IGNORECASE)
823
+ return anonymized
824
+
825
+ def parse_edi_fallback(edi_content):
826
+ """
827
+ Fallback parser that combines address parts into a single line for easier redaction.
828
+ """
829
+ print("Using fallback parser...")
830
+
831
+ segments = edi_content.replace('~', '\n').split('\n')
832
+ segments = [seg.strip() for seg in segments if seg.strip()]
833
+
834
+ parsed_data = {
835
+ 'transaction_info': {}, 'patient_info': {}, 'provider_info': {},
836
+ 'service_info': {}, 'diagnosis_info': {}
837
+ }
838
+
839
+ for segment in segments:
840
+ elements = segment.split('*')
841
+ segment_id = elements[0]
842
+
843
+ if segment_id == 'ST':
844
+ parsed_data['transaction_info']['transaction_type'] = elements[1]
845
+ elif segment_id == 'NM1':
846
+ entity_type = elements[1]
847
+ if entity_type == 'IL': # Patient
848
+ last_name = elements[3] if len(elements) > 3 else ''
849
+ first_name = elements[4] if len(elements) > 4 else ''
850
+ parsed_data['patient_info']['name'] = f"{first_name} {last_name}".strip()
851
+ if len(elements) > 8:
852
+ parsed_data['patient_info']['id'] = elements[9]
853
+ elif entity_type == 'SJ': # Provider
854
+ last_name = elements[3] if len(elements) > 3 else ''
855
+ first_name = elements[4] if len(elements) > 4 else ''
856
+ parsed_data['provider_info']['name'] = f"{first_name} {last_name}".strip()
857
+ if len(elements) > 8:
858
+ parsed_data['provider_info']['npi'] = elements[9]
859
+ elif segment_id == 'N3':
860
+ # Store the first line of the address
861
+ parsed_data['patient_info']['address'] = elements[1]
862
+ elif segment_id == 'N4':
863
+ # Combine City, State, and Zip with the address line
864
+ city = elements[1] if len(elements) > 1 else ''
865
+ state = elements[2] if len(elements) > 2 else ''
866
+ zip_code = elements[3] if len(elements) > 3 else ''
867
+
868
+ full_address_parts = [city, state, zip_code]
869
+
870
+ # If an address line already exists from N3, prepend it
871
+ if 'address' in parsed_data['patient_info']:
872
+ full_address_parts.insert(0, parsed_data['patient_info']['address'])
873
+
874
+ # Join all parts with ", " and filter out any empty parts
875
+ parsed_data['patient_info']['address'] = ", ".join(filter(None, full_address_parts))
876
+
877
+ elif segment_id == 'DMG':
878
+ parsed_data['patient_info']['dob'] = elements[2]
879
+ parsed_data['patient_info']['gender'] = 'Female' if elements[3] == 'F' else 'Male'
880
+ elif segment_id == 'UM':
881
+ parsed_data['service_info']['service_type'] = elements[1]
882
+ parsed_data['service_info']['request_category'] = elements[2]
883
+ parsed_data['service_info']['service_code'] = elements[3]
884
+ if len(elements) > 4:
885
+ parsed_data['service_info']['quantity'] = elements[4]
886
+ elif segment_id == 'HI':
887
+ diagnosis_info = elements[1].split(':')
888
+ if len(diagnosis_info) > 1:
889
+ parsed_data['diagnosis_info']['code_qualifier'] = diagnosis_info[0]
890
+ parsed_data['diagnosis_info']['diagnosis_code'] = diagnosis_info[1]
891
+
892
+ return [{
893
+ 'transaction_type': parsed_data['transaction_info'].get('transaction_type', 'Unknown'),
894
+ 'parsed_data': {
895
+ 'description': 'Health Care Services Review',
896
+ 'patient_info': parsed_data['patient_info'],
897
+ 'provider_info': parsed_data['provider_info'],
898
+ 'service_info': parsed_data['service_info'],
899
+ 'diagnosis_info': parsed_data['diagnosis_info']
900
+ }
901
+ }]
902
+
903
+ def format_output(parsed_transactions, file=None):
904
+ """Format parsed data for display"""
905
+ output_lines = []
906
+ for i, transaction in enumerate(parsed_transactions):
907
+ output_lines.append(f"\n=== TRANSACTION {i+1} ===")
908
+ output_lines.append(f"Transaction Type: {transaction['transaction_type']}")
909
+
910
+ if 'parsed_data' in transaction:
911
+ data = transaction['parsed_data']
912
+
913
+ if 'description' in data:
914
+ output_lines.append(f"Description: {data['description']}")
915
+
916
+ # Patient Information
917
+ if 'patient_info' in data and data['patient_info']:
918
+ output_lines.append("\nPATIENT INFORMATION:")
919
+ for key, value in data['patient_info'].items():
920
+ output_lines.append(f" {key.replace('_', ' ').title()}: {value}")
921
+
922
+ # Provider Information
923
+ if 'provider_info' in data and data['provider_info']:
924
+ output_lines.append("\nPROVIDER INFORMATION:")
925
+ for key, value in data['provider_info'].items():
926
+ output_lines.append(f" {key.replace('_', ' ').title()}: {value}")
927
+
928
+ # Service Information
929
+ if 'service_info' in data and data['service_info']:
930
+ output_lines.append("\nSERVICE INFORMATION:")
931
+ for key, value in data['service_info'].items():
932
+ output_lines.append(f" {key.replace('_', ' ').title()}: {value}")
933
+
934
+ # Diagnosis Information
935
+ if 'diagnosis_info' in data and data['diagnosis_info']:
936
+ output_lines.append("\nDIAGNOSIS INFORMATION:")
937
+ for key, value in data['diagnosis_info'].items():
938
+ output_lines.append(f" {key.replace('_', ' ').title()}: {value}")
939
+ output_str = '\n'.join(output_lines)
940
+ if file:
941
+ file.write(output_str)
942
+ else:
943
+ print(output_str)
944
+ return output_str
945
+
946
+ # --- Main Gradio Processing Function ---
947
+ def analyze_document(file, progress=gr.Progress()):
948
+ """
949
+ This function takes an uploaded file, processes it through the PII detection pipeline,
950
+ and returns the annotated images, a redacted PDF, and a summary report.
951
+ """
952
+ load_models()
953
+ if file is None:
954
+ return None, None, None, None, None, "Please upload a document to begin."
955
+
956
+ unique_id = uuid.uuid4().hex
957
+ if hasattr(file, 'name'):
958
+ filepath = file.name
959
+ else:
960
+ filepath = str(file)
961
+ filename = os.path.basename(filepath)
962
+ extension = os.path.splitext(filename)[1]
963
+
964
+ if extension.lower() == '.csv':
965
+ rows = []
966
+ with open(filepath, 'r', newline='') as csvfile:
967
+ csv_reader = csv.reader(csvfile)
968
+ for row in csv_reader:
969
+ if row:
970
+ rows.append(row[0])
971
+ total_rows = len(rows)
972
+ print(f"Processing {total_rows} rows for job {unique_id}...")
973
+ progress(0.1, desc="Reading CSV rows...")
974
+ report = f"## πŸ” Analysis Report for CSV\n**Total Rows:** {total_rows}\n\n---\n"
975
+ redacted_rows = []
976
+ anonymized_rows = []
977
+ for i, text in enumerate(rows):
978
+ progress(0.4 + (i / total_rows * 0.5), desc=f"Processing Row {i+1}/{total_rows}...")
979
+ format_type = detect_format(text)
980
+ readable_text = convert_to_readable(text, format_type)
981
+ pii_list, llama_raw_output = detect_pii_from_combined_text(readable_text)
982
+ redacted_text = redact_text(readable_text, pii_list)
983
+ anonymized_text = anonymize_text(readable_text, pii_list)
984
+ redacted_rows.append(redacted_text)
985
+ anonymized_rows.append(anonymized_text)
986
+ report += f"### πŸ“„ Row {i+1}\n- **Text Detections:** {len(pii_list)}\n- **PII Found:** {', '.join(pii_list) if pii_list else 'None'}\n\n"
987
+ progress(0.9, desc="Generating final CSVs...")
988
+ redacted_csv_path = os.path.join(RESULTS_FOLDER, f"redacted_{unique_id}.csv")
989
+ with open(redacted_csv_path, 'w', newline='') as csvfile:
990
+ writer = csv.writer(csvfile)
991
+ for txt in redacted_rows:
992
+ writer.writerow([txt])
993
+ anonymized_csv_path = os.path.join(RESULTS_FOLDER, f"anonymized_{unique_id}.csv")
994
+ with open(anonymized_csv_path, 'w', newline='') as csvfile:
995
+ writer = csv.writer(csvfile)
996
+ for txt in anonymized_rows:
997
+ writer.writerow([txt])
998
+ progress(1, desc="Complete!")
999
+ print("Processing Complete.")
1000
+ return [], [], [], redacted_csv_path, anonymized_csv_path, report
1001
+
1002
+ progress(0, desc="Converting document to images...")
1003
+ images_to_process = []
1004
+ try:
1005
+ if extension.lower() == '.pdf':
1006
+ try:
1007
+ images_to_process = [cv2.cvtColor(np.array(page), cv2.COLOR_RGB2BGR) for page in convert_from_path(filepath, dpi=300)]
1008
+ except Exception as e:
1009
+ print(f"PDF conversion error: {e}. Trying fallback method...")
1010
+ try:
1011
+ images_to_process = [cv2.cvtColor(np.array(page), cv2.COLOR_RGB2BGR) for page in convert_from_path(filepath, dpi=150)]
1012
+ except Exception as e2:
1013
+ return None, None, None, None, None, f"πŸ”΄ **Error:** Could not process PDF. Please ensure Poppler is installed.\nDetails: {e2}"
1014
+ else:
1015
+ img = cv2.imread(filepath)
1016
+ if img is not None:
1017
+ images_to_process.append(img)
1018
+ except Exception as e:
1019
+ return None, None, None, None, None, f"πŸ”΄ **Error:** Could not process file. Details: {e}"
1020
+
1021
+ if not images_to_process:
1022
+ return None, None, None, None, None, "πŸ”΄ **Error:** No pages could be extracted from the document."
1023
+
1024
+ total_pages = len(images_to_process)
1025
+ print(f"Processing {total_pages} pages for job {unique_id}...")
1026
+
1027
+ progress(0.1, desc="Extracting text from all pages (OCR)...")
1028
+ combined_text, all_pages_data = "", []
1029
+ for i, img_cv in enumerate(images_to_process):
1030
+ page_text = ""
1031
+ if reader:
1032
+ page_text = extract_sentence_text(reader.readtext(img_cv, paragraph=True))
1033
+ combined_text += f"\n--- Page {i+1} ---\n{page_text}\n"
1034
+ all_pages_data.append({"img_cv": img_cv, "page_num": i + 1})
1035
+
1036
+ progress(0.4, desc="Analyzing text for PII with LLM...")
1037
+ global_pii_list, llama_raw_output = detect_pii_from_combined_text(combined_text)
1038
+
1039
+ annotated_paths, redacted_paths, anonymized_paths = [], [], []
1040
+ redacted_pils, anonymized_pils = [], []
1041
+ report = f"## πŸ” Analysis Report\n**Global PII Found:** `{', '.join(global_pii_list) if global_pii_list else 'None'}`\n\n---\n"
1042
+ pii_map = {}
1043
+
1044
+ for i, page_info in enumerate(all_pages_data):
1045
+ progress(0.5 + (i / total_pages * 0.4), desc=f"Processing Page {i+1}/{total_pages} (Visual & Text)...")
1046
+ img_cv, page_num = page_info["img_cv"], page_info["page_num"]
1047
+ detections = process_page_combined(img_cv, global_pii_list)
1048
+
1049
+ annotated_img = img_cv.copy()
1050
+ redacted_img = img_cv.copy()
1051
+ anonymized_img = img_cv.copy()
1052
+ visual_count = sum(1 for d in detections if d["detection_type"] == "visual")
1053
+ text_count = sum(1 for d in detections if d.get("detection_type") == "text")
1054
+
1055
+ for d in detections:
1056
+ bbox = d.get("bbox", [])
1057
+ if not bbox: continue
1058
+ x1, y1, x2, y2 = map(int, bbox)
1059
+ color = (0, 255, 0) if d.get("detection_type") == "visual" else (0, 0, 255)
1060
+ cv2.rectangle(annotated_img, (x1, y1), (x2, y2), color, 3)
1061
+
1062
+ if d["detection_type"] == "visual":
1063
+ cv2.rectangle(redacted_img, (x1, y1), (x2, y2), (0, 0, 0), -1)
1064
+ cv2.rectangle(anonymized_img, (x1, y1), (x2, y2), (0, 0, 0), -1)
1065
+ else:
1066
+ bg_color = (255, 255, 255)
1067
+ height, width = img_cv.shape[:2]
1068
+ if x2 + 20 < width:
1069
+ sample = img_cv[y1:y2, x2:x2+20]
1070
+ if sample.size > 0:
1071
+ bg_color = tuple(map(int, np.mean(sample, axis=(0,1))))
1072
+ else:
1073
+ if x1 > 20:
1074
+ sample = img_cv[y1:y2, x1-20:x1]
1075
+ if sample.size > 0:
1076
+ bg_color = tuple(map(int, np.mean(sample, axis=(0,1))))
1077
+
1078
+ cv2.rectangle(redacted_img, (x1, y1), (x2, y2), (0, 0, 0), -1)
1079
+ cv2.rectangle(anonymized_img, (x1, y1), (x2, y2), bg_color, -1)
1080
+
1081
+ original_text = d["text"]
1082
+ pii_type = classify_pii(original_text)
1083
+ normalized = normalize_text(original_text)
1084
+ key = 'gender' if pii_type == 'gender' else normalized
1085
+
1086
+ if key not in pii_map:
1087
+ if pii_type == 'gender':
1088
+ base_gender = random.choice(['male', 'female'])
1089
+ fake_text = format_gender(base_gender, original_text)
1090
+ pii_map[key] = base_gender
1091
+ else:
1092
+ fake_text = generate_fake(pii_type, len(original_text), original_text)
1093
+ pii_map[key] = fake_text
1094
+ else:
1095
+ if pii_type == 'gender':
1096
+ base_gender = pii_map[key]
1097
+ fake_text = format_gender(base_gender, original_text)
1098
+ else:
1099
+ fake_text = pii_map[key]
1100
+
1101
+ font = cv2.FONT_HERSHEY_SIMPLEX
1102
+ font_scale = (y2 - y1) / 40.0
1103
+ thickness = 2
1104
+ text_size, _ = cv2.getTextSize(fake_text, font, font_scale, thickness)
1105
+ box_width = x2 - x1
1106
+ if text_size[0] > box_width - 10:
1107
+ font_scale *= (box_width - 10) / text_size[0]
1108
+ text_size, _ = cv2.getTextSize(fake_text, font, font_scale, thickness)
1109
+ text_x = x1 + (box_width - text_size[0]) // 2
1110
+ text_y = y1 + ((y2 - y1) + text_size[1]) // 2
1111
+ bg_brightness = 0.299 * bg_color[2] + 0.587 * bg_color[1] + 0.114 * bg_color[0]
1112
+ text_color = (0, 0, 0) if bg_brightness > 128 else (255, 255, 255)
1113
+ cv2.putText(anonymized_img, fake_text, (text_x, text_y), font, font_scale, text_color, thickness)
1114
+
1115
+ # Save all three versions of the image for the galleries
1116
+ annotated_path = os.path.join(RESULTS_FOLDER, f"annotated_{unique_id}_{page_num}.jpg")
1117
+ redacted_path = os.path.join(RESULTS_FOLDER, f"redacted_preview_{unique_id}_{page_num}.jpg")
1118
+ anonymized_path = os.path.join(RESULTS_FOLDER, f"anonymized_preview_{unique_id}_{page_num}.jpg")
1119
+
1120
+ cv2.imwrite(annotated_path, annotated_img)
1121
+ cv2.imwrite(redacted_path, redacted_img)
1122
+ cv2.imwrite(anonymized_path, anonymized_img)
1123
+
1124
+ annotated_paths.append(annotated_path)
1125
+ redacted_paths.append(redacted_path)
1126
+ anonymized_paths.append(anonymized_path)
1127
+
1128
+ redacted_pils.append(Image.fromarray(cv2.cvtColor(redacted_img, cv2.COLOR_BGR2RGB)))
1129
+ anonymized_pils.append(Image.fromarray(cv2.cvtColor(anonymized_img, cv2.COLOR_BGR2RGB)))
1130
+
1131
+ report += f"### πŸ“„ Page {page_num}\n- **Visual Detections (🟩 Green):** {visual_count}\n- **Text Detections (πŸŸ₯ Red):** {text_count}\n"
1132
+
1133
+ progress(0.9, desc="Generating final PDFs...")
1134
+ redacted_pdf_path, anonymized_pdf_path = None, None
1135
+ if redacted_pils:
1136
+ pdf_path = os.path.join(RESULTS_FOLDER, f"redacted_{unique_id}.pdf")
1137
+ redacted_pils[0].save(pdf_path, "PDF", resolution=100.0, save_all=True, append_images=redacted_pils[1:])
1138
+ redacted_pdf_path = pdf_path
1139
+ if anonymized_pils:
1140
+ pdf_path = os.path.join(RESULTS_FOLDER, f"anonymized_{unique_id}.pdf")
1141
+ anonymized_pils[0].save(pdf_path, "PDF", resolution=100.0, save_all=True, append_images=anonymized_pils[1:])
1142
+ anonymized_pdf_path = pdf_path
1143
+
1144
+ progress(1, desc="Complete!")
1145
+ print("Processing Complete.")
1146
+
1147
+ return annotated_paths, redacted_paths, anonymized_paths, redacted_pdf_path, anonymized_pdf_path, report
1148
+
1149
+ # --- Gradio Interface Definition ---
1150
+ title = "πŸ”’ Combined PII Detection System"
1151
+ description = """
1152
+ ### Advanced Multi-Modal PII Detection
1153
+ This system uses a combination of visual and textual analysis to detect and redact Personally Identifiable Information from your documents.
1154
+ - **πŸ–ΌοΈ Visual Detection (YOLO):** Detects Faces, QR Codes, and Signatures.
1155
+ - **πŸ“ Text Detection (OCR + LLM):** Detects Names, Addresses, Phone Numbers, IDs, and other contextual PII.
1156
+ **How to Use:**
1157
+ 1. Upload a document (PDF or image format).
1158
+ 2. The system will process each page and display annotated previews with colored boxes.
1159
+ 3. A fully redacted PDF with blacked-out PII is generated for you to download.
1160
+ 4. An analysis report summarizes the findings for each page.
1161
+ """
1162
+ with gr.Blocks(theme=gr.themes.Soft()) as demo:
1163
+ gr.Markdown(f"<h1 style='margin-bottom: 0.25rem;'>{title}</h1>")
1164
+ gr.Markdown(
1165
+ "<p style='color:#475569; line-height:1.6;'>"
1166
+ "Upload a PDF or image to detect and redact PII using visual detectors (🟩) and text analysis (πŸŸ₯). "
1167
+ "Fixed for local environment with proper YOLO support."
1168
+ "</p>"
1169
+ )
1170
+ with gr.Accordion("About this tool", open=False):
1171
+ gr.Markdown(description)
1172
+ with gr.Tabs():
1173
+ with gr.Tab("Run"):
1174
+ with gr.Row():
1175
+ with gr.Column(scale=1):
1176
+ file_input = gr.File(
1177
+ label="Upload Document",
1178
+ file_types=['.pdf', '.jpg', '.jpeg', '.png', '.bmp', '.csv'],
1179
+ file_count="single",
1180
+ height=100
1181
+ )
1182
+ submit_btn = gr.Button("πŸš€ Analyze Document", variant="primary")
1183
+ with gr.Accordion("Tips", open=False):
1184
+ gr.Markdown(
1185
+ "- Prefer high-resolution files for better OCR results (300 DPI for PDFs).\n"
1186
+ "- For PDFs, ensure Poppler is installed on your system.\n"
1187
+ "- Visual detections are drawn in green; text-based detections are in red.\n"
1188
+ "- Use the Previews tab to inspect annotated pages and the Report tab to download the redacted PDF."
1189
+ )
1190
+ with gr.Column(scale=1):
1191
+ gr.Markdown("### What happens during analysis")
1192
+ gr.Markdown(
1193
+ "- Convert pages to images\n"
1194
+ "- Run global OCR to build combined text\n"
1195
+ "- Use LLM to extract possible PII strings\n"
1196
+ "- Match PII back to words and merge boxes\n"
1197
+ "- Render annotated previews and build a redacted PDF"
1198
+ )
1199
+ clear_btn = gr.Button("🧹 Clear Results", variant="secondary")
1200
+
1201
+ with gr.Tab("Annotated Preview (Detection)"):
1202
+ gr.Markdown("### Annotated Previews (🟩 Visual, πŸŸ₯ Text)")
1203
+ annotated_gallery_output = gr.Gallery(
1204
+ label="Annotated Pages", show_label=False, elem_id="gallery_annotated",
1205
+ columns=[2], rows=[1], object_fit="contain", height=480
1206
+ )
1207
+
1208
+ with gr.Tab("Redacted Preview"):
1209
+ gr.Markdown("### Redacted Previews (Blacked Out)")
1210
+ redacted_gallery_output = gr.Gallery(
1211
+ label="Redacted Pages", show_label=False, elem_id="gallery_redacted",
1212
+ columns=[2], rows=[1], object_fit="contain", height=480
1213
+ )
1214
+
1215
+ with gr.Tab("Anonymized Preview"):
1216
+ gr.Markdown("### Anonymized Previews (Fake Data)")
1217
+ anonymized_gallery_output = gr.Gallery(
1218
+ label="Anonymized Pages", show_label=False, elem_id="gallery_anonymized",
1219
+ columns=[2], rows=[1], object_fit="contain", height=480
1220
+ )
1221
+
1222
+ with gr.Tab("Report & Download"):
1223
+ with gr.Row():
1224
+ with gr.Column(scale=1):
1225
+ gr.Markdown("### Downloads")
1226
+ redacted_file_output = gr.File(label="Redacted PDF (Blacked Out)")
1227
+ anonymized_file_output = gr.File(label="Anonymized PDF (Fake Data)")
1228
+ with gr.Column(scale=2):
1229
+ gr.Markdown("### Analysis Report")
1230
+ report_output = gr.Markdown(label="Analysis Report")
1231
+
1232
+ outputs_list = [
1233
+ annotated_gallery_output,
1234
+ redacted_gallery_output,
1235
+ anonymized_gallery_output,
1236
+ redacted_file_output,
1237
+ anonymized_file_output,
1238
+ report_output
1239
+ ]
1240
+
1241
+ submit_btn.click(
1242
+ fn=analyze_document,
1243
+ inputs=file_input,
1244
+ outputs=outputs_list
1245
+ )
1246
+
1247
+ clear_btn.click(
1248
+ fn=lambda: ([], [], [], None, None, "Ready. Upload a document and click Analyze."),
1249
+ inputs=None,
1250
+ outputs=outputs_list
1251
+ )
1252
+
1253
+ if __name__ == "__main__":
1254
+ demo.queue().launch(server_port=8000)
requirements.txt ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ gradio==3.50.2
2
+ ultralytics==8.1.34
3
+ pdf2image==1.17.0
4
+ easyocr==1.7.2
5
+ llama-cpp-python==0.3.16
6
+ pydantic==1.10.13
7
+ opencv-python
8
+ torch
9
+ torchvision
10
+ Pillow
unsloth.F16.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:411f2d808152dd6e9f5d350242dce3317bb8b6055eedbc04f7f155430b63c356
3
+ size 2479591968