File size: 33,897 Bytes
41bd0b8
 
 
 
 
 
 
a5778f2
41bd0b8
a5778f2
 
41bd0b8
 
 
 
 
a5778f2
 
41bd0b8
a5778f2
41bd0b8
 
a5778f2
 
 
41bd0b8
 
 
 
 
a5778f2
 
41bd0b8
 
a5778f2
 
 
41bd0b8
 
 
 
 
a5778f2
41bd0b8
a5778f2
41bd0b8
 
a5778f2
41bd0b8
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
 
41bd0b8
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
 
41bd0b8
 
 
 
 
a5778f2
 
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
a5778f2
41bd0b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
"""
RAG (Retrieval-Augmented Generation) Helper for Too Many Cables Chatbot
Integrates knowledge base search with chatbot responses
Now supports both simple keyword search and advanced vector-based semantic search
"""

import json
import logging
from typing import List, Dict, Optional, Tuple
from .knowledge_base_manager import KnowledgeBaseManager

# Set up logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)

# Try to import vector RAG components (fallback to keyword search if not available)
try:
    from .vector_rag_manager import VectorRAGManager
    VECTOR_RAG_AVAILABLE = True
except ImportError:
    VECTOR_RAG_AVAILABLE = False
    print("Vector RAG dependencies not available. Using keyword search fallback.")

class RAGHelper:
    def __init__(self, knowledge_base_path: str = "knowledge_base", use_vector_search: bool = True):
        """Initialize RAG helper with knowledge base and optional vector search"""
        self.knowledge_base_path = knowledge_base_path
        self.use_vector_search = use_vector_search and VECTOR_RAG_AVAILABLE
        
        # Initialize knowledge base manager (always needed)
        self.kb_manager = KnowledgeBaseManager(knowledge_base_path)
        self.kb_manager.scan_documents()
        
        # Initialize vector RAG if available and requested
        self.vector_rag = None
        if self.use_vector_search:
            try:
                print("Initializing Vector RAG Manager...")
                # Use /app/data/vector_db for writable storage in Docker container
                vector_db_path = "/app/data/vector_db"
                self.vector_rag = VectorRAGManager(knowledge_base_path, vector_db_path=vector_db_path)
                print("Vector RAG Manager initialized successfully!")
            except Exception as e:
                print(f"Failed to initialize Vector RAG: {e}")
                self.use_vector_search = False
        
        # Configuration for RAG
        self.max_context_docs = 3
        self.max_context_length = 3000  # Increased from 2000 to better utilize model capacity while maintaining safety
        self.relevance_threshold = 1  # Minimum relevance score to include (for keyword search)
        self.similarity_threshold = 0.30  # Increased from 0.20 - be more selective about what's relevant
        
    def _route_query_to_categories(self, query: str) -> List[str]:
        """
        Route queries to specific KB categories for focused retrieval
        Enhanced with specific product type detection
        """
        query_lower = query.lower()
        prioritized_categories = []
        
        # Check for specific product types first (highest priority)
        product_type_keywords = {
            'lightning': ['lightning', 'iphone', 'ipad', 'ipod', 'apple', 'mfi'],
            'usb-c': ['usb-c', 'usbc', 'type-c', 'usb c'],
            'hdmi': ['hdmi', '4k', '8k', 'video', 'display', 'monitor', 'tv'],
            'audio': ['audio', '3.5mm', 'headphone', 'speaker', 'aux'],
        }
        
        # If query mentions specific products, prioritize product_manuals
        for product_type, keywords in product_type_keywords.items():
            if any(keyword in query_lower for keyword in keywords):
                prioritized_categories = ['product_manuals', 'faqs', 'policies']
                logger.info(f"🎯 Product-specific query detected ({product_type}) - prioritizing product_manuals")
                return prioritized_categories
        
        # Category routing rules (existing logic)
        category_keywords = {
            'policies': [
                'return', 'refund', 'warranty', 'guarantee', 'exchange', 
                'policy', 'shipping', 'delivery', 'payment',
                'money back', 'cancel', 'replacement'
            ],
            'faqs': [
                'how to', 'what is', 'why does', 'when should', 'where can',
                'help', 'problem', 'issue', 'troubleshoot', 'not working',
                'question', 'support', 'compatibility'
            ],
            'product_manuals': [
                'specifications', 'specs', 'technical', 'manual', 'guide',
                'installation', 'setup', 'configure', 'use', 'connect',
                'length', 'size', 'connector', 'pin', 'voltage', 'model', 'recommend'
            ],
            'development': [
                'api', 'endpoint', 'authentication', 'token', 'key', 'secret',
                'development', 'internal', 'admin', 'credentials', 'login',
                'password', 'database', 'security', 'vulnerability', 'documentation'
            ]
        }
        
        # Score each category
        category_scores = {}
        for category, keywords in category_keywords.items():
            score = sum(1 for keyword in keywords if keyword in query_lower)
            if score > 0:
                category_scores[category] = score
        
        # Sort by relevance score
        if category_scores:
            prioritized_categories = sorted(
                category_scores.keys(), 
                key=lambda x: category_scores[x], 
                reverse=True
            )
            logger.info(f"🎯 Query routed to categories: {prioritized_categories} (scores: {category_scores})")
        else:
            # No specific routing, search all categories
            prioritized_categories = ['policies', 'faqs', 'product_manuals', 'development']
            logger.info("πŸ”„ No category match - searching all categories")
        
        return prioritized_categories
    
    def get_relevant_context(self, query: str, max_docs: Optional[int] = None) -> str:
        """
        Get relevant context from knowledge base for a query
        Uses section routing + vector search if available, falls back to keyword search
        Returns formatted context string for LLM
        """
        max_docs = max_docs or self.max_context_docs
        
        logger.info(f"πŸ” RAG LOOKUP TRIGGERED - Query: '{query}' (max_docs: {max_docs})")
        
        # Step 1: Route query to relevant categories
        target_categories = self._route_query_to_categories(query)
        
        # Use vector search with section routing if available
        if self.use_vector_search and self.vector_rag:
            try:
                logger.info(f"πŸ“Š Using SECTION-ROUTED RETRIEVE-AND-RERANK (categories: {target_categories}, threshold: {self.similarity_threshold})")
                
                # Use category-filtered retrieve-and-rerank pipeline
                reranked_results = self.vector_rag.retrieve_and_rerank_filtered(
                    query, 
                    target_categories=target_categories,
                    initial_k=20,  # Retrieve top 20 candidates
                    final_k=max_docs,  # Re-rank to top 3-5
                    similarity_threshold=self.similarity_threshold  # Apply similarity threshold
                )
                
                if reranked_results:
                    # Log reranking effectiveness
                    scores_info = []
                    for r in reranked_results[:3]:
                        vec_sim = r['similarity']
                        rel_score = r['relevance_score']
                        final_score = r['final_score']
                        category = r['metadata']['category']
                        doc_title = r['metadata']['document_title'][:30]  # Truncate title
                        scores_info.append(f"{doc_title}(sim:{vec_sim:.3f},final:{final_score:.3f})")
                    logger.info(f"🎯 Top {len(reranked_results)} results: {', '.join(scores_info)}")
                else:
                    logger.info(f"⚠️ No results met similarity threshold {self.similarity_threshold}")
                
                # Apply content-specific filtering to prioritize most relevant chunks
                filtered_results = self._filter_results_by_content_relevance(reranked_results, query)
                logger.info(f"πŸ” Content filtering retained {len(filtered_results)} of {len(reranked_results)} results")
                
                # Build context from filtered results
                context = self._build_context_from_results(filtered_results)
                logger.info(f"βœ… Section-routed search returned {len(context)} characters of context")
                return context
            except Exception as e:
                logger.warning(f"❌ Section-routed search failed, falling back to keyword search: {e}")
                # Fall through to keyword search
        
        # Fallback to keyword search
        logger.info("πŸ“ Using KEYWORD SEARCH for RAG lookup")
        context = self._get_keyword_context(query, max_docs)
        logger.info(f"βœ… Keyword search returned {len(context)} characters of context")
        return context
    
    def _get_keyword_context(self, query: str, max_docs: int) -> str:
        """
        Original keyword-based context retrieval (fallback method)
        Enhanced with specific product detection
        """
        query_lower = query.lower()
        
        # Check for specific product requests and force load relevant files
        if any(keyword in query_lower for keyword in ['lightning', 'iphone', 'ipad', 'apple', 'mfi']):
            logger.info("🍎 Lightning cable query detected - loading lightning_cables.md")
            lightning_results = self._force_load_specific_document('product_manuals/lightning_cables.md')
            if lightning_results:
                return lightning_results
        
        # Search knowledge base with full query first
        results = self.kb_manager.search_documents(query)
        
        # If no results with full query, try individual important words
        if not results:
            important_words = [
                word.lower().strip('.,!?') for word in query.split() 
                if len(word) > 3 and word.lower() not in ['what', 'how', 'when', 'where', 'why', 'can', 'will', 'would', 'could', 'should', 'the', 'and', 'for', 'with']
            ]
            
            # Try each important word and combine results
            all_results = {}
            for word in important_words:
                word_results = self.kb_manager.search_documents(word)
                for result in word_results:
                    doc_key = result['path']
                    if doc_key in all_results:
                        all_results[doc_key]['relevance_score'] += result['relevance_score']
                    else:
                        all_results[doc_key] = result
            
            results = list(all_results.values())
            results.sort(key=lambda x: x['relevance_score'], reverse=True)
        
        # Filter by relevance and limit results
        relevant_results = [
            r for r in results 
            if r['relevance_score'] >= self.relevance_threshold
        ][:max_docs]
        
        if not relevant_results:
            return ""
        
        # Build context string
        context_parts = []
        total_length = 0
        
        for result in relevant_results:
            # Load document content
            content = self.kb_manager.load_document_content(result['path'])
            if not content:
                continue
            
            # Add document header
            doc_header = f"\n--- {result['title']} (Category: {result['category']}) ---\n"
            
            # Clean markdown formatting but preserve content
            cleaned_content = self._clean_markdown_simple(content)
            
            # Check if adding this document would exceed length limit
            doc_content = cleaned_content[:1200]  # Allow more content since we're not over-compressing
            combined_length = len(doc_header) + len(doc_content)
            
            if total_length + combined_length > self.max_context_length:
                # Truncate to fit within limit
                remaining_space = self.max_context_length - total_length - len(doc_header)
                if remaining_space > 200:  # Only add if we have meaningful space
                    # Try to break at a natural boundary
                    truncated_content = cleaned_content[:remaining_space]
                    last_newline = truncated_content.rfind('\n')
                    if last_newline > remaining_space * 0.7:
                        truncated_content = truncated_content[:last_newline]
                    doc_content = truncated_content + "..."
                    context_parts.append(doc_header + doc_content)
                break
            
            context_parts.append(doc_header + doc_content)
            total_length += combined_length
        
        return "".join(context_parts)
    
    def _compress_chunk(self, chunk: str, doc_title: str) -> str:
        """
        Compress chunk content into clean bulletized facts, removing markdown formatting
        Converts headers and structured content into concise factual statements
        """
        import re
        
        # Step 1: Clean up markdown formatting
        # Remove markdown headers (###, ####, etc.) and convert to clean text
        cleaned_chunk = re.sub(r'^#{1,6}\s+', '', chunk, flags=re.MULTILINE)
        
        # Convert markdown bold (**text** or __text__) to plain text
        cleaned_chunk = re.sub(r'\*\*([^*]+)\*\*', r'\1', cleaned_chunk)
        cleaned_chunk = re.sub(r'__([^_]+)__', r'\1', cleaned_chunk)
        
        # Convert markdown list items (- or *) to our bullet format
        cleaned_chunk = re.sub(r'^[\s]*[-*]\s+', 'β€’ ', cleaned_chunk, flags=re.MULTILINE)
        
        # Handle structured content like "**Key**: Value" -> "Key: Value"
        cleaned_chunk = re.sub(r'\*\*([^:*]+):\*\*\s*', r'\1: ', cleaned_chunk)
        
        # Step 2: Extract and normalize key facts
        lines = [line.strip() for line in cleaned_chunk.split('\n') if line.strip()]
        compressed_points = []
        
        # Process each line to create bulletized facts
        for line in lines:
            if len(line) < 15:  # Skip very short fragments
                continue
                
            # Skip empty bullets or redundant content
            if line in ['β€’', 'β€’ ', 'β€’  '] or line.startswith('β€’') and len(line) < 20:
                continue
                
            # Priority patterns that should be preserved
            key_patterns = [
                r'\d+\s*(days?|hours?|minutes?|months?|years?)',  # Time periods
                r'\$\d+(?:\.\d{2})?',  # Prices
                r'\d+(?:\.\d+)?\s*(?:ft|feet|inch|inches|cm|mm|gb|mb|kb|AM|PM)',  # Measurements/times
                r'warranty|guarantee|return|refund|shipping|delivery|processing',  # Policy terms
                r'compatible|supports?|works?\s+with',  # Compatibility
                r'specifications?|specs?|technical|coverage|tracking',  # Technical info
                r'cutoff|deadline|availability|monday|tuesday|wednesday|thursday|friday',  # Schedule
                r'contact|call|email|support|customer\s+service',  # Contact info
                r'free|cost|price|charge',  # Cost information
            ]
            
            # Check if line contains key information
            line_lower = line.lower()
            is_important = any(re.search(pattern, line_lower) for pattern in key_patterns)
            
            # Also preserve lines with specific product mentions or structured data
            product_terms = ['usb-c', 'hdmi', 'lightning', 'ethernet', 'displayport', 'thunderbolt', 'cable']
            has_product = any(term in line_lower for term in product_terms)
            
            # Handle structured content (like "Processing Time: ...")
            has_structure = ':' in line and len(line.split(':')) == 2
            
            if is_important or has_product or has_structure or len(compressed_points) < 2:
                # Ensure line starts with bullet point
                if not line.startswith('β€’ '):
                    line = f"β€’ {line}"
                
                # Clean up extra whitespace
                clean_line = re.sub(r'\s+', ' ', line).strip()
                
                # Remove redundant "β€’" if already present
                if clean_line.count('β€’') > 1:
                    clean_line = clean_line.replace('β€’ β€’', 'β€’', 1)
                
                if len(clean_line) > 20 and clean_line not in compressed_points:
                    compressed_points.append(clean_line)
        
        # If no good points found, extract from first meaningful content
        if not compressed_points and lines:
            for line in lines[:3]:  # Check first 3 lines
                if len(line.strip()) > 20:
                    clean_line = f"β€’ {line.strip()}"
                    compressed_points.append(clean_line)
                    break
        
        # Add source citation
        citation = f"[Source: {doc_title}]"
        
        # Combine into clean format
        result = '\n'.join(compressed_points)
        if result:
            return f"{result}\n{citation}"
        else:
            return f"β€’ {cleaned_chunk[:100]}...\n{citation}"
    
    def _force_load_specific_document(self, doc_path: str) -> str:
        """
        Force load a specific document for product-specific queries
        """
        try:
            import os
            full_path = os.path.join(self.knowledge_base_path, doc_path)
            if os.path.exists(full_path):
                with open(full_path, 'r', encoding='utf-8') as f:
                    content = f.read()
                
                # Extract document name for header
                doc_name = os.path.basename(doc_path).replace('.md', '').replace('_', ' ').title()
                
                # Format the content
                header = f"--- {doc_name} (Category: product_manuals) ---"
                cleaned_content = self._clean_markdown_simple(content)
                
                return f"{header}\n{cleaned_content}"
        except Exception as e:
            logger.error(f"Error force loading document {doc_path}: {e}")
        
        return ""
    
    def _clean_markdown_simple(self, content: str) -> str:
        """
        Clean markdown formatting but preserve content structure and details
        Simple cleanup that maintains all important information
        """
        import re
        
        # Remove markdown headers (###, ####, etc.) but keep the text
        cleaned = re.sub(r'^#{1,6}\s+', '', content, flags=re.MULTILINE)
        
        # Remove markdown bold/italic formatting but keep the text
        cleaned = re.sub(r'\*\*([^*]+)\*\*', r'\1', cleaned)  # **bold** -> bold
        cleaned = re.sub(r'__([^_]+)__', r'\1', cleaned)      # __bold__ -> bold
        cleaned = re.sub(r'\*([^*]+)\*', r'\1', cleaned)      # *italic* -> italic
        cleaned = re.sub(r'_([^_]+)_', r'\1', cleaned)        # _italic_ -> italic
        
        # Convert markdown list items to clean bullets
        cleaned = re.sub(r'^[\s]*[-*+]\s+', 'β€’ ', cleaned, flags=re.MULTILINE)
        
        # Clean up numbered lists to bullets for consistency
        cleaned = re.sub(r'^\s*\d+\.\s+', 'β€’ ', cleaned, flags=re.MULTILINE)
        
        # Remove extra whitespace but preserve structure
        lines = [line.rstrip() for line in cleaned.split('\n')]
        cleaned_lines = []
        
        for line in lines:
            # Skip empty lines between sections but keep one empty line for readability
            if not line.strip():
                if cleaned_lines and cleaned_lines[-1] != '':
                    cleaned_lines.append('')
            else:
                cleaned_lines.append(line)
        
        # Remove trailing empty lines
        while cleaned_lines and cleaned_lines[-1] == '':
            cleaned_lines.pop()
            
        return '\n'.join(cleaned_lines)

    def _build_context_from_results(self, results: List[Dict]) -> str:
        """
        Build clean, formatted context string from reranked search results
        Cleans markdown formatting but preserves all important details and specifications
        Includes document source/title for transparency
        """
        if not results:
            return ""
        
        context_parts = []
        total_length = 0
        max_context_length = self.max_context_length  # Use class setting instead of hardcoded value
        
        for result in results:
            doc_title = result['metadata']['document_title']
            category = result['metadata']['category']
            chunk_content = result['document']
            
            # Clean markdown formatting but preserve content structure
            cleaned_content = self._clean_markdown_simple(chunk_content)
            
            # FRAGMENT CLEANUP: Fix obvious fragmentation issues
            cleaned_content = self._cleanup_content_fragments(cleaned_content)
            
            # Create clean section header with source document (company policy format)
            section_type = category.replace('_', ' ').upper()
            if 'shipping' in category.lower():
                section_header = f"\nCOMPANY POLICY – SHIPPING (Source: {doc_title}):\n"
            elif 'return' in category.lower() or 'warranty' in category.lower():
                section_header = f"\nCOMPANY POLICY – RETURNS/WARRANTY (Source: {doc_title}):\n"
            elif 'product' in category.lower() or 'manual' in category.lower():
                section_header = f"\nPRODUCT SPECIFICATIONS (Source: {doc_title}):\n"
            elif 'troubleshooting' in category.lower():
                section_header = f"\nTROUBLESHOOTING GUIDE (Source: {doc_title}):\n"
            else:
                section_header = f"\nCOMPANY POLICY – {section_type} (Source: {doc_title}):\n"
            
            # Check length constraints
            combined_length = len(section_header) + len(cleaned_content)
            
            if total_length + combined_length > max_context_length:
                # Truncate content to fit, but keep the most important parts
                remaining_space = max_context_length - total_length - len(section_header)
                if remaining_space > 200:  # Only add if we have meaningful space
                    # Try to break at a natural boundary (end of line)
                    truncated_content = cleaned_content[:remaining_space]
                    last_newline = truncated_content.rfind('\n')
                    if last_newline > remaining_space * 0.7:  # If we can keep most content
                        truncated_content = truncated_content[:last_newline]
                    context_parts.append(section_header + truncated_content + "...")
                break
            
            context_parts.append(section_header + cleaned_content)
            total_length += combined_length
        
        return '\n'.join(context_parts)
    
    def _cleanup_content_fragments(self, content: str) -> str:
        """
        Clean up obvious fragmentation issues in retrieved content
        """
        if not content:
            return content
            
        lines = content.split('\n')
        cleaned_lines = []
        
        for line in lines:
            # Skip lines that start with obvious fragments
            line = line.strip()
            if not line:
                cleaned_lines.append('')
                continue
                
            # Fix common fragmentation patterns
            if line.startswith('lby '):  # "lby TrueHD" -> "Dolby TrueHD"
                line = 'Do' + line
            elif line.startswith('dth '):  # "dth: 18 Gbps" -> "Bandwidth: 18 Gbps"
                line = 'Bandwi' + line
            elif line.startswith('ort: '):  # "ort: HDR10" -> "Support: HDR10"
                line = 'Supp' + line
            elif line.startswith('res: '):  # "res: VRR, ALLM" -> "Features: VRR, ALLM"
                line = 'Featu' + line
                
            # Skip lines that are clearly incomplete fragments (less than 4 chars)
            if len(line.strip()) >= 4:
                cleaned_lines.append(line)
        
        return '\n'.join(cleaned_lines)
    
    def _filter_results_by_content_relevance(self, results: List[Dict], query: str) -> List[Dict]:
        """
        Filter and prioritize results based on content-specific keywords in the query
        This helps focus on the most relevant product specs when multiple products are retrieved
        """
        if not results:
            return results
            
        query_lower = query.lower()
        
        # Extract product-specific keywords from query
        product_keywords = {
            'hdmi': ['hdmi', '8k', '4k', '2.1', '2.0', 'ultra high speed', 'high speed'],
            'usb': ['usb-c', 'usb c', 'usb', 'charging', 'power delivery', 'pd'],
            'lightning': ['lightning', 'iphone', 'ipad', 'apple', 'mfi'],
            'audio': ['audio', '3.5mm', 'aux', 'headphone', 'speaker'],
            'ethernet': ['ethernet', 'cat5', 'cat6', 'network', 'gigabit']
        }
        
        # Determine primary product type from query
        primary_product = None
        max_matches = 0
        
        for product_type, keywords in product_keywords.items():
            matches = sum(1 for kw in keywords if kw in query_lower)
            if matches > max_matches:
                max_matches = matches
                primary_product = product_type
        
        if not primary_product:
            # If no specific product detected, return original results
            return results
        
        # Filter results to prioritize chunks containing the primary product keywords
        relevant_results = []
        secondary_results = []
        
        primary_keywords = product_keywords[primary_product]
        
        for result in results:
            content_lower = result['document'].lower()
            content_matches = sum(1 for kw in primary_keywords if kw in content_lower)
            
            # Prioritize chunks that contain multiple keywords from the primary product
            if content_matches >= 2:  # Strong match - multiple keywords
                relevant_results.append(result)
            elif content_matches >= 1:  # Weak match - single keyword
                secondary_results.append(result)
        
        # Return prioritized results: strong matches first, then weak matches, limit to original count
        final_results = relevant_results + secondary_results
        original_count = len(results)
        return final_results[:original_count]
    
    def enhance_prompt(self, user_message: str, base_prompt: str) -> str:
        """
        Enhance a prompt with relevant context from knowledge base
        """
        logger.info(f"πŸš€ PROMPT ENHANCEMENT REQUESTED for query: '{user_message}'")
        context = self.get_relevant_context(user_message)
        
        if not context:
            logger.info("❌ No relevant context found - returning base prompt")
            return base_prompt + f"\n\nUser Query: {user_message}"
        
        logger.info("βœ… Context found - enhancing prompt with RAG data")
        
        # Place the provided base_prompt at the very top and append compressed RAG context
        # Do NOT re-introduce a duplicate system lead-in here; `base_prompt` should contain that.
        enhanced_prompt = f"""{base_prompt}

        COMPANY KNOWLEDGE BASE:
        {context}

        Please use the information above to create your answer.

        """

        return enhanced_prompt
    
    def get_suggested_questions(self, category: Optional[str] = None) -> List[str]:
        """
        Get suggested questions based on knowledge base content
        """
        suggestions = []
        
        # Get documents by category or all
        if category:
            docs = self.kb_manager.get_document_by_category(category)
        else:
            docs = []
            for cat_docs in self.kb_manager.documents.values():
                docs.extend(cat_docs['documents'])
        
        # Generate suggestions based on document titles and content
        common_questions = [
            "What types of cables do you sell?",
            "What is your return policy?",
            "How long is the warranty on your products?",
            "Do you offer free shipping?",
            "How do I troubleshoot a cable that isn't working?",
            "What's the difference between USB-C and USB-A?",
            "Do you sell HDMI cables for 4K displays?",
            "How do I contact customer service?",
            "Can I return a cable if it doesn't fit my device?",
            "What payment methods do you accept?"
        ]
        
        return common_questions[:5]  # Return top 5 suggestions
    
    def analyze_query_intent(self, query: str) -> Dict[str, any]:
        """
        Analyze user query to determine intent and relevant categories
        """
        query_lower = query.lower()
        
        intent_analysis = {
            'categories': [],
            'product_types': [],
            'intent_type': 'general',
            'keywords': []
        }
        
        # Category mapping
        category_keywords = {
            'product_manuals': ['how to use', 'specifications', 'specs', 'manual', 'guide'],
            'policies': ['return', 'warranty', 'shipping', 'policy', 'refund', 'exchange'],
            'faqs': ['question', 'help', 'what is', 'how do', 'troubleshoot', 'problem']
        }
        
        # Product type keywords
        product_keywords = {
            'usb-c': ['usb-c', 'usbc', 'usb c', 'type-c'],
            'hdmi': ['hdmi', 'display', '4k', '8k', 'monitor', 'tv'],
            'usb-a': ['usb-a', 'usba', 'usb a', 'standard usb'],
            'lightning': ['lightning', 'iphone', 'ipad', 'apple']
        }
        
        # Intent type keywords
        intent_keywords = {
            'troubleshooting': ['not working', 'broken', 'fix', 'problem', 'issue', 'troubleshoot'],
            'product_inquiry': ['buy', 'purchase', 'price', 'cost', 'available', 'sell'],
            'support': ['help', 'support', 'contact', 'customer service'],
            'policy': ['return', 'warranty', 'shipping', 'policy']
        }
        
        # Analyze categories
        for category, keywords in category_keywords.items():
            if any(keyword in query_lower for keyword in keywords):
                intent_analysis['categories'].append(category)
        
        # Analyze product types
        for product, keywords in product_keywords.items():
            if any(keyword in query_lower for keyword in keywords):
                intent_analysis['product_types'].append(product)
        
        # Analyze intent type
        for intent, keywords in intent_keywords.items():
            if any(keyword in query_lower for keyword in keywords):
                intent_analysis['intent_type'] = intent
                break
        
        # Extract key terms
        important_words = [
            word for word in query_lower.split() 
            if len(word) > 3 and word not in ['what', 'how', 'when', 'where', 'why', 'can', 'will', 'would', 'could', 'should']
        ]
        intent_analysis['keywords'] = important_words[:5]
        
        return intent_analysis
    
    def get_knowledge_base_stats(self) -> Dict:
        """Get current knowledge base statistics"""
        stats = self.kb_manager.get_stats()
        
        # Add vector search information
        stats['vector_search_available'] = VECTOR_RAG_AVAILABLE
        stats['vector_search_enabled'] = self.use_vector_search
        
        if self.use_vector_search and self.vector_rag:
            try:
                vector_stats = self.vector_rag.get_collection_stats()
                stats['vector_database'] = vector_stats
            except Exception as e:
                stats['vector_database'] = {"error": str(e)}
        
        return stats
    
    def ensure_vector_index(self, force_reindex: bool = False) -> Dict:
        """
        Ensure vector index is built and up to date
        Returns indexing statistics
        """
        if not self.use_vector_search or not self.vector_rag:
            return {"error": "Vector search not available"}
        
        try:
            return self.vector_rag.index_documents(force_reindex=force_reindex)
        except Exception as e:
            return {"error": f"Failed to index documents: {e}"}

def test_rag_helper():
    """Test the RAG helper functionality"""
    print("Testing RAG Helper...")
    
    rag = RAGHelper()
    
    # Test query analysis
    test_queries = [
        "How do I return a broken USB-C cable?",
        "What HDMI cables do you sell for 4K gaming?",
        "My Lightning cable isn't charging my iPhone",
        "What is your shipping policy?"
    ]
    
    for query in test_queries:
        print(f"\n--- Testing Query: '{query}' ---")
        
        # Analyze intent
        intent = rag.analyze_query_intent(query)
        print(f"Intent Analysis: {intent}")
        
        # Get relevant context
        context = rag.get_relevant_context(query)
        print(f"Context Length: {len(context)} characters")
        
        # Show enhanced prompt (truncated)
        enhanced = rag.enhance_prompt(query)
        print(f"Enhanced Prompt Length: {len(enhanced)} characters")
        print(f"Context Preview: {context[:200]}..." if context else "No relevant context found")
    
    # Show knowledge base stats
    print(f"\nKnowledge Base Stats: {rag.get_knowledge_base_stats()}")

if __name__ == "__main__":
    test_rag_helper()