""" Example usage and testing script for IDP system Demonstrates the complete workflow from document to structured data """ import json import time from pathlib import Path # Import components from preprocessing import DocumentPreprocessor from ocr_engine import LightweightOCR from classifier_model import DocumentClassifierInference from ner_model import DocumentNERInference from postprocessing import PostProcessor from inference_pipeline import IDPPipeline def example_1_component_by_component(): """ Example using individual components step-by-step Useful for understanding the pipeline """ print("\n" + "="*60) print("Example 1: Component-by-Component Processing") print("="*60) # Sample text (simulating OCR output) sample_text = """ INVOICE Company ABC Ltd. 123 Business Street City, State 12345 Invoice #: INV-2025-001 Date: 28/11/2025 Bill To: Customer XYZ Corp 456 Client Avenue Item Qty Price Amount Product A 10 100.00 1,000.00 Product B 5 200.00 1,000.00 Subtotal: 2,000.00 Tax (12.5%): 250.00 Total: 2,250.00 GST ID: 29ABCDE1234F1Z5 """ # Simulate OCR boxes ocr_boxes = [ {'text': 'INVOICE', 'bbox': [100, 50, 200, 70], 'confidence': 0.98}, {'text': 'INV-2025-001', 'bbox': [150, 120, 250, 140], 'confidence': 0.95}, {'text': '28/11/2025', 'bbox': [150, 150, 230, 170], 'confidence': 0.93}, {'text': '2,250.00', 'bbox': [400, 450, 480, 470], 'confidence': 0.96}, ] # Simulate NER entities ner_entities = [ {'entity': 'INVOICE_NUMBER', 'text': 'INV-2025-001', 'confidence': 0.94, 'start': 80, 'end': 92}, {'entity': 'DATE', 'text': '28/11/2025', 'confidence': 0.91, 'start': 100, 'end': 110}, {'entity': 'TOTAL_AMOUNT', 'text': '2,250.00', 'confidence': 0.93, 'start': 350, 'end': 358}, {'entity': 'TAX_AMOUNT', 'text': '250.00', 'confidence': 0.89, 'start': 330, 'end': 336}, {'entity': 'VENDOR_NAME', 'text': 'Company ABC Ltd.', 'confidence': 0.87, 'start': 20, 'end': 36}, {'entity': 'GST_ID', 'text': '29ABCDE1234F1Z5', 'confidence': 0.92, 'start': 400, 'end': 415}, ] # Post-process processor = PostProcessor() result = processor.process_document( document_type='INVOICE', classification_confidence=0.96, ocr_text=sample_text, ocr_boxes=ocr_boxes, ner_entities=ner_entities ) # Display results print(f"\nDocument Type: {result['document_type']}") print(f"Classification Confidence: {result['classification_confidence']:.3f}") print(f"\nExtracted Fields:") for field_name, field_data in result['fields'].items(): print(f"\n {field_name}:") print(f" Value: {field_data['value']}") print(f" Confidence: {field_data['confidence']:.3f}") print(f" Source: {field_data['source']}") if 'normalized' in field_data: print(f" Normalized: {field_data['normalized']}") def example_2_full_pipeline(image_path: str = None): """ Example using the complete inference pipeline This is the recommended approach for production """ print("\n" + "="*60) print("Example 2: Full Pipeline Processing") print("="*60) if not image_path: print("\nNote: No image path provided, skipping this example") print("Usage: Provide path to invoice/receipt image") print(" python examples.py /path/to/invoice.pdf") return # Initialize pipeline print("\nInitializing pipeline...") pipeline = IDPPipeline( use_gpu=False, ocr_confidence_threshold=0.5 ) # Process document print(f"\nProcessing: {image_path}") start_time = time.time() result = pipeline.process_document(image_path) elapsed = time.time() - start_time # Display results print(f"\nProcessing complete in {elapsed:.2f}s") print(f"\nFile Type: {result['file_type']}") print(f"Total Pages: {result['total_pages']}") for page in result['pages']: print(f"\n--- Page {page.get('page_number', 1)} ---") print(f"Document Type: {page['document_type']}") print(f"Classification Confidence: {page['classification_confidence']:.3f}") print(f"\nExtracted Fields ({len(page['fields'])} total):") for field_name, field_data in page['fields'].items(): print(f" • {field_name}: {field_data['value']} (conf: {field_data['confidence']:.3f})") print(f"\nProcessing Time Breakdown:") for step, duration in page['processing_time'].items(): print(f" {step}: {duration:.3f}s") # Save result output_file = "example_output.json" with open(output_file, 'w', encoding='utf-8') as f: json.dump(result, f, indent=2, ensure_ascii=False) print(f"\nFull results saved to: {output_file}") def example_3_api_client(): """ Example of calling the API programmatically (requires API server to be running) """ print("\n" + "="*60) print("Example 3: API Client Usage") print("="*60) import requests api_url = "http://localhost:7860" # Health check print(f"\nChecking API health at {api_url}...") try: response = requests.get(f"{api_url}/health") if response.ok: health = response.json() print(f"✓ API Status: {health['status']}") print(f"✓ Models Loaded: {health['models_loaded']}") else: print(f"✗ Health check failed: {response.status_code}") return except Exception as e: print(f"✗ Could not connect to API: {str(e)}") print("Note: Make sure API server is running with: python api_server.py") return # Process document # Uncomment and provide actual file path to test """ print("\nProcessing document via API...") with open('sample_invoice.pdf', 'rb') as f: files = {'file': f} response = requests.post(f"{api_url}/process", files=files) if response.ok: result = response.json() print(f"✓ Document processed successfully") print(f" Document Type: {result['pages'][0]['document_type']}") print(f" Fields Extracted: {len(result['pages'][0]['fields'])}") else: error = response.json() print(f"✗ Processing failed: {error.get('detail', 'Unknown error')}") """ def example_4_batch_processing(): """ Example of processing multiple documents """ print("\n" + "="*60) print("Example 4: Batch Processing") print("="*60) # Simulated batch of documents documents = [ {'type': 'INVOICE', 'id': 'INV-001'}, {'type': 'RECEIPT', 'id': 'REC-001'}, {'type': 'FORM', 'id': 'FORM-001'}, ] print(f"\nProcessing {len(documents)} documents...") results = [] for doc in documents: # Simulate processing result = { 'document_id': doc['id'], 'document_type': doc['type'], 'status': 'success', 'fields_extracted': 5, 'confidence': 0.92 } results.append(result) print(f" ✓ {doc['id']}: {doc['type']} (confidence: {result['confidence']:.3f})") print(f"\nBatch processing complete!") print(f" Total: {len(results)}") print(f" Success: {sum(1 for r in results if r['status'] == 'success')}") print(f" Average confidence: {sum(r['confidence'] for r in results) / len(results):.3f}") if __name__ == "__main__": import sys print("\n" + "="*60) print("IDP System - Usage Examples") print("="*60) # Run examples example_1_component_by_component() # Run full pipeline if image provided if len(sys.argv) > 1: image_path = sys.argv[1] if Path(image_path).exists(): example_2_full_pipeline(image_path) else: print(f"\nWarning: File not found: {image_path}") else: example_2_full_pipeline() # Will skip with message example_3_api_client() example_4_batch_processing() print("\n" + "="*60) print("Examples complete!") print("="*60) print("\nNext steps:") print(" 1. Train your models: python train_classifier.py && python train_ner.py") print(" 2. Test pipeline: python inference_pipeline.py your_invoice.pdf") print(" 3. Start API: python api_server.py") print(" 4. Deploy to HF Spaces: See deployment_guide.md") print("="*60 + "\n")