IDP-Machine-learning / examples.py
mrrobot2610's picture
Initial commit: IDP (Intelligent Document Processing) System
1a7ee60
Raw History Blame Contribute Delete
9.15 kB
"""
Example usage and testing script for IDP system
Demonstrates the complete workflow from document to structured data
"""
import json
import time
from pathlib import Path
# Import components
from preprocessing import DocumentPreprocessor
from ocr_engine import LightweightOCR
from classifier_model import DocumentClassifierInference
from ner_model import DocumentNERInference
from postprocessing import PostProcessor
from inference_pipeline import IDPPipeline
def example_1_component_by_component():
"""
Example using individual components step-by-step
Useful for understanding the pipeline
"""
print("\n" + "="*60)
print("Example 1: Component-by-Component Processing")
print("="*60)
# Sample text (simulating OCR output)
sample_text = """
INVOICE
Company ABC Ltd.
123 Business Street
City, State 12345
Invoice #: INV-2025-001
Date: 28/11/2025
Bill To:
Customer XYZ Corp
456 Client Avenue
Item Qty Price Amount
Product A 10 100.00 1,000.00
Product B 5 200.00 1,000.00
Subtotal: 2,000.00
Tax (12.5%): 250.00
Total: 2,250.00
GST ID: 29ABCDE1234F1Z5
"""
# Simulate OCR boxes
ocr_boxes = [
{'text': 'INVOICE', 'bbox': [100, 50, 200, 70], 'confidence': 0.98},
{'text': 'INV-2025-001', 'bbox': [150, 120, 250, 140], 'confidence': 0.95},
{'text': '28/11/2025', 'bbox': [150, 150, 230, 170], 'confidence': 0.93},
{'text': '2,250.00', 'bbox': [400, 450, 480, 470], 'confidence': 0.96},
]
# Simulate NER entities
ner_entities = [
{'entity': 'INVOICE_NUMBER', 'text': 'INV-2025-001', 'confidence': 0.94, 'start': 80, 'end': 92},
{'entity': 'DATE', 'text': '28/11/2025', 'confidence': 0.91, 'start': 100, 'end': 110},
{'entity': 'TOTAL_AMOUNT', 'text': '2,250.00', 'confidence': 0.93, 'start': 350, 'end': 358},
{'entity': 'TAX_AMOUNT', 'text': '250.00', 'confidence': 0.89, 'start': 330, 'end': 336},
{'entity': 'VENDOR_NAME', 'text': 'Company ABC Ltd.', 'confidence': 0.87, 'start': 20, 'end': 36},
{'entity': 'GST_ID', 'text': '29ABCDE1234F1Z5', 'confidence': 0.92, 'start': 400, 'end': 415},
]
# Post-process
processor = PostProcessor()
result = processor.process_document(
document_type='INVOICE',
classification_confidence=0.96,
ocr_text=sample_text,
ocr_boxes=ocr_boxes,
ner_entities=ner_entities
)
# Display results
print(f"\nDocument Type: {result['document_type']}")
print(f"Classification Confidence: {result['classification_confidence']:.3f}")
print(f"\nExtracted Fields:")
for field_name, field_data in result['fields'].items():
print(f"\n {field_name}:")
print(f" Value: {field_data['value']}")
print(f" Confidence: {field_data['confidence']:.3f}")
print(f" Source: {field_data['source']}")
if 'normalized' in field_data:
print(f" Normalized: {field_data['normalized']}")
def example_2_full_pipeline(image_path: str = None):
"""
Example using the complete inference pipeline
This is the recommended approach for production
"""
print("\n" + "="*60)
print("Example 2: Full Pipeline Processing")
print("="*60)
if not image_path:
print("\nNote: No image path provided, skipping this example")
print("Usage: Provide path to invoice/receipt image")
print(" python examples.py /path/to/invoice.pdf")
return
# Initialize pipeline
print("\nInitializing pipeline...")
pipeline = IDPPipeline(
use_gpu=False,
ocr_confidence_threshold=0.5
)
# Process document
print(f"\nProcessing: {image_path}")
start_time = time.time()
result = pipeline.process_document(image_path)
elapsed = time.time() - start_time
# Display results
print(f"\nProcessing complete in {elapsed:.2f}s")
print(f"\nFile Type: {result['file_type']}")
print(f"Total Pages: {result['total_pages']}")
for page in result['pages']:
print(f"\n--- Page {page.get('page_number', 1)} ---")
print(f"Document Type: {page['document_type']}")
print(f"Classification Confidence: {page['classification_confidence']:.3f}")
print(f"\nExtracted Fields ({len(page['fields'])} total):")
for field_name, field_data in page['fields'].items():
print(f" • {field_name}: {field_data['value']} (conf: {field_data['confidence']:.3f})")
print(f"\nProcessing Time Breakdown:")
for step, duration in page['processing_time'].items():
print(f" {step}: {duration:.3f}s")
# Save result
output_file = "example_output.json"
with open(output_file, 'w', encoding='utf-8') as f:
json.dump(result, f, indent=2, ensure_ascii=False)
print(f"\nFull results saved to: {output_file}")
def example_3_api_client():
"""
Example of calling the API programmatically
(requires API server to be running)
"""
print("\n" + "="*60)
print("Example 3: API Client Usage")
print("="*60)
import requests
api_url = "http://localhost:7860"
# Health check
print(f"\nChecking API health at {api_url}...")
try:
response = requests.get(f"{api_url}/health")
if response.ok:
health = response.json()
print(f"✓ API Status: {health['status']}")
print(f"✓ Models Loaded: {health['models_loaded']}")
else:
print(f"✗ Health check failed: {response.status_code}")
return
except Exception as e:
print(f"✗ Could not connect to API: {str(e)}")
print("Note: Make sure API server is running with: python api_server.py")
return
# Process document
# Uncomment and provide actual file path to test
"""
print("\nProcessing document via API...")
with open('sample_invoice.pdf', 'rb') as f:
files = {'file': f}
response = requests.post(f"{api_url}/process", files=files)
if response.ok:
result = response.json()
print(f"✓ Document processed successfully")
print(f" Document Type: {result['pages'][0]['document_type']}")
print(f" Fields Extracted: {len(result['pages'][0]['fields'])}")
else:
error = response.json()
print(f"✗ Processing failed: {error.get('detail', 'Unknown error')}")
"""
def example_4_batch_processing():
"""
Example of processing multiple documents
"""
print("\n" + "="*60)
print("Example 4: Batch Processing")
print("="*60)
# Simulated batch of documents
documents = [
{'type': 'INVOICE', 'id': 'INV-001'},
{'type': 'RECEIPT', 'id': 'REC-001'},
{'type': 'FORM', 'id': 'FORM-001'},
]
print(f"\nProcessing {len(documents)} documents...")
results = []
for doc in documents:
# Simulate processing
result = {
'document_id': doc['id'],
'document_type': doc['type'],
'status': 'success',
'fields_extracted': 5,
'confidence': 0.92
}
results.append(result)
print(f" ✓ {doc['id']}: {doc['type']} (confidence: {result['confidence']:.3f})")
print(f"\nBatch processing complete!")
print(f" Total: {len(results)}")
print(f" Success: {sum(1 for r in results if r['status'] == 'success')}")
print(f" Average confidence: {sum(r['confidence'] for r in results) / len(results):.3f}")
if __name__ == "__main__":
import sys
print("\n" + "="*60)
print("IDP System - Usage Examples")
print("="*60)
# Run examples
example_1_component_by_component()
# Run full pipeline if image provided
if len(sys.argv) > 1:
image_path = sys.argv[1]
if Path(image_path).exists():
example_2_full_pipeline(image_path)
else:
print(f"\nWarning: File not found: {image_path}")
else:
example_2_full_pipeline() # Will skip with message
example_3_api_client()
example_4_batch_processing()
print("\n" + "="*60)
print("Examples complete!")
print("="*60)
print("\nNext steps:")
print(" 1. Train your models: python train_classifier.py && python train_ner.py")
print(" 2. Test pipeline: python inference_pipeline.py your_invoice.pdf")
print(" 3. Start API: python api_server.py")
print(" 4. Deploy to HF Spaces: See deployment_guide.md")
print("="*60 + "\n")