Document Question Answering
Transformers
PyTorch
English
document-processing
ocr
ner
text-classification
information-extraction
invoice
receipt
form
Instructions to use mrrobot2610/IDP-Machine-learning with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mrrobot2610/IDP-Machine-learning with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("document-question-answering", model="mrrobot2610/IDP-Machine-learning")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("mrrobot2610/IDP-Machine-learning", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download examples.py from mrrobot2610/IDP-Machine-learning: direct link, hf CLI and curl.
- Browser
- Download file 9.15 kB
-
https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/examples.py
- Command line
-
hf download hf://mrrobot2610/IDP-Machine-learning/examples.py
-
curl -L -o examples.py https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/examples.py
9.15 kB
| """ | |
| Example usage and testing script for IDP system | |
| Demonstrates the complete workflow from document to structured data | |
| """ | |
| import json | |
| import time | |
| from pathlib import Path | |
| # Import components | |
| from preprocessing import DocumentPreprocessor | |
| from ocr_engine import LightweightOCR | |
| from classifier_model import DocumentClassifierInference | |
| from ner_model import DocumentNERInference | |
| from postprocessing import PostProcessor | |
| from inference_pipeline import IDPPipeline | |
| def example_1_component_by_component(): | |
| """ | |
| Example using individual components step-by-step | |
| Useful for understanding the pipeline | |
| """ | |
| print("\n" + "="*60) | |
| print("Example 1: Component-by-Component Processing") | |
| print("="*60) | |
| # Sample text (simulating OCR output) | |
| sample_text = """ | |
| INVOICE | |
| Company ABC Ltd. | |
| 123 Business Street | |
| City, State 12345 | |
| Invoice #: INV-2025-001 | |
| Date: 28/11/2025 | |
| Bill To: | |
| Customer XYZ Corp | |
| 456 Client Avenue | |
| Item Qty Price Amount | |
| Product A 10 100.00 1,000.00 | |
| Product B 5 200.00 1,000.00 | |
| Subtotal: 2,000.00 | |
| Tax (12.5%): 250.00 | |
| Total: 2,250.00 | |
| GST ID: 29ABCDE1234F1Z5 | |
| """ | |
| # Simulate OCR boxes | |
| ocr_boxes = [ | |
| {'text': 'INVOICE', 'bbox': [100, 50, 200, 70], 'confidence': 0.98}, | |
| {'text': 'INV-2025-001', 'bbox': [150, 120, 250, 140], 'confidence': 0.95}, | |
| {'text': '28/11/2025', 'bbox': [150, 150, 230, 170], 'confidence': 0.93}, | |
| {'text': '2,250.00', 'bbox': [400, 450, 480, 470], 'confidence': 0.96}, | |
| ] | |
| # Simulate NER entities | |
| ner_entities = [ | |
| {'entity': 'INVOICE_NUMBER', 'text': 'INV-2025-001', 'confidence': 0.94, 'start': 80, 'end': 92}, | |
| {'entity': 'DATE', 'text': '28/11/2025', 'confidence': 0.91, 'start': 100, 'end': 110}, | |
| {'entity': 'TOTAL_AMOUNT', 'text': '2,250.00', 'confidence': 0.93, 'start': 350, 'end': 358}, | |
| {'entity': 'TAX_AMOUNT', 'text': '250.00', 'confidence': 0.89, 'start': 330, 'end': 336}, | |
| {'entity': 'VENDOR_NAME', 'text': 'Company ABC Ltd.', 'confidence': 0.87, 'start': 20, 'end': 36}, | |
| {'entity': 'GST_ID', 'text': '29ABCDE1234F1Z5', 'confidence': 0.92, 'start': 400, 'end': 415}, | |
| ] | |
| # Post-process | |
| processor = PostProcessor() | |
| result = processor.process_document( | |
| document_type='INVOICE', | |
| classification_confidence=0.96, | |
| ocr_text=sample_text, | |
| ocr_boxes=ocr_boxes, | |
| ner_entities=ner_entities | |
| ) | |
| # Display results | |
| print(f"\nDocument Type: {result['document_type']}") | |
| print(f"Classification Confidence: {result['classification_confidence']:.3f}") | |
| print(f"\nExtracted Fields:") | |
| for field_name, field_data in result['fields'].items(): | |
| print(f"\n {field_name}:") | |
| print(f" Value: {field_data['value']}") | |
| print(f" Confidence: {field_data['confidence']:.3f}") | |
| print(f" Source: {field_data['source']}") | |
| if 'normalized' in field_data: | |
| print(f" Normalized: {field_data['normalized']}") | |
| def example_2_full_pipeline(image_path: str = None): | |
| """ | |
| Example using the complete inference pipeline | |
| This is the recommended approach for production | |
| """ | |
| print("\n" + "="*60) | |
| print("Example 2: Full Pipeline Processing") | |
| print("="*60) | |
| if not image_path: | |
| print("\nNote: No image path provided, skipping this example") | |
| print("Usage: Provide path to invoice/receipt image") | |
| print(" python examples.py /path/to/invoice.pdf") | |
| return | |
| # Initialize pipeline | |
| print("\nInitializing pipeline...") | |
| pipeline = IDPPipeline( | |
| use_gpu=False, | |
| ocr_confidence_threshold=0.5 | |
| ) | |
| # Process document | |
| print(f"\nProcessing: {image_path}") | |
| start_time = time.time() | |
| result = pipeline.process_document(image_path) | |
| elapsed = time.time() - start_time | |
| # Display results | |
| print(f"\nProcessing complete in {elapsed:.2f}s") | |
| print(f"\nFile Type: {result['file_type']}") | |
| print(f"Total Pages: {result['total_pages']}") | |
| for page in result['pages']: | |
| print(f"\n--- Page {page.get('page_number', 1)} ---") | |
| print(f"Document Type: {page['document_type']}") | |
| print(f"Classification Confidence: {page['classification_confidence']:.3f}") | |
| print(f"\nExtracted Fields ({len(page['fields'])} total):") | |
| for field_name, field_data in page['fields'].items(): | |
| print(f" • {field_name}: {field_data['value']} (conf: {field_data['confidence']:.3f})") | |
| print(f"\nProcessing Time Breakdown:") | |
| for step, duration in page['processing_time'].items(): | |
| print(f" {step}: {duration:.3f}s") | |
| # Save result | |
| output_file = "example_output.json" | |
| with open(output_file, 'w', encoding='utf-8') as f: | |
| json.dump(result, f, indent=2, ensure_ascii=False) | |
| print(f"\nFull results saved to: {output_file}") | |
| def example_3_api_client(): | |
| """ | |
| Example of calling the API programmatically | |
| (requires API server to be running) | |
| """ | |
| print("\n" + "="*60) | |
| print("Example 3: API Client Usage") | |
| print("="*60) | |
| import requests | |
| api_url = "http://localhost:7860" | |
| # Health check | |
| print(f"\nChecking API health at {api_url}...") | |
| try: | |
| response = requests.get(f"{api_url}/health") | |
| if response.ok: | |
| health = response.json() | |
| print(f"✓ API Status: {health['status']}") | |
| print(f"✓ Models Loaded: {health['models_loaded']}") | |
| else: | |
| print(f"✗ Health check failed: {response.status_code}") | |
| return | |
| except Exception as e: | |
| print(f"✗ Could not connect to API: {str(e)}") | |
| print("Note: Make sure API server is running with: python api_server.py") | |
| return | |
| # Process document | |
| # Uncomment and provide actual file path to test | |
| """ | |
| print("\nProcessing document via API...") | |
| with open('sample_invoice.pdf', 'rb') as f: | |
| files = {'file': f} | |
| response = requests.post(f"{api_url}/process", files=files) | |
| if response.ok: | |
| result = response.json() | |
| print(f"✓ Document processed successfully") | |
| print(f" Document Type: {result['pages'][0]['document_type']}") | |
| print(f" Fields Extracted: {len(result['pages'][0]['fields'])}") | |
| else: | |
| error = response.json() | |
| print(f"✗ Processing failed: {error.get('detail', 'Unknown error')}") | |
| """ | |
| def example_4_batch_processing(): | |
| """ | |
| Example of processing multiple documents | |
| """ | |
| print("\n" + "="*60) | |
| print("Example 4: Batch Processing") | |
| print("="*60) | |
| # Simulated batch of documents | |
| documents = [ | |
| {'type': 'INVOICE', 'id': 'INV-001'}, | |
| {'type': 'RECEIPT', 'id': 'REC-001'}, | |
| {'type': 'FORM', 'id': 'FORM-001'}, | |
| ] | |
| print(f"\nProcessing {len(documents)} documents...") | |
| results = [] | |
| for doc in documents: | |
| # Simulate processing | |
| result = { | |
| 'document_id': doc['id'], | |
| 'document_type': doc['type'], | |
| 'status': 'success', | |
| 'fields_extracted': 5, | |
| 'confidence': 0.92 | |
| } | |
| results.append(result) | |
| print(f" ✓ {doc['id']}: {doc['type']} (confidence: {result['confidence']:.3f})") | |
| print(f"\nBatch processing complete!") | |
| print(f" Total: {len(results)}") | |
| print(f" Success: {sum(1 for r in results if r['status'] == 'success')}") | |
| print(f" Average confidence: {sum(r['confidence'] for r in results) / len(results):.3f}") | |
| if __name__ == "__main__": | |
| import sys | |
| print("\n" + "="*60) | |
| print("IDP System - Usage Examples") | |
| print("="*60) | |
| # Run examples | |
| example_1_component_by_component() | |
| # Run full pipeline if image provided | |
| if len(sys.argv) > 1: | |
| image_path = sys.argv[1] | |
| if Path(image_path).exists(): | |
| example_2_full_pipeline(image_path) | |
| else: | |
| print(f"\nWarning: File not found: {image_path}") | |
| else: | |
| example_2_full_pipeline() # Will skip with message | |
| example_3_api_client() | |
| example_4_batch_processing() | |
| print("\n" + "="*60) | |
| print("Examples complete!") | |
| print("="*60) | |
| print("\nNext steps:") | |
| print(" 1. Train your models: python train_classifier.py && python train_ner.py") | |
| print(" 2. Test pipeline: python inference_pipeline.py your_invoice.pdf") | |
| print(" 3. Start API: python api_server.py") | |
| print(" 4. Deploy to HF Spaces: See deployment_guide.md") | |
| print("="*60 + "\n") | |