Document Question Answering
Transformers
PyTorch
English
document-processing
ocr
ner
text-classification
information-extraction
invoice
receipt
form
Instructions to use mrrobot2610/IDP-Machine-learning with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mrrobot2610/IDP-Machine-learning with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("document-question-answering", model="mrrobot2610/IDP-Machine-learning")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("mrrobot2610/IDP-Machine-learning", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download api_server.py from mrrobot2610/IDP-Machine-learning: direct link, hf CLI and curl.
- Browser
- Download file 9.54 kB
-
https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/api_server.py
- Command line
-
hf download hf://mrrobot2610/IDP-Machine-learning/api_server.py
-
curl -L -o api_server.py https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/api_server.py
9.54 kB
| """ | |
| FastAPI Server for IDP System | |
| Provides REST API endpoints for Hugging Face Spaces deployment | |
| CORS-enabled for Next.js frontend integration | |
| """ | |
| # Fix for TensorFlow/PaddlePaddle mutex warnings on macOS | |
| import os | |
| os.environ['KMP_DUPLICATE_LIB_OK'] = 'TRUE' | |
| os.environ['OMP_NUM_THREADS'] = '1' | |
| os.environ['OPENBLAS_NUM_THREADS'] = '1' | |
| os.environ['MKL_NUM_THREADS'] = '1' | |
| os.environ['VECLIB_MAXIMUM_THREADS'] = '1' | |
| os.environ['NUMEXPR_NUM_THREADS'] = '1' | |
| os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2' | |
| import warnings | |
| warnings.filterwarnings('ignore') | |
| from fastapi import FastAPI, File, UploadFile, HTTPException | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from fastapi.responses import JSONResponse | |
| import uvicorn | |
| import logging | |
| from typing import Optional | |
| import tempfile | |
| from pathlib import Path | |
| import traceback | |
| from inference_pipeline import IDPPipeline | |
| # Configure logging | |
| logging.basicConfig( | |
| level=logging.INFO, | |
| format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' | |
| ) | |
| logger = logging.getLogger(__name__) | |
| # Initialize FastAPI app | |
| app = FastAPI( | |
| title="IDP API", | |
| description="Intelligent Document Processing API for invoices, receipts, and forms", | |
| version="1.0.0" | |
| ) | |
| # Configure CORS for Next.js frontend | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=[ | |
| "*", # Allow all origins (for development) | |
| # For production, specify your Vercel domain: | |
| # "https://your-app.vercel.app", | |
| # "https://*.vercel.app", | |
| ], | |
| allow_credentials=True, | |
| allow_methods=["*"], # Allow all HTTP methods | |
| allow_headers=["*"], # Allow all headers | |
| ) | |
| # Global pipeline instance (loaded once at startup) | |
| pipeline: Optional[IDPPipeline] = None | |
| async def startup_event(): | |
| """Initialize pipeline on server startup""" | |
| global pipeline | |
| logger.info("Starting IDP API server...") | |
| logger.info("Initializing inference pipeline...") | |
| try: | |
| # Initialize with CPU by default (change use_gpu=True if GPU available) | |
| pipeline = IDPPipeline( | |
| classifier_model_path="models/classifier/best_classifier.pt", | |
| ner_model_path="models/ner/best_ner.pt", | |
| use_gpu=False, # Set to True if deploying on GPU | |
| ocr_confidence_threshold=0.5 | |
| ) | |
| logger.info("Pipeline initialized successfully!") | |
| except Exception as e: | |
| logger.error(f"Failed to initialize pipeline: {str(e)}") | |
| logger.error(traceback.format_exc()) | |
| # Continue startup anyway to allow health check | |
| async def shutdown_event(): | |
| """Cleanup on server shutdown""" | |
| logger.info("Shutting down IDP API server...") | |
| async def root(): | |
| """Root endpoint""" | |
| return { | |
| "message": "IDP API is running", | |
| "version": "1.0.0", | |
| "endpoints": { | |
| "health_check": "GET /health", | |
| "process_document": "POST /process", | |
| } | |
| } | |
| async def health_check(): | |
| """ | |
| Health check endpoint | |
| Returns status and model loading state | |
| Next.js usage: | |
| ```javascript | |
| const response = await fetch('https://your-space.hf.space/health'); | |
| const data = await response.json(); | |
| console.log(data.status); // "ok" | |
| ``` | |
| """ | |
| models_loaded = pipeline is not None | |
| return { | |
| "status": "ok", | |
| "models_loaded": models_loaded, | |
| "version": "1.0.0" | |
| } | |
| async def process_document( | |
| file: UploadFile = File(...), | |
| adaptive_threshold: bool = False, | |
| page_number: Optional[int] = None | |
| ): | |
| """ | |
| Process a document (PDF or image) and extract structured data | |
| Args: | |
| file: Uploaded file (PDF, PNG, JPEG, JPG) | |
| adaptive_threshold: Apply adaptive thresholding for poor quality scans | |
| page_number: For PDFs, process specific page (None = all pages) | |
| Returns: | |
| JSON response with extracted document fields | |
| Next.js usage: | |
| ```javascript | |
| const formData = new FormData(); | |
| formData.append('file', fileBlob); // File from input element | |
| const response = await fetch('https://your-space.hf.space/process', { | |
| method: 'POST', | |
| body: formData, | |
| }); | |
| const result = await response.json(); | |
| console.log(result.pages[0].document_type); // "INVOICE", "RECEIPT", etc. | |
| console.log(result.pages[0].fields); // Extracted fields | |
| ``` | |
| Response format: | |
| ```json | |
| { | |
| "file_type": "image" | "pdf", | |
| "total_pages": 1, | |
| "processed_pages": 1, | |
| "pages": [ | |
| { | |
| "document_type": "INVOICE", | |
| "classification_confidence": 0.96, | |
| "fields": { | |
| "invoice_number": { | |
| "value": "INV-12345", | |
| "confidence": 0.92, | |
| "bbox": [x1, y1, x2, y2], | |
| "source": "ner" | |
| }, | |
| "date": { | |
| "value": "2025-11-28", | |
| "confidence": 0.88, | |
| ... | |
| }, | |
| ... | |
| }, | |
| "processing_time": { | |
| "total": 1.23, | |
| ... | |
| } | |
| } | |
| ] | |
| } | |
| ``` | |
| """ | |
| # Check if pipeline is loaded | |
| if pipeline is None: | |
| logger.error("Pipeline not initialized") | |
| raise HTTPException( | |
| status_code=503, | |
| detail="Service unavailable: Pipeline not initialized" | |
| ) | |
| # Validate file type | |
| allowed_extensions = {'.pdf', '.png', '.jpg', '.jpeg', '.bmp', '.tiff'} | |
| file_ext = Path(file.filename).suffix.lower() | |
| if file_ext not in allowed_extensions: | |
| raise HTTPException( | |
| status_code=400, | |
| detail=f"Unsupported file type: {file_ext}. Allowed: {allowed_extensions}" | |
| ) | |
| # Check file size (limit to 10MB) | |
| max_size = 10 * 1024 * 1024 # 10 MB | |
| file_size = 0 | |
| # Save uploaded file to temporary location | |
| try: | |
| with tempfile.NamedTemporaryFile(delete=False, suffix=file_ext) as tmp: | |
| # Read and write in chunks to check size | |
| while True: | |
| chunk = await file.read(1024 * 1024) # Read 1MB at a time | |
| if not chunk: | |
| break | |
| file_size += len(chunk) | |
| if file_size > max_size: | |
| os.remove(tmp.name) | |
| raise HTTPException( | |
| status_code=413, | |
| detail=f"File too large: {file_size / (1024*1024):.2f}MB. Max: 10MB" | |
| ) | |
| tmp.write(chunk) | |
| tmp_path = tmp.name | |
| logger.info(f"Processing file: {file.filename} ({file_size / 1024:.2f}KB)") | |
| # Process document | |
| result = pipeline.process_document( | |
| file_path=tmp_path, | |
| adaptive_threshold=adaptive_threshold, | |
| page_number=page_number | |
| ) | |
| # Add filename to response | |
| result['filename'] = file.filename | |
| result['file_size_kb'] = file_size / 1024 | |
| logger.info(f"Successfully processed {file.filename}") | |
| return JSONResponse(content=result) | |
| except HTTPException: | |
| # Re-raise HTTP exceptions | |
| raise | |
| except Exception as e: | |
| logger.error(f"Error processing document: {str(e)}") | |
| logger.error(traceback.format_exc()) | |
| raise HTTPException( | |
| status_code=500, | |
| detail=f"Error processing document: {str(e)}" | |
| ) | |
| finally: | |
| # Clean up temporary file | |
| if 'tmp_path' in locals() and os.path.exists(tmp_path): | |
| try: | |
| os.remove(tmp_path) | |
| except: | |
| pass | |
| async def process_batch( | |
| files: list[UploadFile] = File(...) | |
| ): | |
| """ | |
| Process multiple documents in batch | |
| Args: | |
| files: List of uploaded files | |
| Returns: | |
| JSON response with results for each file | |
| Note: For large batches, consider using the single /process endpoint | |
| in parallel from the client side for better control | |
| """ | |
| if len(files) > 5: | |
| raise HTTPException( | |
| status_code=400, | |
| detail="Maximum 5 files per batch request" | |
| ) | |
| results = [] | |
| for file in files: | |
| try: | |
| result = await process_document(file) | |
| results.append({ | |
| "filename": file.filename, | |
| "status": "success", | |
| "data": result | |
| }) | |
| except Exception as e: | |
| logger.error(f"Error processing {file.filename}: {str(e)}") | |
| results.append({ | |
| "filename": file.filename, | |
| "status": "error", | |
| "error": str(e) | |
| }) | |
| return JSONResponse(content={"results": results}) | |
| if __name__ == "__main__": | |
| # Run server | |
| # For development: | |
| uvicorn.run( | |
| app, | |
| host="0.0.0.0", | |
| port=7860, # Default Hugging Face Spaces port | |
| log_level="info" | |
| ) | |
| # For production on Hugging Face Spaces, Dockerfile will handle this | |