Document Question Answering
Transformers
PyTorch
English
document-processing
ocr
ner
text-classification
information-extraction
invoice
receipt
form
Instructions to use mrrobot2610/IDP-Machine-learning with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mrrobot2610/IDP-Machine-learning with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("document-question-answering", model="mrrobot2610/IDP-Machine-learning")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("mrrobot2610/IDP-Machine-learning", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download train_ner.py from mrrobot2610/IDP-Machine-learning: direct link, hf CLI and curl.
- Browser
- Download file 12.4 kB
-
https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/train_ner.py
- Command line
-
hf download hf://mrrobot2610/IDP-Machine-learning/train_ner.py
-
curl -L -o train_ner.py https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/train_ner.py
12.4 kB
| """ | |
| Training Script for NER Model | |
| Fine-tunes DistilBERT on document entity extraction | |
| Supports CORD and FUNSD datasets with BIO tagging | |
| """ | |
| import json | |
| import logging | |
| import os | |
| from typing import Dict, List, Tuple | |
| import numpy as np | |
| import torch | |
| import torch.nn as nn | |
| from seqeval.metrics import ( | |
| classification_report, | |
| f1_score, | |
| precision_score, | |
| recall_score, | |
| ) | |
| from sklearn.model_selection import train_test_split | |
| from torch.optim import AdamW | |
| from torch.utils.data import DataLoader, Dataset | |
| from tqdm import tqdm | |
| from transformers import AutoTokenizer, get_linear_schedule_with_warmup | |
| from dataset_loader import UnifiedDatasetLoader | |
| from ner_model import DocumentNERModel | |
| logging.basicConfig(level=logging.INFO) | |
| logger = logging.getLogger(__name__) | |
| class NERDataset(Dataset): | |
| """PyTorch Dataset for NER""" | |
| def __init__(self, tokenized_inputs: Dict, labels: List[List[int]]): | |
| self.input_ids = tokenized_inputs["input_ids"] | |
| self.attention_mask = tokenized_inputs["attention_mask"] | |
| self.labels = labels | |
| def __len__(self): | |
| return len(self.input_ids) | |
| def __getitem__(self, idx): | |
| return { | |
| "input_ids": torch.tensor(self.input_ids[idx], dtype=torch.long), | |
| "attention_mask": torch.tensor(self.attention_mask[idx], dtype=torch.long), | |
| "labels": torch.tensor(self.labels[idx], dtype=torch.long), | |
| } | |
| class NERDataPreprocessor: | |
| """Preprocess NER data with proper tokenization and label alignment""" | |
| def __init__(self, tokenizer, label2id: Dict, max_length: int = 512): | |
| self.tokenizer = tokenizer | |
| self.label2id = label2id | |
| self.max_length = max_length | |
| def tokenize_and_align_labels( | |
| self, examples: List[Dict] | |
| ) -> Tuple[Dict, List[List[int]]]: | |
| """ | |
| Tokenize texts and align labels with subword tokens | |
| Args: | |
| examples: List of dicts with 'tokens' and 'labels' | |
| Returns: | |
| Tuple of (tokenized_inputs, aligned_labels) | |
| """ | |
| tokenized_inputs = {"input_ids": [], "attention_mask": []} | |
| aligned_labels = [] | |
| for example in examples: | |
| tokens = example["tokens"] | |
| labels = example["labels"] | |
| # Tokenize with word boundaries | |
| encoding = self.tokenizer( | |
| tokens, | |
| is_split_into_words=True, | |
| max_length=self.max_length, | |
| padding="max_length", | |
| truncation=True, | |
| return_tensors=None, | |
| ) | |
| # Align labels | |
| word_ids = encoding.word_ids() | |
| label_ids = [] | |
| previous_word_idx = None | |
| for word_idx in word_ids: | |
| if word_idx is None: | |
| # Special token (CLS, SEP, PAD) | |
| label_ids.append(-100) | |
| elif word_idx != previous_word_idx: | |
| # First subword of a word | |
| if word_idx < len(labels): | |
| label = labels[word_idx] | |
| label_ids.append(self.label2id.get(label, self.label2id["O"])) | |
| else: | |
| label_ids.append(-100) | |
| else: | |
| # Continuation of previous word (subword) | |
| # Use same label or -100 | |
| if word_idx < len(labels): | |
| label = labels[word_idx] | |
| # Convert B- to I- for subwords | |
| if label.startswith("B-"): | |
| label = "I-" + label[2:] | |
| label_ids.append(self.label2id.get(label, self.label2id["O"])) | |
| else: | |
| label_ids.append(-100) | |
| previous_word_idx = word_idx | |
| tokenized_inputs["input_ids"].append(encoding["input_ids"]) | |
| tokenized_inputs["attention_mask"].append(encoding["attention_mask"]) | |
| aligned_labels.append(label_ids) | |
| return tokenized_inputs, aligned_labels | |
| class NERTrainer: | |
| """Trainer for NER model""" | |
| def __init__( | |
| self, | |
| model_name: str = "distilbert-base-uncased", | |
| output_dir: str = "models/ner", | |
| device: str = None, | |
| ): | |
| self.model_name = model_name | |
| self.output_dir = output_dir | |
| self.device = ( | |
| device if device else ("cuda" if torch.cuda.is_available() else "cpu") | |
| ) | |
| os.makedirs(output_dir, exist_ok=True) | |
| logger.info(f"NER Trainer initialized on device: {self.device}") | |
| if self.device == "cuda": | |
| logger.info(f"Using GPU: {torch.cuda.get_device_name(0)}") | |
| def prepare_data(self, test_size: float = 0.1, val_size: float = 0.1): | |
| """Load and prepare NER datasets""" | |
| logger.info("Loading NER datasets...") | |
| # Load data | |
| loader = UnifiedDatasetLoader() | |
| train_data = loader.load_ner_dataset( | |
| datasets=["cord", "sroie", "funsd"], split="train" | |
| ) | |
| # Get label mappings | |
| mappings = loader.get_label_mappings() | |
| self.label2id = mappings["ner"] | |
| self.id2label = mappings["ner_id2label"] | |
| logger.info(f"Loaded {len(train_data)} examples") | |
| logger.info(f"Number of labels: {len(self.label2id)}") | |
| # Split data | |
| train_examples, temp_examples = train_test_split( | |
| train_data, test_size=(test_size + val_size), random_state=42 | |
| ) | |
| val_examples, test_examples = train_test_split( | |
| temp_examples, test_size=test_size / (test_size + val_size), random_state=42 | |
| ) | |
| logger.info( | |
| f"Train: {len(train_examples)}, Val: {len(val_examples)}, Test: {len(test_examples)}" | |
| ) | |
| return train_examples, val_examples, test_examples | |
| def train( | |
| self, | |
| train_examples: List[Dict], | |
| val_examples: List[Dict], | |
| num_epochs: int = 25, | |
| batch_size: int = 8, | |
| learning_rate: float = 3e-5, | |
| warmup_ratio: float = 0.1, | |
| early_stopping_patience: int = 5, | |
| ): | |
| """Train the NER model""" | |
| # Load tokenizer and model | |
| tokenizer = AutoTokenizer.from_pretrained(self.model_name) | |
| model = DocumentNERModel( | |
| model_name=self.model_name, num_labels=len(self.label2id) | |
| ) | |
| model.to(self.device) | |
| # Preprocess data | |
| preprocessor = NERDataPreprocessor(tokenizer, self.label2id) | |
| logger.info("Preprocessing training data...") | |
| train_inputs, train_labels = preprocessor.tokenize_and_align_labels( | |
| train_examples | |
| ) | |
| train_dataset = NERDataset(train_inputs, train_labels) | |
| logger.info("Preprocessing validation data...") | |
| val_inputs, val_labels = preprocessor.tokenize_and_align_labels(val_examples) | |
| val_dataset = NERDataset(val_inputs, val_labels) | |
| # Create dataloaders | |
| train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True) | |
| val_loader = DataLoader(val_dataset, batch_size=batch_size) | |
| # Optimizer and scheduler | |
| optimizer = AdamW(model.parameters(), lr=learning_rate) | |
| total_steps = len(train_loader) * num_epochs | |
| warmup_steps = int(total_steps * warmup_ratio) | |
| scheduler = get_linear_schedule_with_warmup( | |
| optimizer, num_warmup_steps=warmup_steps, num_training_steps=total_steps | |
| ) | |
| # Training loop | |
| best_val_f1 = 0 | |
| patience_counter = 0 | |
| for epoch in range(num_epochs): | |
| logger.info(f"\nEpoch {epoch + 1}/{num_epochs}") | |
| # Training | |
| model.train() | |
| train_loss = 0 | |
| progress_bar = tqdm(train_loader, desc="Training") | |
| for batch in progress_bar: | |
| optimizer.zero_grad() | |
| input_ids = batch["input_ids"].to(self.device) | |
| attention_mask = batch["attention_mask"].to(self.device) | |
| labels = batch["labels"].to(self.device) | |
| outputs = model(input_ids, attention_mask, labels) | |
| loss = outputs["loss"] | |
| loss.backward() | |
| torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0) | |
| optimizer.step() | |
| scheduler.step() | |
| train_loss += loss.item() | |
| progress_bar.set_postfix({"loss": loss.item()}) | |
| avg_train_loss = train_loss / len(train_loader) | |
| logger.info(f"Train Loss: {avg_train_loss:.4f}") | |
| # Validation | |
| model.eval() | |
| val_loss = 0 | |
| all_preds = [] | |
| all_labels = [] | |
| with torch.no_grad(): | |
| for batch in tqdm(val_loader, desc="Validation"): | |
| input_ids = batch["input_ids"].to(self.device) | |
| attention_mask = batch["attention_mask"].to(self.device) | |
| labels = batch["labels"].to(self.device) | |
| outputs = model(input_ids, attention_mask, labels) | |
| val_loss += outputs["loss"].item() | |
| preds = torch.argmax(outputs["logits"], dim=-1) | |
| # Collect predictions and labels (excluding -100) | |
| for pred_seq, label_seq in zip(preds, labels): | |
| pred_list = [] | |
| label_list = [] | |
| for p, l in zip(pred_seq, label_seq): | |
| if l != -100: | |
| pred_list.append(self.id2label[p.item()]) | |
| label_list.append(self.id2label[l.item()]) | |
| if pred_list: | |
| all_preds.append(pred_list) | |
| all_labels.append(label_list) | |
| avg_val_loss = val_loss / len(val_loader) | |
| # Calculate metrics using seqeval | |
| val_f1 = f1_score(all_labels, all_preds) | |
| val_precision = precision_score(all_labels, all_preds) | |
| val_recall = recall_score(all_labels, all_preds) | |
| logger.info(f"Val Loss: {avg_val_loss:.4f}") | |
| logger.info( | |
| f"Val F1: {val_f1:.4f}, Precision: {val_precision:.4f}, Recall: {val_recall:.4f}" | |
| ) | |
| # Early stopping | |
| if val_f1 > best_val_f1: | |
| best_val_f1 = val_f1 | |
| patience_counter = 0 | |
| # Save best model | |
| model_path = os.path.join(self.output_dir, "best_ner.pt") | |
| torch.save(model.state_dict(), model_path) | |
| logger.info(f"Saved best model with F1: {best_val_f1:.4f}") | |
| else: | |
| patience_counter += 1 | |
| if patience_counter >= early_stopping_patience: | |
| logger.info(f"Early stopping triggered after {epoch + 1} epochs") | |
| break | |
| # Save final model and metadata | |
| torch.save(model.state_dict(), os.path.join(self.output_dir, "final_ner.pt")) | |
| metadata = { | |
| "model_name": self.model_name, | |
| "num_labels": len(self.label2id), | |
| "label2id": self.label2id, | |
| "id2label": self.id2label, | |
| "best_val_f1": best_val_f1, | |
| } | |
| with open(os.path.join(self.output_dir, "metadata.json"), "w") as f: | |
| json.dump(metadata, f, indent=2) | |
| logger.info("NER training complete!") | |
| return best_val_f1 | |
| if __name__ == "__main__": | |
| # Training configuration | |
| trainer = NERTrainer(model_name="distilbert-base-uncased", output_dir="models/ner") | |
| # Prepare data | |
| train_examples, val_examples, test_examples = trainer.prepare_data( | |
| test_size=0.1, val_size=0.1 | |
| ) | |
| # Train model | |
| best_f1 = trainer.train( | |
| train_examples=train_examples, | |
| val_examples=val_examples, | |
| num_epochs=30, | |
| batch_size=32, | |
| learning_rate=3e-5, | |
| early_stopping_patience=5, | |
| ) | |
| print(f"\nBest validation F1: {best_f1:.4f}") | |