Text Classification
Transformers
PyTorch
TensorBoard
Safetensors
English
roberta
text-embeddings-inference
Instructions to use smeintadmin/image_intents with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use smeintadmin/image_intents with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="smeintadmin/image_intents")# Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("smeintadmin/image_intents") model = AutoModelForSequenceClassification.from_pretrained("smeintadmin/image_intents", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments | |
| from datasets import Dataset, DatasetDict, load_from_disk, concatenate_datasets | |
| import torch | |
| import os | |
| import tensorflow as tf | |
| OUTPUT_DIR = "./results" | |
| DATASET_NAME = 'dataset_with_split.csv' | |
| MODEL_NAME = "roberta-large" | |
| LOG_DIR = "./logs" | |
| SAVE_MODEL_FOLDER = "img_intents_model" | |
| POS_NAME = "POSITIVE" | |
| NEG_NAME = "NEGATIVE" | |
| BATCH_SIZE_TRAIN = 16 | |
| BATCH_SIZE_EVAL = 64 | |
| EPOCS = 10 | |
| WARMUP_STEPS = 500 | |
| tf.debugging.experimental.enable_dump_debug_info( | |
| LOG_DIR, | |
| tensor_debug_mode="FULL_HEALTH", | |
| circular_buffer_size=1000, | |
| op_regex=None, | |
| tensor_dtypes=None | |
| ) | |
| # Load the dataset from the CSV file | |
| dataset = Dataset.from_csv(DATASET_NAME) | |
| # Create a DatasetDict object containing train, validation, and test datasets | |
| datasets = DatasetDict({ | |
| 'train': dataset.filter(lambda example: example['split'] == 'train'), | |
| 'validation': dataset.filter(lambda example: example['split'] == 'validation'), | |
| 'test': dataset.filter(lambda example: example['split'] == 'test'), | |
| }) | |
| # Balance the datasets | |
| for split in datasets.keys(): | |
| num_positive = len(datasets[split].filter(lambda example: example['label'] == POS_NAME)) | |
| num_negative = len(datasets[split].filter(lambda example: example['label'] == NEG_NAME)) | |
| if num_positive > num_negative: | |
| # Downsample the positive examples | |
| datasets[split] = concatenate_datasets([ | |
| datasets[split].filter(lambda example: example['label'] == POS_NAME).shuffle(seed=42).select(range(num_negative)), | |
| datasets[split].filter(lambda example: example['label'] == NEG_NAME) | |
| ]) | |
| else: | |
| # Downsample the negative examples | |
| datasets[split] = concatenate_datasets([ | |
| datasets[split].filter(lambda example: example['label'] == POS_NAME), | |
| datasets[split].filter(lambda example: example['label'] == NEG_NAME).shuffle(seed=42).select(range(num_positive)) | |
| ]) | |
| # Shuffle the dataset to mix positive and negative examples | |
| datasets[split] = datasets[split].shuffle(seed=42) | |
| # Specify the model name | |
| model_name = MODEL_NAME # Or whatever model you want to use | |
| # Load the tokenizer associated with your model | |
| tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| # Load your datasets | |
| train_dataset = datasets['train'] | |
| val_dataset = datasets['validation'] | |
| test_dataset = datasets['test'] | |
| # Preprocessing function | |
| def preprocess_function(examples): | |
| # Replace None in 'text' field with an empty string | |
| examples["text"] = [text if text is not None else "" for text in examples["text"]] | |
| # Convert labels from string to int | |
| examples["label"] = [1 if label == POS_NAME else 0 for label in examples["label"]] | |
| # Tokenize the texts | |
| return tokenizer(examples["text"], truncation=True, max_length=512, padding='max_length') | |
| train_dataset = train_dataset.map(preprocess_function, batched=True) | |
| val_dataset = val_dataset.map(preprocess_function, batched=True) | |
| test_dataset = test_dataset.map(preprocess_function, batched=True) | |
| # Make sure all your tensors are the same size for batching together | |
| train_dataset = train_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") | |
| val_dataset = val_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") | |
| test_dataset = test_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") | |
| # Load a pre-trained model for sequence classification | |
| model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=2) # You have two labels: POSITIVE and NEGATIVE | |
| # TrainingArguments | |
| training_args = TrainingArguments( | |
| output_dir=OUTPUT_DIR, | |
| num_train_epochs=EPOCS, | |
| per_device_train_batch_size=BATCH_SIZE_TRAIN, # decrease this if necessary | |
| per_device_eval_batch_size=BATCH_SIZE_EVAL, | |
| warmup_steps=WARMUP_STEPS, | |
| weight_decay=0.01, | |
| logging_dir=LOG_DIR, | |
| logging_strategy='steps', # Log after every training step | |
| logging_steps=10, # Adjust this to change how often logging occurs | |
| evaluation_strategy='steps', # Evaluate after every training step | |
| eval_steps=100, # Adjust this to change how often evaluation occurs | |
| save_strategy='steps', # Save after every training step | |
| save_steps=500, # Adjust this to change how often saving occurs | |
| no_cuda=False, # use GPU | |
| gradient_accumulation_steps=2, # if necessary | |
| fp16=True, # use mixed precision training | |
| report_to='tensorboard' | |
| ) | |
| # Create a Trainer | |
| trainer = Trainer( | |
| model=model, | |
| args=training_args, | |
| train_dataset=train_dataset, | |
| eval_dataset=val_dataset, | |
| ) | |
| # Train the model | |
| trainer.train() | |
| # Save the model | |
| trainer.save_model(SAVE_MODEL_FOLDER) | |
| # Save the tokenizer | |
| tokenizer.save_pretrained(OUTPUT_DIR) | |
| # Save the training arguments | |
| torch.save(training_args, os.path.join(OUTPUT_DIR, "training_args.bin")) | |
| # Evaluate the model and print the results | |
| eval_results = trainer.evaluate(test_dataset) | |
| print(f"Test set evaluation results: {eval_results}") |