Text Classification
Transformers
PyTorch
TensorBoard
Safetensors
English
roberta
text-embeddings-inference
Instructions to use smeintadmin/image_intents with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use smeintadmin/image_intents with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="smeintadmin/image_intents")# Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("smeintadmin/image_intents") model = AutoModelForSequenceClassification.from_pretrained("smeintadmin/image_intents", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 2,826 Bytes
1ae8986 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 | from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments
from datasets import Dataset, load_from_disk, concatenate_datasets
import os
import torch
import numpy as np
from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix
MODEL_NAME = "roberta-large"
SAVE_MODEL_FOLDER = "img_intents_model"
OUTPUT_DIR = "./results"
NEG_NAME = "NEGATIVE"
POS_NAME = "POSITIVE"
# Load the model and tokenizer
model = AutoModelForSequenceClassification.from_pretrained(SAVE_MODEL_FOLDER)
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
# Load the training arguments
training_args = torch.load(os.path.join(OUTPUT_DIR, "training_args.bin"))
# Load the sentences from the text files into lists
with open('test_positives.txt', 'r') as file:
positives_texts = [line.strip() for line in file.readlines()]
with open('test_negatives.txt', 'r') as file:
negatives_texts = [line.strip() for line in file.readlines()]
# Create datasets from the lists and add a 'label' column
positives_dataset = Dataset.from_dict({'text': positives_texts, 'label': [1]*len(positives_texts)})
negatives_dataset = Dataset.from_dict({'text': negatives_texts, 'label': [0]*len(negatives_texts)})
# Combine into a single dataset
test_dataset = concatenate_datasets([positives_dataset, negatives_dataset])
# Preprocessing function
def preprocess_function(examples):
# Tokenize the texts
return tokenizer(examples["text"], truncation=True, max_length=512, padding='max_length')
test_dataset = test_dataset.map(preprocess_function, batched=True)
# Make sure all your tensors are the same size for batching together
test_dataset = test_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch")
# Create the Trainer object
trainer = Trainer(
model=model,
args=training_args,
)
# Evaluate the model and save predictions and labels
predictions, labels, _ = trainer.predict(test_dataset)
# Convert predictions to binary (0 or 1)
binary_predictions = np.argmax(predictions, axis=1)
# Print overall metrics
accuracy = accuracy_score(labels, binary_predictions)
precision = precision_score(labels, binary_predictions)
recall = recall_score(labels, binary_predictions)
f1 = f1_score(labels, binary_predictions)
print(f"Overall accuracy: {accuracy}")
print(f"Overall precision: {precision}")
print(f"Overall recall: {recall}")
print(f"Overall F1 score: {f1}")
# Print the report for each class
cm = confusion_matrix(labels, binary_predictions)
for i, class_name in enumerate([NEG_NAME, POS_NAME]):
total = cm[i].sum()
correct = cm[i][i]
loss = total - correct
print(f"\n{class_name}:")
print(f"Total: {total}")
print(f"Confirmed: {correct}")
print(f"Loss: {loss} ({loss / total * 100:.2f}%)")
|