from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments from datasets import Dataset, DatasetDict, load_from_disk, concatenate_datasets import torch import os import tensorflow as tf OUTPUT_DIR = "./results" DATASET_NAME = 'dataset_with_split.csv' MODEL_NAME = "roberta-large" LOG_DIR = "./logs" SAVE_MODEL_FOLDER = "img_intents_model" POS_NAME = "POSITIVE" NEG_NAME = "NEGATIVE" BATCH_SIZE_TRAIN = 16 BATCH_SIZE_EVAL = 64 EPOCS = 10 WARMUP_STEPS = 500 tf.debugging.experimental.enable_dump_debug_info( LOG_DIR, tensor_debug_mode="FULL_HEALTH", circular_buffer_size=1000, op_regex=None, tensor_dtypes=None ) # Load the dataset from the CSV file dataset = Dataset.from_csv(DATASET_NAME) # Create a DatasetDict object containing train, validation, and test datasets datasets = DatasetDict({ 'train': dataset.filter(lambda example: example['split'] == 'train'), 'validation': dataset.filter(lambda example: example['split'] == 'validation'), 'test': dataset.filter(lambda example: example['split'] == 'test'), }) # Balance the datasets for split in datasets.keys(): num_positive = len(datasets[split].filter(lambda example: example['label'] == POS_NAME)) num_negative = len(datasets[split].filter(lambda example: example['label'] == NEG_NAME)) if num_positive > num_negative: # Downsample the positive examples datasets[split] = concatenate_datasets([ datasets[split].filter(lambda example: example['label'] == POS_NAME).shuffle(seed=42).select(range(num_negative)), datasets[split].filter(lambda example: example['label'] == NEG_NAME) ]) else: # Downsample the negative examples datasets[split] = concatenate_datasets([ datasets[split].filter(lambda example: example['label'] == POS_NAME), datasets[split].filter(lambda example: example['label'] == NEG_NAME).shuffle(seed=42).select(range(num_positive)) ]) # Shuffle the dataset to mix positive and negative examples datasets[split] = datasets[split].shuffle(seed=42) # Specify the model name model_name = MODEL_NAME # Or whatever model you want to use # Load the tokenizer associated with your model tokenizer = AutoTokenizer.from_pretrained(model_name) # Load your datasets train_dataset = datasets['train'] val_dataset = datasets['validation'] test_dataset = datasets['test'] # Preprocessing function def preprocess_function(examples): # Replace None in 'text' field with an empty string examples["text"] = [text if text is not None else "" for text in examples["text"]] # Convert labels from string to int examples["label"] = [1 if label == POS_NAME else 0 for label in examples["label"]] # Tokenize the texts return tokenizer(examples["text"], truncation=True, max_length=512, padding='max_length') train_dataset = train_dataset.map(preprocess_function, batched=True) val_dataset = val_dataset.map(preprocess_function, batched=True) test_dataset = test_dataset.map(preprocess_function, batched=True) # Make sure all your tensors are the same size for batching together train_dataset = train_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") val_dataset = val_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") test_dataset = test_dataset.remove_columns(["text"]).rename_column("label", "labels").with_format("torch") # Load a pre-trained model for sequence classification model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=2) # You have two labels: POSITIVE and NEGATIVE # TrainingArguments training_args = TrainingArguments( output_dir=OUTPUT_DIR, num_train_epochs=EPOCS, per_device_train_batch_size=BATCH_SIZE_TRAIN, # decrease this if necessary per_device_eval_batch_size=BATCH_SIZE_EVAL, warmup_steps=WARMUP_STEPS, weight_decay=0.01, logging_dir=LOG_DIR, logging_strategy='steps', # Log after every training step logging_steps=10, # Adjust this to change how often logging occurs evaluation_strategy='steps', # Evaluate after every training step eval_steps=100, # Adjust this to change how often evaluation occurs save_strategy='steps', # Save after every training step save_steps=500, # Adjust this to change how often saving occurs no_cuda=False, # use GPU gradient_accumulation_steps=2, # if necessary fp16=True, # use mixed precision training report_to='tensorboard' ) # Create a Trainer trainer = Trainer( model=model, args=training_args, train_dataset=train_dataset, eval_dataset=val_dataset, ) # Train the model trainer.train() # Save the model trainer.save_model(SAVE_MODEL_FOLDER) # Save the tokenizer tokenizer.save_pretrained(OUTPUT_DIR) # Save the training arguments torch.save(training_args, os.path.join(OUTPUT_DIR, "training_args.bin")) # Evaluate the model and print the results eval_results = trainer.evaluate(test_dataset) print(f"Test set evaluation results: {eval_results}")