YAML Metadata Warning:empty or missing yaml metadata in repo card
Check out the documentation for more information.
- Load datasets
- Import necessary libraries
- Set environment variable to adjust MPS memory usage
- Load the IELTS Evaluations dataset
- Initialize the tokenizer
- Define the tokenization function for IELTS Evaluations
- Apply tokenization to the dataset
- Set up training arguments with reduced batch size and disabled fp16
- Initialize the model
- Initialize the Trainer
- Train the model
- Evaluate the model
- Save the trained model and tokenizer
from transformers import BartTokenizer, BartForConditionalGeneration, Trainer, TrainingArguments from datasets import load_dataset import torch
Load datasets
ielts_evaluations = load_dataset("chillies/IELTS_evaluations") ielts_feedback = load_dataset("chillies/IELTS_essay_human_feedback") wi_locness = load_dataset("bea2019st/wi_locness", 'wi')
Import necessary libraries
from transformers import T5Tokenizer, T5ForConditionalGeneration, Trainer, TrainingArguments from datasets import load_dataset import torch import os
Set environment variable to adjust MPS memory usage
os.environ['PYTORCH_MPS_HIGH_WATERMARK_RATIO'] = '0.0'
Load the IELTS Evaluations dataset
ielts_evaluations = load_dataset("chillies/IELTS_evaluations")
Initialize the tokenizer
tokenizer = T5Tokenizer.from_pretrained('t5-large')
Define the tokenization function for IELTS Evaluations
def tokenize_ielts_evaluations(examples, max_length=512): inputs = examples['essay'] targets = examples['evaluation'] # The model's training output is the evaluation score model_inputs = tokenizer(inputs, max_length=max_length, truncation=True, padding="max_length", return_tensors="pt")
# Tokenize the targets (evaluation scores)
labels = tokenizer(targets, max_length=max_length, truncation=True, padding="max_length", return_tensors="pt").input_ids
# Replace padding token ID in labels with -100 to ignore these tokens during loss calculation
labels = torch.where(labels == tokenizer.pad_token_id, -100, labels)
model_inputs["labels"] = labels
return model_inputs
Apply tokenization to the dataset
ielts_evaluations = ielts_evaluations.map(tokenize_ielts_evaluations, batched=True)
Set up training arguments with reduced batch size and disabled fp16
training_args_evaluations = TrainingArguments( output_dir='./results_ielts_evaluations', evaluation_strategy="epoch", learning_rate=2e-5, per_device_train_batch_size=1, # Further reduced batch size per_device_eval_batch_size=1, # Further reduced eval batch size num_train_epochs=3, weight_decay=0.01, gradient_accumulation_steps=4, # Adjust as needed fp16=False, # Disable fp16 as MPS backend does not support it bf16=True # Enable bf16 if using MPS backend )
Initialize the model
model = T5ForConditionalGeneration.from_pretrained('t5-large')
Initialize the Trainer
trainer_evaluations = Trainer( model=model, args=training_args_evaluations, train_dataset=ielts_evaluations['train'], eval_dataset=ielts_evaluations['test'], )
Train the model
trainer_evaluations.train()
Evaluate the model
eval_results_evaluations = trainer_evaluations.evaluate() print(f"Evaluation results (Overall Band Score Prediction): {eval_results_evaluations}")
Save the trained model and tokenizer
model.save_pretrained('./trained_t5_model_ielts_evaluations') tokenizer.save_pretrained('./trained_t5_model_ielts_evaluations')