# define text preprocessing def preprocess_function(examples): return tokenizer(examples["text"], padding='max_length', truncation=True) # tokenize all datasetse tokenized_data = data.map(preprocess_function, batched=True) tokenized_data.set_format(type='torch', columns=['input_ids', 'attention_mask', 'labels'])