# define text preprocessing def preprocess_function(examples): # return tokenized text with truncation return tokenizer(examples["text"], truncation=True) # preprocess all datasets tokenized_data = dataset_dict.map(preprocess_function, batched=True)