from PIL import Image # load dataset dataset = load_dataset("shawhin/yt-title-thumbnail-pairs") # define preprocessing function def preprocess(batch): """ Preprocessing data without augmentations for test set """ # get images from urls image_list = [Image.open(requests.get(url, stream=True).raw) for url in batch["thumbnail_url"]] # return columns with standard names return { "anchor": image_list, "positive": batch["title"], "negative": batch["title_neg"] } # remove columns not relevant to training columns_to_remove = [col for col in dataset['train'].column_names if col not in ['anchor', 'positive', 'negative']] # apply transformations dataset = dataset.map(preprocess, batched=True, remove_columns=columns_to_remove)