from trl import SFTConfig from transformers import Qwen2VLProcessor from qwen_vl_utils import process_vision_info args = SFTConfig( output_dir="qwen2-7b-instruct-amazon-description", # directory to save and repository id num_train_epochs=3, # number of training epochs per_device_train_batch_size=4, # batch size per device during training gradient_accumulation_steps=8, # number of steps before performing a backward/update pass gradient_checkpointing=True, # use gradient checkpointing to save memory optim="adamw_torch_fused", # use fused adamw optimizer logging_steps=5, # log every 10 steps save_strategy="epoch", # save checkpoint every epoch learning_rate=2e-4, # learning rate, based on QLoRA paper bf16=True, # use bfloat16 precision tf32=True, # use tf32 precision max_grad_norm=0.3, # max gradient norm based on QLoRA paper warmup_ratio=0.03, # warmup ratio based on QLoRA paper lr_scheduler_type="constant", # use constant learning rate scheduler push_to_hub=True, # push model to hub report_to="tensorboard", # report metrics to tensorboard gradient_checkpointing_kwargs = {"use_reentrant": False}, # use reentrant checkpointing dataset_text_field="", # need a dummy field for collator dataset_kwargs = {"skip_prepare_dataset": True} # important for collator ) args.remove_unused_columns=False # Create a data collator to encode text and image pairs def collate_fn(examples): # Get the texts and images, and apply the chat template texts = [processor.apply_chat_template(example["messages"], tokenize=False) for example in examples] image_inputs = [process_vision_info(example["messages"])[0] for example in examples] # Tokenize the texts and process the images batch = processor(text=texts, images=image_inputs, return_tensors="pt", padding=True) # The labels are the input_ids, and we mask the padding tokens in the loss computation labels = batch["input_ids"].clone() labels[labels == processor.tokenizer.pad_token_id] = -100 # # Ignore the image token index in the loss computation (model specific) if isinstance(processor, Qwen2VLProcessor): image_tokens = [151652,151653,151655] else: image_tokens = [processor.tokenizer.convert_tokens_to_ids(processor.image_token)] for image_token_id in image_tokens: labels[labels == image_token_id] = -100 batch["labels"] = labels return batch