from transformers import TrainingArguments args = TrainingArguments( output_dir="doplhin-dpo", # directory to save and repository id num_train_epochs=1, # number of training epochs per_device_train_batch_size=12, # batch size per device during training per_device_eval_batch_size=4, # batch size for evaluation gradient_accumulation_steps=1, # number of steps before performing a backward/update pass gradient_checkpointing=True, # use gradient checkpointing to save memory optim="adamw_torch_fused", # use fused adamw optimizer learning_rate=5e-5, # 10x higher LR than QLoRA paper max_grad_norm=0.3, # max gradient norm based on QLoRA paper warmup_ratio=0.1, # warmup ratio based on QLoRA paper lr_scheduler_type="cosine", # use cosine learning rate scheduler logging_steps=25, # log every 25 steps save_steps=500, # when to save checkpoint save_total_limit=2, # limit the total amount of checkpoints evaluation_strategy="steps", # evaluate every 1000 steps eval_steps=700, # when to evaluate bf16=True, # use bfloat16 precision tf32=True, # use tf32 precision push_to_hub=False, # push model to hub report_to="tensorboard", # report metrics to tensorboard ) dpo_args = { "beta": 0.1, # The beta factor in DPO loss. Higher beta means less divergence "loss_type": "sigmoid" # The loss type for DPO. }