from torch.distributed.algorithms.ddp_comm_hooks import powerSGD_hook def configure_ddp(model, rank): dist.init_process_group("nccl") model = DDP(model, device_ids=[rank], static_graph=True, gradient_as_bucket_view=True, bucket_cap_mb=2000) state = powerSGD_hook.PowerSGDState( process_group=None ) model.register_comm_hook( state, hook=powerSGD_hook.powerSGD_hook ) return model