hpcaitech · flybird11111 · Aug 18, 2023 · Aug 16, 2023 · Aug 18, 2023 · Aug 18, 2023
diff --git a/colossalai/shardformer/layer/_operation.py b/colossalai/shardformer/layer/_operation.py
@@ -154,7 +154,7 @@ class _LinearWithGatherForwardReduceScatterBackward(torch.autograd.Function):
     """
 
     @staticmethod
-    def forward(ctx, input_, weight, bias, process_group, async_grad_reduce_scatter, dim, overlap):
+    def forward(ctx, input_, weight, bias, process_group, async_grad_reduce_scatter, dim, overlap=True):
         ctx.save_for_backward(input_, weight)
         ctx.use_bias = bias is not None
         ctx.process_group = process_group
@@ -217,9 +217,7 @@ def backward(ctx, grad_output):
             # do all gather in default stream
             input_ = input_.contiguous()
             world_size = dist.get_world_size(process_group)
-            rank = dist.get_rank(process_group)
             tensor_list = [torch.empty_like(input_) for _ in range(world_size)]
-            tensor_list[rank] = input_
             gather_handle = dist.all_gather(tensor_list, input_, group=process_group, async_op=True)
 
             # calculate gradient in calculate_stream
@@ -469,9 +467,7 @@ def _gather(input_, dim=-1, process_group=None):
 
     # all gather
     input_ = input_.contiguous()
-    rank = dist.get_rank(process_group)
     tensor_list = [torch.empty_like(input_) for _ in range(world_size)]
-    tensor_list[rank] = input_
     torch.distributed.all_gather(tensor_list, input_, group=process_group)
 
     # concat