Commit d6be0c7e authored by Myle Ott's avatar Myle Ott
Browse files

Use FP32 for multi-head attention softmax

parent 2d27ae08
Loading
Loading
Loading
Loading
+1 −1
Changes for fairseq/modules/multihead_attention.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -129,7 +129,7 @@ class MultiheadAttention(nn.Module):
                float('-inf'),
            ).type_as(attn_weights)  # FP16 support: cast to float and back
            attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len)
        attn_weights = F.softmax(attn_weights, dim=-1)
        attn_weights = F.softmax(attn_weights.float(), dim=-1).type_as(attn_weights)
        attn_weights = F.dropout(attn_weights, p=self.dropout, training=self.training)

        attn = torch.bmm(attn_weights, v)