Loading fairseq/modules/multihead_attention.py +1 −1 Changes for fairseq/modules/multihead_attention.py: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -129,7 +129,7 @@ class MultiheadAttention(nn.Module): float('-inf'), ).type_as(attn_weights) # FP16 support: cast to float and back attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) attn_weights = F.softmax(attn_weights, dim=-1) attn_weights = F.softmax(attn_weights.float(), dim=-1).type_as(attn_weights) attn_weights = F.dropout(attn_weights, p=self.dropout, training=self.training) attn = torch.bmm(attn_weights, v) Loading Loading
fairseq/modules/multihead_attention.py +1 −1 Changes for fairseq/modules/multihead_attention.py: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -129,7 +129,7 @@ class MultiheadAttention(nn.Module): float('-inf'), ).type_as(attn_weights) # FP16 support: cast to float and back attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) attn_weights = F.softmax(attn_weights, dim=-1) attn_weights = F.softmax(attn_weights.float(), dim=-1).type_as(attn_weights) attn_weights = F.dropout(attn_weights, p=self.dropout, training=self.training) attn = torch.bmm(attn_weights, v) Loading