Commit e89329d6 authored by Myle Ott's avatar Myle Ott
Browse files

Updates for latest PyTorch

parent ff68a9ef
Loading
Loading
Loading
Loading
+1 −1
Changes for fairseq/criterions/adaptive_loss.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -26,7 +26,7 @@ class AdaptiveLoss(FairseqCriterion):
        """Compute the loss for the given sample.

        Returns a tuple with three elements:
        1) the loss, as a Variable
        1) the loss
        2) the sample size, which is used as the denominator for the gradient
        3) logging outputs to display while training
        """
+1 −1
Changes for fairseq/criterions/cross_entropy.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -23,7 +23,7 @@ class CrossEntropyCriterion(FairseqCriterion):
        """Compute the loss for the given sample.

        Returns a tuple with three elements:
        1) the loss, as a Variable
        1) the loss
        2) the sample size, which is used as the denominator for the gradient
        3) logging outputs to display while training
        """
+1 −1
Changes for fairseq/criterions/fairseq_criterion.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -24,7 +24,7 @@ class FairseqCriterion(_Loss):
        """Compute the loss for the given sample.

        Returns a tuple with three elements:
        1) the loss, as a Variable
        1) the loss
        2) the sample size, which is used as the denominator for the gradient
        3) logging outputs to display while training
        """
+1 −1
Changes for fairseq/criterions/label_smoothed_cross_entropy.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -29,7 +29,7 @@ class LabelSmoothedCrossEntropyCriterion(FairseqCriterion):
        """Compute the loss for the given sample.

        Returns a tuple with three elements:
        1) the loss, as a Variable
        1) the loss
        2) the sample size, which is used as the denominator for the gradient
        3) logging outputs to display while training
        """
+10 −10
Changes for fairseq/models/fconv.py: 10 added lines, 10 removed lines.
Original line number Diff line number Diff line
@@ -565,23 +565,23 @@ def extend_conv_spec(convolutions):

def Embedding(num_embeddings, embedding_dim, padding_idx):
    m = nn.Embedding(num_embeddings, embedding_dim, padding_idx=padding_idx)
    nn.init.normal(m.weight, 0, 0.1)
    nn.init.constant(m.weight[padding_idx], 0)
    nn.init.normal_(m.weight, 0, 0.1)
    nn.init.constant_(m.weight[padding_idx], 0)
    return m


def PositionalEmbedding(num_embeddings, embedding_dim, padding_idx, left_pad):
    m = LearnedPositionalEmbedding(num_embeddings, embedding_dim, padding_idx, left_pad)
    nn.init.normal(m.weight, 0, 0.1)
    nn.init.constant(m.weight[padding_idx], 0)
    nn.init.normal_(m.weight, 0, 0.1)
    nn.init.constant_(m.weight[padding_idx], 0)
    return m


def Linear(in_features, out_features, dropout=0):
    """Weight-normalized Linear layer (input: N x T x C)"""
    m = nn.Linear(in_features, out_features)
    nn.init.normal(m.weight, mean=0, std=math.sqrt((1 - dropout) / in_features))
    nn.init.constant(m.bias, 0)
    nn.init.normal_(m.weight, mean=0, std=math.sqrt((1 - dropout) / in_features))
    nn.init.constant_(m.bias, 0)
    return nn.utils.weight_norm(m)


@@ -589,8 +589,8 @@ def LinearizedConv1d(in_channels, out_channels, kernel_size, dropout=0, **kwargs
    """Weight-normalized Conv1d layer optimized for decoding"""
    m = LinearizedConvolution(in_channels, out_channels, kernel_size, **kwargs)
    std = math.sqrt((4 * (1.0 - dropout)) / (m.kernel_size[0] * in_channels))
    nn.init.normal(m.weight, mean=0, std=std)
    nn.init.constant(m.bias, 0)
    nn.init.normal_(m.weight, mean=0, std=std)
    nn.init.constant_(m.bias, 0)
    return nn.utils.weight_norm(m, dim=2)


@@ -599,8 +599,8 @@ def ConvTBC(in_channels, out_channels, kernel_size, dropout=0, **kwargs):
    from fairseq.modules import ConvTBC
    m = ConvTBC(in_channels, out_channels, kernel_size, **kwargs)
    std = math.sqrt((4 * (1.0 - dropout)) / (m.kernel_size[0] * in_channels))
    nn.init.normal(m.weight, mean=0, std=std)
    nn.init.constant(m.bias, 0)
    nn.init.normal_(m.weight, mean=0, std=std)
    nn.init.constant_(m.bias, 0)
    return nn.utils.weight_norm(m, dim=2)


Loading