Loading fairseq/data.py +20 −0 Changes for fairseq/data.py: 20 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -12,7 +12,9 @@ import math import numbers import numpy as np import os import torch from torch.autograd import Variable import torch.utils.data from fairseq.dictionary import Dictionary Loading Loading @@ -435,3 +437,21 @@ def numpy_seed(seed): yield finally: np.random.set_state(state) def get_dummy_batch(ntokens, src_dict, dst_dict, src_len=128, tgt_len=128): bsz = int(ntokens / max(src_len, tgt_len)) bsz = (bsz // 8) * 8 assert src_dict.pad() == dst_dict.pad() pad_idx = src_dict.pad() src_vocab, dst_vocab = len(src_dict), len(dst_dict) dummy_batch = {} dummy_batch['id'] = Variable(torch.arange(bsz).long().cuda()) dummy_batch['ntokens'] = tgt_len * bsz dummy_batch['target'] = Variable(torch.Tensor(bsz, tgt_len).uniform_(pad_idx + 1, dst_vocab - 1).long().cuda()) input = {} input['prev_output_tokens'] = Variable(dummy_batch['target'].data.clone()) input['src_lengths'] = Variable(torch.LongTensor(bsz).fill_(src_len).cuda()) input['src_tokens'] = Variable(torch.Tensor(bsz, src_len).uniform_(pad_idx + 1, src_vocab - 1).long().cuda()) dummy_batch['net_input'] = input return dummy_batch fairseq/distributed_utils.py +0 −52 Changes for fairseq/distributed_utils.py: 0 added lines, 52 removed lines. Original line number Diff line number Diff line Loading @@ -53,58 +53,6 @@ def suppress_output(): __builtin__.print = print def all_reduce_and_rescale_tensors(tensors, rescale_denom, buffer_size=10485760): """All-reduce and rescale tensors in chunks of the specified size. Args: tensors: list of Tensors to all-reduce rescale_denom: denominator for rescaling summed Tensors buffer_size: all-reduce chunk size in bytes """ # buffer size is in bytes, determine equiv. # of elements based on data type buffer_t = tensors[0].new(math.ceil(buffer_size / tensors[0].element_size())).zero_() buffer = [] def all_reduce_buffer(): # copy tensors into buffer_t offset = 0 for t in buffer: numel = t.numel() buffer_t[offset:offset+numel].copy_(t.view(-1)) offset += numel # all-reduce and rescale torch.distributed.all_reduce(buffer_t[:offset]) buffer_t.div_(rescale_denom) # copy all-reduced buffer back into tensors offset = 0 for t in buffer: numel = t.numel() t.view(-1).copy_(buffer_t[offset:offset+numel]) offset += numel filled = 0 for t in tensors: sz = t.numel() * t.element_size() if sz > buffer_size: # tensor is bigger than buffer, all-reduce and rescale directly torch.distributed.all_reduce(t) t.div_(rescale_denom) elif filled + sz > buffer_size: # buffer is full, all-reduce and replace buffer with grad all_reduce_buffer() buffer = [t] filled = sz else: # add tensor to buffer buffer.append(t) filled += sz if len(buffer) > 0: all_reduce_buffer() def all_gather_list(data, max_size=4096): """Gathers arbitrary data from all nodes into a list.""" world_size = torch.distributed.get_world_size() Loading fairseq/fp16_trainer.py 0 → 100644 +141 −0 Changes for fairseq/fp16_trainer.py: 141 added lines, 0 removed lines. Original line number Diff line number Diff line # Copyright (c) 2017-present, Facebook, Inc. # All rights reserved. # # This source code is licensed under the license found in the LICENSE file in # the root directory of this source tree. An additional grant of patent rights # can be found in the PATENTS file in the same directory. """ Train a network on multiple GPUs. """ import math import torch from fairseq import optim from fairseq.meters import AverageMeter from fairseq.optim import lr_scheduler from fairseq.trainer import Trainer class DynamicLossScaler: def __init__(self, init_scale=2.**15, scale_factor=2., scale_window=2000): self.loss_scale = init_scale self.scale_factor = scale_factor self.scale_window = scale_window self._iter = 0 self._last_overflow_iter = -1 def update_scale(self, overflow): if overflow: self.loss_scale /= self.scale_factor self._last_overflow_iter = self._iter elif (self._iter - self._last_overflow_iter) % self.scale_window == 0: self.loss_scale *= self.scale_factor self._iter += 1 @staticmethod def has_overflow(grad_norm): # detect inf and nan if grad_norm == float('inf') or grad_norm != grad_norm: return True return False class FP16Trainer(Trainer): """Modified trainer for FP16. We maintain two copies of the model's parameters, both in FP16 and FP32. We do forward/backward with FP16 and compute the loss + optimize with FP32. """ def __init__(self, args, model, criterion): super().__init__(args, model, criterion) # convert model to FP16 (but keep criterion FP32) self.model.half() # dynamically scale loss to reduce overflow self.scaler = DynamicLossScaler(init_scale=2.**7) self.meters['loss_scale'] = AverageMeter() def _build_optimizer(self): # create FP32 copy of parameters and grads params = [p for p in self.model.parameters() if p.requires_grad] total_param_size = sum(p.data.numel() for p in params) self.fp32_params = params[0].new(0).float().new(total_param_size) offset = 0 for p in params: numel = p.data.numel() self.fp32_params[offset:offset+numel].copy_(p.data.view(-1)) offset += numel self.fp32_params = torch.nn.Parameter(self.fp32_params) self.fp32_params.grad = self.fp32_params.data.new(total_param_size) # create optimizer using the copied FP32 params self.optimizer = optim.build_optimizer(self.args, [self.fp32_params]) self.lr_scheduler = lr_scheduler.build_lr_scheduler(self.args, self.optimizer) def save_checkpoint(self, filename, extra_state): """Save all training state in a checkpoint file.""" extra_state['loss_scale'] = self.scaler.loss_scale super().save_checkpoint(filename, extra_state) def load_checkpoint(self, filename): """Load all training state from a checkpoint file.""" extra_state = super().load_checkpoint(filename) if extra_state is not None and 'loss_scale' in extra_state: self.scaler.loss_scale = extra_state['loss_scale'] return extra_state def zero_grad(self): # zero both the FP16 and FP32 grads self.model.zero_grad() # FP16 self.optimizer.zero_grad() # FP32 def _backward(self, loss): self.meters['loss_scale'].reset() self.meters['loss_scale'].update(self.scaler.loss_scale) if loss is not None: # dynamically rescale loss to stay in FP16 range loss = loss * self.scaler.loss_scale return super()._backward(loss) def _all_reduce_and_rescale(self, grad_denom): # undo effect of dynamic loss scaling on gradients grad_denom *= self.scaler.loss_scale # all-reduce and rescale gradients grad_norm = super()._all_reduce_and_rescale(grad_denom) # detect overflow and adjust loss scale overflow = DynamicLossScaler.has_overflow(grad_norm) self.scaler.update_scale(overflow) if overflow: raise OverflowError('setting loss scale to: ' + str(self.scaler.loss_scale)) return grad_norm def _get_flat_grads(self, out=None): if out is None: out = self.fp32_params.grad return super()._get_flat_grads(out) def _set_flat_grads(self, new_grads): # no-op assert new_grads.data_ptr() == self.fp32_params.grad.data.data_ptr() def _opt(self): # take an optimization step using the FP32 params and grads super()._opt() # copy FP32 params back into FP16 model offset = 0 for p in self.model.parameters(): if not p.requires_grad: continue numel = p.data.numel() p.data.copy_(self.fp32_params.data[offset:offset+numel].view_as(p.data)) offset += numel fairseq/models/fairseq_decoder.py +1 −1 Changes for fairseq/models/fairseq_decoder.py: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -21,7 +21,7 @@ class FairseqDecoder(nn.Module): def get_normalized_probs(self, net_output, log_probs): """Get normalized probabilities (or log probs) from a net's output.""" logits = net_output[0] logits = net_output[0].float() if log_probs: return F.log_softmax(logits, dim=-1) else: Loading fairseq/options.py +3 −0 Changes for fairseq/options.py: 3 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -155,6 +155,9 @@ def add_optimization_args(parser): ' (default is to normalize by number of tokens)') group.add_argument('--update-freq', default='1', metavar='N', help='update parameters every N_i batches, when in epoch i') has_tensor_cores = torch.cuda.device_count() > 0 and torch.cuda.get_device_capability(0)[0] >= 7 group.add_argument('--fp16', action='store_true', default=has_tensor_cores, help='use FP16 during training') # Optimizer definitions can be found under fairseq/optim/ group.add_argument('--optimizer', default='nag', metavar='OPT', Loading Loading
fairseq/data.py +20 −0 Changes for fairseq/data.py: 20 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -12,7 +12,9 @@ import math import numbers import numpy as np import os import torch from torch.autograd import Variable import torch.utils.data from fairseq.dictionary import Dictionary Loading Loading @@ -435,3 +437,21 @@ def numpy_seed(seed): yield finally: np.random.set_state(state) def get_dummy_batch(ntokens, src_dict, dst_dict, src_len=128, tgt_len=128): bsz = int(ntokens / max(src_len, tgt_len)) bsz = (bsz // 8) * 8 assert src_dict.pad() == dst_dict.pad() pad_idx = src_dict.pad() src_vocab, dst_vocab = len(src_dict), len(dst_dict) dummy_batch = {} dummy_batch['id'] = Variable(torch.arange(bsz).long().cuda()) dummy_batch['ntokens'] = tgt_len * bsz dummy_batch['target'] = Variable(torch.Tensor(bsz, tgt_len).uniform_(pad_idx + 1, dst_vocab - 1).long().cuda()) input = {} input['prev_output_tokens'] = Variable(dummy_batch['target'].data.clone()) input['src_lengths'] = Variable(torch.LongTensor(bsz).fill_(src_len).cuda()) input['src_tokens'] = Variable(torch.Tensor(bsz, src_len).uniform_(pad_idx + 1, src_vocab - 1).long().cuda()) dummy_batch['net_input'] = input return dummy_batch
fairseq/distributed_utils.py +0 −52 Changes for fairseq/distributed_utils.py: 0 added lines, 52 removed lines. Original line number Diff line number Diff line Loading @@ -53,58 +53,6 @@ def suppress_output(): __builtin__.print = print def all_reduce_and_rescale_tensors(tensors, rescale_denom, buffer_size=10485760): """All-reduce and rescale tensors in chunks of the specified size. Args: tensors: list of Tensors to all-reduce rescale_denom: denominator for rescaling summed Tensors buffer_size: all-reduce chunk size in bytes """ # buffer size is in bytes, determine equiv. # of elements based on data type buffer_t = tensors[0].new(math.ceil(buffer_size / tensors[0].element_size())).zero_() buffer = [] def all_reduce_buffer(): # copy tensors into buffer_t offset = 0 for t in buffer: numel = t.numel() buffer_t[offset:offset+numel].copy_(t.view(-1)) offset += numel # all-reduce and rescale torch.distributed.all_reduce(buffer_t[:offset]) buffer_t.div_(rescale_denom) # copy all-reduced buffer back into tensors offset = 0 for t in buffer: numel = t.numel() t.view(-1).copy_(buffer_t[offset:offset+numel]) offset += numel filled = 0 for t in tensors: sz = t.numel() * t.element_size() if sz > buffer_size: # tensor is bigger than buffer, all-reduce and rescale directly torch.distributed.all_reduce(t) t.div_(rescale_denom) elif filled + sz > buffer_size: # buffer is full, all-reduce and replace buffer with grad all_reduce_buffer() buffer = [t] filled = sz else: # add tensor to buffer buffer.append(t) filled += sz if len(buffer) > 0: all_reduce_buffer() def all_gather_list(data, max_size=4096): """Gathers arbitrary data from all nodes into a list.""" world_size = torch.distributed.get_world_size() Loading
fairseq/fp16_trainer.py 0 → 100644 +141 −0 Changes for fairseq/fp16_trainer.py: 141 added lines, 0 removed lines. Original line number Diff line number Diff line # Copyright (c) 2017-present, Facebook, Inc. # All rights reserved. # # This source code is licensed under the license found in the LICENSE file in # the root directory of this source tree. An additional grant of patent rights # can be found in the PATENTS file in the same directory. """ Train a network on multiple GPUs. """ import math import torch from fairseq import optim from fairseq.meters import AverageMeter from fairseq.optim import lr_scheduler from fairseq.trainer import Trainer class DynamicLossScaler: def __init__(self, init_scale=2.**15, scale_factor=2., scale_window=2000): self.loss_scale = init_scale self.scale_factor = scale_factor self.scale_window = scale_window self._iter = 0 self._last_overflow_iter = -1 def update_scale(self, overflow): if overflow: self.loss_scale /= self.scale_factor self._last_overflow_iter = self._iter elif (self._iter - self._last_overflow_iter) % self.scale_window == 0: self.loss_scale *= self.scale_factor self._iter += 1 @staticmethod def has_overflow(grad_norm): # detect inf and nan if grad_norm == float('inf') or grad_norm != grad_norm: return True return False class FP16Trainer(Trainer): """Modified trainer for FP16. We maintain two copies of the model's parameters, both in FP16 and FP32. We do forward/backward with FP16 and compute the loss + optimize with FP32. """ def __init__(self, args, model, criterion): super().__init__(args, model, criterion) # convert model to FP16 (but keep criterion FP32) self.model.half() # dynamically scale loss to reduce overflow self.scaler = DynamicLossScaler(init_scale=2.**7) self.meters['loss_scale'] = AverageMeter() def _build_optimizer(self): # create FP32 copy of parameters and grads params = [p for p in self.model.parameters() if p.requires_grad] total_param_size = sum(p.data.numel() for p in params) self.fp32_params = params[0].new(0).float().new(total_param_size) offset = 0 for p in params: numel = p.data.numel() self.fp32_params[offset:offset+numel].copy_(p.data.view(-1)) offset += numel self.fp32_params = torch.nn.Parameter(self.fp32_params) self.fp32_params.grad = self.fp32_params.data.new(total_param_size) # create optimizer using the copied FP32 params self.optimizer = optim.build_optimizer(self.args, [self.fp32_params]) self.lr_scheduler = lr_scheduler.build_lr_scheduler(self.args, self.optimizer) def save_checkpoint(self, filename, extra_state): """Save all training state in a checkpoint file.""" extra_state['loss_scale'] = self.scaler.loss_scale super().save_checkpoint(filename, extra_state) def load_checkpoint(self, filename): """Load all training state from a checkpoint file.""" extra_state = super().load_checkpoint(filename) if extra_state is not None and 'loss_scale' in extra_state: self.scaler.loss_scale = extra_state['loss_scale'] return extra_state def zero_grad(self): # zero both the FP16 and FP32 grads self.model.zero_grad() # FP16 self.optimizer.zero_grad() # FP32 def _backward(self, loss): self.meters['loss_scale'].reset() self.meters['loss_scale'].update(self.scaler.loss_scale) if loss is not None: # dynamically rescale loss to stay in FP16 range loss = loss * self.scaler.loss_scale return super()._backward(loss) def _all_reduce_and_rescale(self, grad_denom): # undo effect of dynamic loss scaling on gradients grad_denom *= self.scaler.loss_scale # all-reduce and rescale gradients grad_norm = super()._all_reduce_and_rescale(grad_denom) # detect overflow and adjust loss scale overflow = DynamicLossScaler.has_overflow(grad_norm) self.scaler.update_scale(overflow) if overflow: raise OverflowError('setting loss scale to: ' + str(self.scaler.loss_scale)) return grad_norm def _get_flat_grads(self, out=None): if out is None: out = self.fp32_params.grad return super()._get_flat_grads(out) def _set_flat_grads(self, new_grads): # no-op assert new_grads.data_ptr() == self.fp32_params.grad.data.data_ptr() def _opt(self): # take an optimization step using the FP32 params and grads super()._opt() # copy FP32 params back into FP16 model offset = 0 for p in self.model.parameters(): if not p.requires_grad: continue numel = p.data.numel() p.data.copy_(self.fp32_params.data[offset:offset+numel].view_as(p.data)) offset += numel
fairseq/models/fairseq_decoder.py +1 −1 Changes for fairseq/models/fairseq_decoder.py: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -21,7 +21,7 @@ class FairseqDecoder(nn.Module): def get_normalized_probs(self, net_output, log_probs): """Get normalized probabilities (or log probs) from a net's output.""" logits = net_output[0] logits = net_output[0].float() if log_probs: return F.log_softmax(logits, dim=-1) else: Loading
fairseq/options.py +3 −0 Changes for fairseq/options.py: 3 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -155,6 +155,9 @@ def add_optimization_args(parser): ' (default is to normalize by number of tokens)') group.add_argument('--update-freq', default='1', metavar='N', help='update parameters every N_i batches, when in epoch i') has_tensor_cores = torch.cuda.device_count() > 0 and torch.cuda.get_device_capability(0)[0] >= 7 group.add_argument('--fp16', action='store_true', default=has_tensor_cores, help='use FP16 during training') # Optimizer definitions can be found under fairseq/optim/ group.add_argument('--optimizer', default='nag', metavar='OPT', Loading