Commit d494485f authored by Alexei Baevski's avatar Alexei Baevski Committed by Myle Ott
Browse files

fix raw text for language modeling

parent 7358296b
Loading
Loading
Loading
Loading
+2 −2
Changes for fairseq/data/token_block_dataset.py: 2 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -47,7 +47,7 @@ class TokenBlockDataset(torch.utils.data.Dataset):

            self.slice_indices = [block_at(i) for i in range(length)]
        elif break_mode == 'complete':
            assert sizes is not None and sum(sizes) == len(tokens)
            assert sizes is not None and sum(sizes) == len(tokens), '{} != {}'.format(sum(sizes), len(tokens))
            tok_idx = 0
            sz_idx = 0
            curr_size = 0
@@ -62,7 +62,7 @@ class TokenBlockDataset(torch.utils.data.Dataset):
            if curr_size > 0:
                self.slice_indices.append((tok_idx, tok_idx + curr_size))
        elif break_mode == 'eos':
            assert sizes is not None and sum(sizes) == len(tokens)
            assert sizes is not None and sum(sizes) == len(tokens), '{} != {}'.format(sum(sizes), len(tokens))
            curr = 0
            for sz in sizes:
                # skip samples with just 1 example (which would be just the eos token)
+1 −1
Changes for fairseq/tasks/language_modeling.py: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -48,7 +48,7 @@ class LanguageModelingTask(FairseqTask):
        path = os.path.join(self.args.data, split)
        if self.args.raw_text and IndexedRawTextDataset.exists(path):
            ds = IndexedRawTextDataset(path, self.dictionary)
            tokens = ds.tokens_list
            tokens = [t for l in ds.tokens_list for t in l]
        elif not self.args.raw_text and IndexedInMemoryDataset.exists(path):
            ds = IndexedInMemoryDataset(path, fix_lua_indexing=True)
            tokens = ds.buffer