Compare commits

...
Author SHA1 Message Date
Sam Shleifer c2788f19aa boom boom 2020-08-21 19:55:50 -04:00
Sam Shleifer 6aaac683c5 merge batch parity 2020-08-21 19:54:03 -04:00
Sam Shleifer d68c671132 Merge branch 'master' into dropper-celoss 2020-08-21 19:50:36 -04:00
Sam Shleifer f5d68bb13b fix tests 2020-08-21 17:15:54 -04:00
Sam Shleifer 0da45503e8 split out bart tokenizer tests 2020-08-21 17:09:37 -04:00
Sam Shleifer aebbe11a67 Merge branch 'master' into batch-parity-cleaner 2020-08-21 16:31:49 -04:00
Sam Shleifer f9613aaf24 Merge remote-tracking branch 'upstream/master' into batch-parity-cleaner 2020-08-20 14:37:55 -04:00
Sam Shleifer b2c039bfad boom boom 2020-08-20 11:37:02 -04:00
Sam Shleifer f328679d93 boom boom 2020-08-20 11:31:40 -04:00
Sam Shleifer 2207e5d8cb tests pass 2020-08-19 22:41:38 -04:00
Sam Shleifer 75978bf5e9 batch parity 2020-08-19 22:39:03 -04:00
Sam Shleifer 018c1bb3da broken test 2020-08-19 14:33:04 -04:00
Sam Shleifer b9ca11e3ae Dropc 0.5 2020-08-18 22:03:35 -04:00
Sam Shleifer 6bdf998dff dropc03 2020-08-18 10:17:15 -04:00
Sam Shleifer fcb96a2c1e test dropc=0, ce loss 2020-08-17 23:48:58 -04:00
Sam Shleifer e5a3f09cd4 dropc_zero 2020-08-17 23:44:48 -04:00
Sam Shleifer 7d5bdb82eb Merge branch 'master' into dropper 2020-08-17 23:42:36 -04:00
Sam Shleifer 556fdea687 boom boom 2020-08-16 22:50:05 -04:00
Sam Shleifer 9ba04a609e Merge branch 'master' into dropper 2020-08-16 22:50:02 -04:00
Sam Shleifer d63f811af5 Doesnt break 2020-08-16 22:43:36 -04:00
Sam Shleifer faac0718cb boom boom 2020-08-16 21:36:09 -04:00
Sam Shleifer 7618c08e55 merge master 2020-08-16 21:06:19 -04:00
Sam Shleifer 1734ba169b asked for help 2020-07-22 10:54:11 -04:00
18 changed files with 254 additions and 162 deletions
+4 -4
View File
@@ -106,7 +106,7 @@ The following command should work on a 16GB GPU:
--train_batch_size=1 \
--eval_batch_size=1 \
--output_dir=xsum_results \
--num_train_epochs 1 \
--num_train_epochs 6 \
--model_name_or_path facebook/bart-large
```
@@ -122,8 +122,8 @@ Best performing command:
export ENRO_DIR='wmt_en_ro' # Download instructions above
# export WANDB_PROJECT="MT" # optional
export MAX_LEN=128
export BS=4
./train_mbart_cc25_enro.sh --output_dir enro_finetune_baseline --label_smoothing 0.1 --fp16_opt_level=O1 --logger_name wandb --sortish_sampler
export BS=8
./train_mbart_cc25_enro.sh --output_dir enro_finetune_baseline_dropper --label_smoothing 0 --fp16_opt_level=O1 --logger_name wandb --sortish_sampler
```
This should take < 6h/epoch on a 16GB v100 and achieve test BLEU above 26
To get results in line with fairseq, you need to do some postprocessing. (see `romanian_postprocessing.md`)
@@ -141,7 +141,7 @@ export BS=4
As you train, `output_dir` will be filled with files, that look kind of like this (comments are mine).
Some of them are metrics, some of them are checkpoints, some of them are metadata. Here is a quick tour:
```bash
```
output_dir
├── best_tfmr # this is a huggingface checkpoint generated by save_pretrained. It is the same model as the PL .ckpt file below
│   ├── config.json
+25 -5
View File
@@ -14,6 +14,7 @@ from torch.utils.data import DataLoader
from lightning_base import BaseTransformer, add_generic_args, generic_train
from transformers import MarianTokenizer, MBartTokenizer, T5ForConditionalGeneration
from transformers.modeling_bart import shift_tokens_right
try:
@@ -36,6 +37,7 @@ try:
)
from .callbacks import Seq2SeqLoggingCallback, get_checkpoint_callback, get_early_stopping_callback
from .loss_dropper import LossDropper
except ImportError:
from utils import (
Seq2SeqDataset,
@@ -55,19 +57,21 @@ except ImportError:
label_smoothed_nll_loss,
)
from callbacks import Seq2SeqLoggingCallback, get_checkpoint_callback, get_early_stopping_callback
from loss_dropper import LossDropper
logger = logging.getLogger(__name__)
class SummarizationModule(BaseTransformer):
mode = "summarization"
loss_names = ["loss"]
loss_names = ["loss", "dropper_mask_mean"]
metric_names = ROUGE_KEYS
val_metric = "rouge2"
def __init__(self, hparams, **kwargs):
super().__init__(hparams, num_labels=None, mode=self.mode, **kwargs)
use_task_specific_params(self.model, "summarization")
self.dropper = LossDropper(dropc=.05)
save_git_info(self.hparams.output_dir)
self.metrics_save_path = Path(self.output_dir) / "metrics.json"
self.hparams_save_path = Path(self.output_dir) / "hparams.pkl"
@@ -137,7 +141,10 @@ class SummarizationModule(BaseTransformer):
pad_token_id = self.tokenizer.pad_token_id
source_ids, source_mask, target_ids = batch["input_ids"], batch["attention_mask"], batch["decoder_input_ids"]
if isinstance(self.model, T5ForConditionalGeneration):
if "labels" in batch:
lm_labels = batch["labels"]
decoder_input_ids = shift_tokens_right(lm_labels, pad_token_id)
elif isinstance(self.model, T5ForConditionalGeneration):
decoder_input_ids = self.model._shift_right(target_ids)
lm_labels = target_ids
else:
@@ -145,19 +152,32 @@ class SummarizationModule(BaseTransformer):
lm_labels = target_ids[:, 1:].clone() # why clone?
outputs = self(source_ids, attention_mask=source_mask, decoder_input_ids=decoder_input_ids, use_cache=False)
bs = source_ids.shape[0]
if self.hparams.label_smoothing == 0:
# Same behavior as modeling_bart.py
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=pad_token_id)
loss_fct = torch.nn.CrossEntropyLoss(reduction='none', ignore_index=pad_token_id)
lm_logits = outputs[0]
assert lm_logits.shape[-1] == self.model.config.vocab_size
#loss_fct = torch.nn.NLLLoss(reduction='none', ignore_index=pad_token_id)
#logit_shape =
#weights = torch.ones(logit_shape
loss = loss_fct(lm_logits.view(-1, lm_logits.shape[-1]), lm_labels.view(-1))
loss = loss.view(-1, bs)
loss = loss.mean(dim=0)
mask = self.dropper(loss)
loss *= mask
loss = loss.mean()
return (loss, 1-mask.mean())
#loss = loss.view(-1, bs)
else:
lprobs = torch.nn.functional.log_softmax(outputs[0], dim=-1)
loss, nll_loss = label_smoothed_nll_loss(
lprobs, lm_labels, self.hparams.label_smoothing, ignore_index=pad_token_id
)
return (loss,)
return (loss,torch.tensor(1.))
@property
def pad(self) -> int:
@@ -302,6 +322,7 @@ class SummarizationModule(BaseTransformer):
"--task", type=str, default="summarization", required=False, help="# examples. -1 means use all."
)
parser.add_argument("--label_smoothing", type=float, default=0.0, required=False)
parser.add_argument("--loss_dropper", type=float, default=0.0, required=False)
parser.add_argument("--src_lang", type=str, default="", required=False)
parser.add_argument("--tgt_lang", type=str, default="", required=False)
parser.add_argument(
@@ -316,7 +337,6 @@ class SummarizationModule(BaseTransformer):
class TranslationModule(SummarizationModule):
mode = "translation"
loss_names = ["loss"]
metric_names = ["bleu"]
val_metric = "bleu"
+53
View File
@@ -0,0 +1,53 @@
import numpy as np
import torch.nn as nn
class LossDropper(nn.Module):
def __init__(
self,
dropc=0.4,
min_count=10000,
recompute=10000,
verbose=True
):
super().__init__()
self.keepc = 1. - dropc
self.count = 0
self.min_count = min_count
self.recompute = recompute
self.last_computed = 0
self.percentile_val = 100000000.
self.cur_idx = 0
self.verbose = verbose
self.vals = np.zeros(self.recompute, dtype=np.float32)
def forward(self, loss):
if loss is None:
return loss
self.last_computed += loss.numel()
self.count += loss.numel()
if self.count < len(self.vals):
self.vals[self.count - loss.numel():self.count] = loss.detach().cpu().numpy().flatten()
self.cur_idx += loss.numel()
return (loss < np.inf).type(loss.dtype)
else:
for idx, item in enumerate(loss):
self.vals[self.cur_idx] = item
self.cur_idx += 1
if self.cur_idx >= len(self.vals):
self.cur_idx = 0
if self.count < self.min_count:
return (loss < np.inf).type(loss.dtype)
if self.last_computed > self.recompute:
self.percentile_val = np.percentile(self.vals, self.keepc * 100)
if self.verbose:
print('Using cutoff', self.percentile_val)
self.last_computed = 0
mask = (loss < self.percentile_val).type(loss.dtype)
return mask
+1 -1
View File
@@ -6,7 +6,7 @@ export GAS=1
python finetune.py \
--learning_rate=3e-5 \
--fp16 \
--fp16 --fp16_opt_level=O1 \
--gpus 1 \
--do_train \
--do_predict \
+1 -1
View File
@@ -6,7 +6,7 @@ python distillation.py \
--learning_rate=3e-4 \
--do_train \
--do_predict \
--fp16 \
--fp16 --fp16_opt_level=O1 \
--val_check_interval 0.1 --n_val 1000 \
--teacher facebook/bart-large-xsum --data_dir $XSUM_DIR \
--max_target_length=60 --val_max_target_length=60 --test_max_target_length=100 \
@@ -14,5 +14,4 @@ python finetune.py \
--task translation \
--warmup_steps 500 \
--freeze_embeds \
--model_name_or_path=facebook/mbart-large-cc25 \
"$@"
-1
View File
@@ -299,7 +299,6 @@ if is_torch_available():
BartModel,
BartForConditionalGeneration,
BartForQuestionAnswering,
BART_PRETRAINED_MODEL_ARCHIVE_LIST,
)
from .modeling_mbart import MBartForConditionalGeneration
from .modeling_marian import MarianMTModel
+2 -9
View File
@@ -51,15 +51,7 @@ _CONFIG_FOR_DOC = "BartConfig"
_TOKENIZER_FOR_DOC = "BartTokenizer"
BART_PRETRAINED_MODEL_ARCHIVE_LIST = [
"facebook/bart-base",
"facebook/bart-large",
"facebook/bart-large-mnli",
"facebook/bart-large-cnn",
"facebook/bart-large-xsum",
"facebook/mbart-large-en-ro",
# See all BART models at https://huggingface.co/models?filter=bart
]
# See all BART models at https://huggingface.co/models?filter=bart
BART_START_DOCSTRING = r"""
@@ -1023,6 +1015,7 @@ class BartForConditionalGeneration(PretrainedBartModel):
if labels is not None:
use_cache = False
decoder_input_ids = shift_tokens_right(labels, self.config.pad_token_id)
outputs = self.model(
input_ids,
+7 -10
View File
@@ -33,6 +33,7 @@ _all_bart_models = [
"facebook/bart-large-cnn",
"facebook/bart-large-xsum",
"yjernite/bart_eli5",
# This is not exhaustive: see https://huggingface.co/models?filter=bart
]
@@ -133,7 +134,7 @@ class BartTokenizer(RobertaTokenizer):
# Process tgt_texts
if max_target_length is None:
max_target_length = max_length
decoder_inputs: BatchEncoding = self(
labels = self(
tgt_texts,
add_special_tokens=True,
return_tensors=return_tensors,
@@ -141,10 +142,8 @@ class BartTokenizer(RobertaTokenizer):
max_length=max_target_length,
truncation=truncation,
**kwargs,
)
for k, v in decoder_inputs.items():
model_inputs[f"decoder_{k}"] = v
)["input_ids"]
model_inputs["labels"] = labels
return model_inputs
@@ -245,7 +244,7 @@ class BartTokenizerFast(RobertaTokenizerFast):
# Process tgt_texts
if max_target_length is None:
max_target_length = max_length
decoder_inputs: BatchEncoding = self(
labels = self(
tgt_texts,
add_special_tokens=True,
return_tensors=return_tensors,
@@ -253,8 +252,6 @@ class BartTokenizerFast(RobertaTokenizerFast):
max_length=max_target_length,
truncation=truncation,
**kwargs,
)
for k, v in decoder_inputs.items():
model_inputs[f"decoder_{k}"] = v
)["input_ids"]
model_inputs["labels"] = labels
return model_inputs
+42 -33
View File
@@ -56,6 +56,15 @@ FAIRSEQ_LANGUAGE_CODES = [
]
def shift_tokens_right(input_ids, pad_token_id):
"""Shift input ids one token to the right, and wrap the last non pad token (usually <eos>)."""
prev_output_tokens = input_ids.clone()
index_of_eos = (input_ids.ne(pad_token_id).sum(dim=1) - 1).unsqueeze(-1)
prev_output_tokens[:, 0] = input_ids.gather(1, index_of_eos).squeeze()
prev_output_tokens[:, 1:] = input_ids[:, :-1]
return prev_output_tokens
class MBartTokenizer(XLMRobertaTokenizer):
"""
This inherits from XLMRobertaTokenizer. ``prepare_seq2seq_batch`` should be used to encode inputs.
@@ -98,32 +107,6 @@ class MBartTokenizer(XLMRobertaTokenizer):
self._additional_special_tokens = list(self.lang_code_to_id.keys())
self.set_src_lang_special_tokens(kwargs.get("src_lang", "en_XX"))
def build_inputs_with_special_tokens(
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
) -> List[int]:
"""
Build model inputs from a sequence or a pair of sequence for sequence classification tasks
by concatenating and adding special tokens. The special tokens depend on calling set_lang.
An MBART sequence has the following format, where ``X`` represents the sequence:
- ``input_ids`` (for encoder) ``X [eos, src_lang_code]``
- ``decoder_input_ids``: (for decoder) ``[tgt_lang_code] X [eos]``
BOS is never used.
Pairs of sequences are not the expected use case, but they will be handled without a separator.
Args:
token_ids_0 (:obj:`List[int]`):
List of IDs to which the special tokens will be added
token_ids_1 (:obj:`List[int]`, `optional`, defaults to :obj:`None`):
Optional second list of IDs for sequence pairs.
Returns:
:obj:`List[int]`: list of `input IDs <../glossary.html#input-ids>`__ with the appropriate special tokens.
"""
if token_ids_1 is None:
return self.prefix_tokens + token_ids_0 + self.suffix_tokens
# We don't expect to process pairs, but leave the pair logic for API consistency
return self.prefix_tokens + token_ids_0 + token_ids_1 + self.suffix_tokens
def get_special_tokens_mask(
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None, already_has_special_tokens: bool = False
) -> List[int]:
@@ -156,6 +139,32 @@ class MBartTokenizer(XLMRobertaTokenizer):
return prefix_ones + ([0] * len(token_ids_0)) + suffix_ones
return prefix_ones + ([0] * len(token_ids_0)) + ([0] * len(token_ids_1)) + suffix_ones
def build_inputs_with_special_tokens(
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
) -> List[int]:
"""
Build model inputs from a sequence or a pair of sequence for sequence classification tasks
by concatenating and adding special tokens. The special tokens depend on calling set_lang.
An MBART sequence has the following format, where ``X`` represents the sequence:
- ``input_ids`` (for encoder) ``X [eos, src_lang_code]``
- ``decoder_input_ids``: (for decoder) ``[tgt_lang_code] X [eos]``
BOS is never used.
Pairs of sequences are not the expected use case, but they will be handled without a separator.
Args:
token_ids_0 (:obj:`List[int]`):
List of IDs to which the special tokens will be added
token_ids_1 (:obj:`List[int]`, `optional`, defaults to :obj:`None`):
Optional second list of IDs for sequence pairs.
Returns:
:obj:`List[int]`: list of `input IDs <../glossary.html#input-ids>`__ with the appropriate special tokens.
"""
if token_ids_1 is None:
return self.prefix_tokens + token_ids_0 + self.suffix_tokens
# We don't expect to process pairs, but leave the pair logic for API consistency
return self.prefix_tokens + token_ids_0 + token_ids_1 + self.suffix_tokens
@add_start_docstrings_to_callable(PREPARE_SEQ2SEQ_BATCH_DOCSTRING)
def prepare_seq2seq_batch(
self,
@@ -251,7 +260,8 @@ class MBartTokenizer(XLMRobertaTokenizer):
if max_target_length is None:
max_target_length = max_length
self.set_tgt_lang_special_tokens(tgt_lang)
decoder_inputs: BatchEncoding = self(
labels = self(
tgt_texts,
add_special_tokens=True,
return_tensors=return_tensors,
@@ -259,10 +269,9 @@ class MBartTokenizer(XLMRobertaTokenizer):
max_length=max_target_length,
truncation=True,
**kwargs,
)
for k, v in decoder_inputs.items():
model_inputs[f"decoder_{k}"] = v
)["input_ids"]
model_inputs["decoder_input_ids"] = shift_tokens_right(labels, self.pad_token_id)
model_inputs["labels"] = labels
self.set_src_lang_special_tokens(src_lang) # sets to src_lang
return model_inputs
@@ -275,5 +284,5 @@ class MBartTokenizer(XLMRobertaTokenizer):
def set_tgt_lang_special_tokens(self, lang: str) -> None:
"""Reset the special tokens to the target language setting. Prefix [tgt_lang_code], suffix =[eos]."""
self.cur_lang_code = self.lang_code_to_id[lang]
self.prefix_tokens = [self.cur_lang_code]
self.suffix_tokens = [self.eos_token_id]
self.prefix_tokens = []
self.suffix_tokens = [self.eos_token_id, self.cur_lang_code]
+4 -3
View File
@@ -133,7 +133,8 @@ class PegasusTokenizer(ReformerTokenizer):
return model_inputs
if max_target_length is not None:
tokenizer_kwargs["max_length"] = max_target_length
decoder_inputs: BatchEncoding = self(tgt_texts, **tokenizer_kwargs)
for k, v in decoder_inputs.items():
model_inputs[f"decoder_{k}"] = v
labels: BatchEncoding = self(tgt_texts, **tokenizer_kwargs)["input_ids"]
model_inputs["labels"] = labels
# for k, v in decoder_inputs.items():
# model_inputs[f"decoder_{k}"] = v
return model_inputs
+3 -5
View File
@@ -303,7 +303,7 @@ class T5Tokenizer(PreTrainedTokenizer):
if max_length is None:
max_length = self.max_len
self.prefix_tokens = []
model_inputs: BatchEncoding = self(
model_inputs = self(
src_texts,
add_special_tokens=True,
return_tensors=return_tensors,
@@ -319,7 +319,7 @@ class T5Tokenizer(PreTrainedTokenizer):
max_target_length = max_length
# set prefix_tokens for target text
self.prefix_tokens = [self.pad_token_id]
decoder_inputs: BatchEncoding = self(
model_inputs["labels"] = self(
tgt_texts,
add_special_tokens=True,
return_tensors=return_tensors,
@@ -327,9 +327,7 @@ class T5Tokenizer(PreTrainedTokenizer):
max_length=max_target_length,
truncation=truncation,
**kwargs,
)
for k, v in decoder_inputs.items():
model_inputs[f"decoder_{k}"] = v
)["input_ids"]
self.prefix_tokens = []
return model_inputs
+2 -80
View File
@@ -18,7 +18,7 @@ import unittest
import timeout_decorator # noqa
from transformers import BatchEncoding, is_torch_available
from transformers import is_torch_available
from transformers.file_utils import cached_property
from transformers.testing_utils import require_torch, slow, torch_device
@@ -486,7 +486,7 @@ class BartModelIntegrationTests(unittest.TestCase):
def test_xsum_summarization_same_as_fairseq(self):
model = BartForConditionalGeneration.from_pretrained("facebook/bart-large-xsum").to(torch_device)
self.assertFalse(model.config.is_valid_mbart())
tok = BartTokenizer.from_pretrained("facebook/bart-large")
tok = self.default_tokenizer
EXPECTED_SUMMARY = "California's largest power company has begun shutting off electricity to thousands of customers in the state."
dct = tok.batch_encode_plus(
@@ -568,84 +568,6 @@ class BartModelIntegrationTests(unittest.TestCase):
# TODO(SS): run fairseq again with num_beams=2, min_len=20.
# TODO(SS): add test case that hits max_length
def test_prepare_seq2seq_batch(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
tgt_text = [
"Summary of the text.",
"Another summary.",
]
expected_src_tokens = [0, 250, 251, 17818, 13, 32933, 21645, 1258, 4, 2]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_length=len(expected_src_tokens), return_tensors="pt"
)
self.assertIsInstance(batch, BatchEncoding)
self.assertEqual((2, 10), batch.input_ids.shape)
self.assertEqual((2, 10), batch.attention_mask.shape)
result = batch.input_ids.tolist()[0]
self.assertListEqual(expected_src_tokens, result)
# Test that special tokens are reset
def test_empty_target_text(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(src_text, return_tensors="pt")
# check if input_ids are returned and no decoder_input_ids
self.assertIn("input_ids", batch)
self.assertIn("attention_mask", batch)
self.assertNotIn("decoder_input_ids", batch)
self.assertNotIn("decoder_attention_mask", batch)
def test_max_target_length(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
tgt_text = [
"Summary of the text.",
"Another summary.",
]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_target_length=32, padding="max_length", return_tensors="pt"
)
self.assertEqual(32, batch["decoder_input_ids"].shape[1])
self.assertEqual(32, batch["decoder_attention_mask"].shape[1])
# test None max_target_length
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_length=32, padding="max_length", return_tensors="pt"
)
self.assertEqual(32, batch["decoder_input_ids"].shape[1])
self.assertEqual(32, batch["decoder_attention_mask"].shape[1])
def test_outputs_not_longer_than_maxlen(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
["I am a small frog" * 1024, "I am a small frog"], return_tensors="pt"
)
self.assertIsInstance(batch, BatchEncoding)
self.assertEqual(batch.input_ids.shape, (2, 1024))
def test_special_tokens(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization."]
tgt_text = [
"Summary of the text.",
]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, return_tensors="pt")
input_ids = batch["input_ids"]
decoder_input_ids = batch["decoder_input_ids"]
self.assertTrue((input_ids[:, 0] == tokenizer.bos_token_id).all().item())
self.assertTrue((decoder_input_ids[:, 0] == tokenizer.bos_token_id).all().item())
self.assertTrue((input_ids[:, -1] == tokenizer.eos_token_id).all().item())
self.assertTrue((decoder_input_ids[:, -1] == tokenizer.eos_token_id).all().item())
@require_torch
class TestSinusoidalPositionalEmbeddings(unittest.TestCase):
+1 -3
View File
@@ -31,9 +31,7 @@ class PegasusXSUMIntegrationTest(AbstractSeq2SeqIntegrationTest):
@slow
def test_pegasus_xsum_summary(self):
assert self.tokenizer.model_max_length == 512
inputs = self.tokenizer(self.src_text, return_tensors="pt", truncation=True, max_length=512, padding=True).to(
torch_device
)
inputs = self.tokenizer(self.src_text, return_tensors="pt", truncation=True, padding=True).to(torch_device)
assert inputs.input_ids.shape == (2, 421)
translated_tokens = self.model.generate(**inputs)
decoded = self.tokenizer.batch_decode(translated_tokens, skip_special_tokens=True)
+90
View File
@@ -0,0 +1,90 @@
import unittest
from transformers import BartTokenizer, BartTokenizerFast, BatchEncoding
from transformers.file_utils import cached_property
class TestTokenizationBart(unittest.TestCase):
@cached_property
def default_tokenizer(self):
return BartTokenizer.from_pretrained("facebook/bart-large")
@cached_property
def default_tokenizer_fast(self):
return BartTokenizerFast.from_pretrained("facebook/bart-large")
def test_prepare_seq2seq_batch(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
tgt_text = [
"Summary of the text.",
"Another summary.",
]
expected_src_tokens = [0, 250, 251, 17818, 13, 32933, 21645, 1258, 4, 2]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_length=len(expected_src_tokens), return_tensors="pt"
)
self.assertIsInstance(batch, BatchEncoding)
self.assertEqual((2, 10), batch.input_ids.shape)
self.assertEqual((2, 10), batch.attention_mask.shape)
result = batch.input_ids.tolist()[0]
self.assertListEqual(expected_src_tokens, result)
# Test that special tokens are reset
def test_empty_target_text(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(src_text, return_tensors="pt")
# check if input_ids are returned and no labels
self.assertIn("input_ids", batch)
self.assertIn("attention_mask", batch)
self.assertNotIn("labels", batch)
self.assertNotIn("decoder_attention_mask", batch)
def test_max_target_length(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
tgt_text = [
"Summary of the text.",
"Another summary.",
]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_target_length=32, padding="max_length", return_tensors="pt"
)
self.assertEqual(32, batch["labels"].shape[1])
# test None max_target_length
batch = tokenizer.prepare_seq2seq_batch(
src_text, tgt_texts=tgt_text, max_length=32, padding="max_length", return_tensors="pt"
)
self.assertEqual(32, batch["labels"].shape[1])
def test_outputs_not_longer_than_maxlen(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(
["I am a small frog" * 1024, "I am a small frog"], return_tensors="pt"
)
self.assertIsInstance(batch, BatchEncoding)
self.assertEqual(batch.input_ids.shape, (2, 1024))
def test_special_tokens(self):
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
src_text = ["A long paragraph for summrization."]
tgt_text = [
"Summary of the text.",
]
for tokenizer in tokenizers:
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, return_tensors="pt")
input_ids = batch["input_ids"]
labels = batch["labels"]
self.assertTrue((input_ids[:, 0] == tokenizer.bos_token_id).all().item())
self.assertTrue((labels[:, 0] == tokenizer.bos_token_id).all().item())
self.assertTrue((input_ids[:, -1] == tokenizer.eos_token_id).all().item())
self.assertTrue((labels[:, -1] == tokenizer.eos_token_id).all().item())
+2 -2
View File
@@ -1544,11 +1544,11 @@ class TokenizerTesterMixin:
src_texts=src_text, tgt_texts=tgt_text, max_length=3, max_target_length=10, return_tensors="pt"
)
self.assertEqual(batch.input_ids.shape[1], 3)
self.assertEqual(batch.decoder_input_ids.shape[1], 10)
self.assertEqual(batch.labels.shape[1], 10)
# max_target_length will default to max_length if not specified
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, max_length=3)
self.assertEqual(batch.input_ids.shape[1], 3)
self.assertEqual(batch.decoder_input_ids.shape[1], 3)
self.assertEqual(batch.labels.shape[1], 3)
batch_encoder_only = tokenizer.prepare_seq2seq_batch(
src_texts=src_text, max_length=3, max_target_length=10, return_tensors="pt"
+14
View File
@@ -182,3 +182,17 @@ class MBartEnroIntegrationTest(unittest.TestCase):
self.tokenizer.save_pretrained(tmpdirname)
new_tok = MBartTokenizer.from_pretrained(tmpdirname)
self.assertDictEqual(new_tok.fairseq_tokens_to_ids, original_special_tokens)
def test_batch_fairseq_parity(self):
batch: BatchEncoding = self.tokenizer.prepare_seq2seq_batch(
self.src_text, tgt_texts=self.tgt_text, return_tensors="pt"
)
for k in batch:
batch[k] = batch[k].tolist()
# batch = {k: v.tolist() for k,v in batch.items()}
# fairseq batch: https://gist.github.com/sshleifer/cba08bc2109361a74ac3760a7e30e4f4
# batch.decoder_inputs_ids[0][0] ==
assert batch.input_ids[1][-2:] == [2, EN_CODE]
assert batch.decoder_input_ids[1][0] == RO_CODE
assert batch.decoder_input_ids[1][-1] == 2
assert batch.labels[1][-2:] == [2, RO_CODE]
+3 -4
View File
@@ -63,7 +63,6 @@ class PegasusTokenizationTest(TokenizerTesterMixin, unittest.TestCase):
batch = self.pegasus_large_tokenizer.prepare_seq2seq_batch(src_texts, tgt_texts=tgt_texts, max_target_length=5)
assert batch.input_ids.shape == (2, 1024)
assert batch.attention_mask.shape == (2, 1024)
assert "decoder_input_ids" in batch # because tgt_texts was specified
assert batch.decoder_input_ids.shape == (2, 5)
assert batch.decoder_attention_mask.shape == (2, 5)
assert len(batch) == 4 # no extra keys
assert "labels" in batch # because tgt_texts was specified
assert batch.labels.shape == (2, 5)
assert len(batch) == 3 # input_ids, attention_mask, labels. Other things make by BartModel