Compare commits
23
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c2788f19aa | ||
|
|
6aaac683c5 | ||
|
|
d68c671132 | ||
|
|
f5d68bb13b | ||
|
|
0da45503e8 | ||
|
|
aebbe11a67 | ||
|
|
f9613aaf24 | ||
|
|
b2c039bfad | ||
|
|
f328679d93 | ||
|
|
2207e5d8cb | ||
|
|
75978bf5e9 | ||
|
|
018c1bb3da | ||
|
|
b9ca11e3ae | ||
|
|
6bdf998dff | ||
|
|
fcb96a2c1e | ||
|
|
e5a3f09cd4 | ||
|
|
7d5bdb82eb | ||
|
|
556fdea687 | ||
|
|
9ba04a609e | ||
|
|
d63f811af5 | ||
|
|
faac0718cb | ||
|
|
7618c08e55 | ||
|
|
1734ba169b |
@@ -106,7 +106,7 @@ The following command should work on a 16GB GPU:
|
||||
--train_batch_size=1 \
|
||||
--eval_batch_size=1 \
|
||||
--output_dir=xsum_results \
|
||||
--num_train_epochs 1 \
|
||||
--num_train_epochs 6 \
|
||||
--model_name_or_path facebook/bart-large
|
||||
```
|
||||
|
||||
@@ -122,8 +122,8 @@ Best performing command:
|
||||
export ENRO_DIR='wmt_en_ro' # Download instructions above
|
||||
# export WANDB_PROJECT="MT" # optional
|
||||
export MAX_LEN=128
|
||||
export BS=4
|
||||
./train_mbart_cc25_enro.sh --output_dir enro_finetune_baseline --label_smoothing 0.1 --fp16_opt_level=O1 --logger_name wandb --sortish_sampler
|
||||
export BS=8
|
||||
./train_mbart_cc25_enro.sh --output_dir enro_finetune_baseline_dropper --label_smoothing 0 --fp16_opt_level=O1 --logger_name wandb --sortish_sampler
|
||||
```
|
||||
This should take < 6h/epoch on a 16GB v100 and achieve test BLEU above 26
|
||||
To get results in line with fairseq, you need to do some postprocessing. (see `romanian_postprocessing.md`)
|
||||
@@ -141,7 +141,7 @@ export BS=4
|
||||
As you train, `output_dir` will be filled with files, that look kind of like this (comments are mine).
|
||||
Some of them are metrics, some of them are checkpoints, some of them are metadata. Here is a quick tour:
|
||||
|
||||
```bash
|
||||
```
|
||||
output_dir
|
||||
├── best_tfmr # this is a huggingface checkpoint generated by save_pretrained. It is the same model as the PL .ckpt file below
|
||||
│ ├── config.json
|
||||
|
||||
@@ -14,6 +14,7 @@ from torch.utils.data import DataLoader
|
||||
|
||||
from lightning_base import BaseTransformer, add_generic_args, generic_train
|
||||
from transformers import MarianTokenizer, MBartTokenizer, T5ForConditionalGeneration
|
||||
from transformers.modeling_bart import shift_tokens_right
|
||||
|
||||
|
||||
try:
|
||||
@@ -36,6 +37,7 @@ try:
|
||||
)
|
||||
|
||||
from .callbacks import Seq2SeqLoggingCallback, get_checkpoint_callback, get_early_stopping_callback
|
||||
from .loss_dropper import LossDropper
|
||||
except ImportError:
|
||||
from utils import (
|
||||
Seq2SeqDataset,
|
||||
@@ -55,19 +57,21 @@ except ImportError:
|
||||
label_smoothed_nll_loss,
|
||||
)
|
||||
from callbacks import Seq2SeqLoggingCallback, get_checkpoint_callback, get_early_stopping_callback
|
||||
from loss_dropper import LossDropper
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class SummarizationModule(BaseTransformer):
|
||||
mode = "summarization"
|
||||
loss_names = ["loss"]
|
||||
loss_names = ["loss", "dropper_mask_mean"]
|
||||
metric_names = ROUGE_KEYS
|
||||
val_metric = "rouge2"
|
||||
|
||||
def __init__(self, hparams, **kwargs):
|
||||
super().__init__(hparams, num_labels=None, mode=self.mode, **kwargs)
|
||||
use_task_specific_params(self.model, "summarization")
|
||||
self.dropper = LossDropper(dropc=.05)
|
||||
save_git_info(self.hparams.output_dir)
|
||||
self.metrics_save_path = Path(self.output_dir) / "metrics.json"
|
||||
self.hparams_save_path = Path(self.output_dir) / "hparams.pkl"
|
||||
@@ -137,7 +141,10 @@ class SummarizationModule(BaseTransformer):
|
||||
pad_token_id = self.tokenizer.pad_token_id
|
||||
source_ids, source_mask, target_ids = batch["input_ids"], batch["attention_mask"], batch["decoder_input_ids"]
|
||||
|
||||
if isinstance(self.model, T5ForConditionalGeneration):
|
||||
if "labels" in batch:
|
||||
lm_labels = batch["labels"]
|
||||
decoder_input_ids = shift_tokens_right(lm_labels, pad_token_id)
|
||||
elif isinstance(self.model, T5ForConditionalGeneration):
|
||||
decoder_input_ids = self.model._shift_right(target_ids)
|
||||
lm_labels = target_ids
|
||||
else:
|
||||
@@ -145,19 +152,32 @@ class SummarizationModule(BaseTransformer):
|
||||
lm_labels = target_ids[:, 1:].clone() # why clone?
|
||||
|
||||
outputs = self(source_ids, attention_mask=source_mask, decoder_input_ids=decoder_input_ids, use_cache=False)
|
||||
bs = source_ids.shape[0]
|
||||
|
||||
if self.hparams.label_smoothing == 0:
|
||||
|
||||
# Same behavior as modeling_bart.py
|
||||
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=pad_token_id)
|
||||
loss_fct = torch.nn.CrossEntropyLoss(reduction='none', ignore_index=pad_token_id)
|
||||
lm_logits = outputs[0]
|
||||
assert lm_logits.shape[-1] == self.model.config.vocab_size
|
||||
|
||||
#loss_fct = torch.nn.NLLLoss(reduction='none', ignore_index=pad_token_id)
|
||||
#logit_shape =
|
||||
#weights = torch.ones(logit_shape
|
||||
loss = loss_fct(lm_logits.view(-1, lm_logits.shape[-1]), lm_labels.view(-1))
|
||||
loss = loss.view(-1, bs)
|
||||
loss = loss.mean(dim=0)
|
||||
mask = self.dropper(loss)
|
||||
loss *= mask
|
||||
loss = loss.mean()
|
||||
return (loss, 1-mask.mean())
|
||||
#loss = loss.view(-1, bs)
|
||||
else:
|
||||
lprobs = torch.nn.functional.log_softmax(outputs[0], dim=-1)
|
||||
loss, nll_loss = label_smoothed_nll_loss(
|
||||
lprobs, lm_labels, self.hparams.label_smoothing, ignore_index=pad_token_id
|
||||
)
|
||||
return (loss,)
|
||||
return (loss,torch.tensor(1.))
|
||||
|
||||
@property
|
||||
def pad(self) -> int:
|
||||
@@ -302,6 +322,7 @@ class SummarizationModule(BaseTransformer):
|
||||
"--task", type=str, default="summarization", required=False, help="# examples. -1 means use all."
|
||||
)
|
||||
parser.add_argument("--label_smoothing", type=float, default=0.0, required=False)
|
||||
parser.add_argument("--loss_dropper", type=float, default=0.0, required=False)
|
||||
parser.add_argument("--src_lang", type=str, default="", required=False)
|
||||
parser.add_argument("--tgt_lang", type=str, default="", required=False)
|
||||
parser.add_argument(
|
||||
@@ -316,7 +337,6 @@ class SummarizationModule(BaseTransformer):
|
||||
|
||||
class TranslationModule(SummarizationModule):
|
||||
mode = "translation"
|
||||
loss_names = ["loss"]
|
||||
metric_names = ["bleu"]
|
||||
val_metric = "bleu"
|
||||
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
import numpy as np
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
class LossDropper(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
dropc=0.4,
|
||||
min_count=10000,
|
||||
recompute=10000,
|
||||
verbose=True
|
||||
):
|
||||
super().__init__()
|
||||
self.keepc = 1. - dropc
|
||||
self.count = 0
|
||||
self.min_count = min_count
|
||||
|
||||
self.recompute = recompute
|
||||
self.last_computed = 0
|
||||
self.percentile_val = 100000000.
|
||||
self.cur_idx = 0
|
||||
|
||||
self.verbose = verbose
|
||||
|
||||
self.vals = np.zeros(self.recompute, dtype=np.float32)
|
||||
|
||||
def forward(self, loss):
|
||||
if loss is None:
|
||||
return loss
|
||||
|
||||
self.last_computed += loss.numel()
|
||||
self.count += loss.numel()
|
||||
if self.count < len(self.vals):
|
||||
self.vals[self.count - loss.numel():self.count] = loss.detach().cpu().numpy().flatten()
|
||||
self.cur_idx += loss.numel()
|
||||
return (loss < np.inf).type(loss.dtype)
|
||||
else:
|
||||
for idx, item in enumerate(loss):
|
||||
self.vals[self.cur_idx] = item
|
||||
self.cur_idx += 1
|
||||
if self.cur_idx >= len(self.vals):
|
||||
self.cur_idx = 0
|
||||
if self.count < self.min_count:
|
||||
return (loss < np.inf).type(loss.dtype)
|
||||
|
||||
if self.last_computed > self.recompute:
|
||||
self.percentile_val = np.percentile(self.vals, self.keepc * 100)
|
||||
if self.verbose:
|
||||
print('Using cutoff', self.percentile_val)
|
||||
self.last_computed = 0
|
||||
|
||||
mask = (loss < self.percentile_val).type(loss.dtype)
|
||||
return mask
|
||||
@@ -6,7 +6,7 @@ export GAS=1
|
||||
|
||||
python finetune.py \
|
||||
--learning_rate=3e-5 \
|
||||
--fp16 \
|
||||
--fp16 --fp16_opt_level=O1 \
|
||||
--gpus 1 \
|
||||
--do_train \
|
||||
--do_predict \
|
||||
|
||||
@@ -6,7 +6,7 @@ python distillation.py \
|
||||
--learning_rate=3e-4 \
|
||||
--do_train \
|
||||
--do_predict \
|
||||
--fp16 \
|
||||
--fp16 --fp16_opt_level=O1 \
|
||||
--val_check_interval 0.1 --n_val 1000 \
|
||||
--teacher facebook/bart-large-xsum --data_dir $XSUM_DIR \
|
||||
--max_target_length=60 --val_max_target_length=60 --test_max_target_length=100 \
|
||||
|
||||
@@ -14,5 +14,4 @@ python finetune.py \
|
||||
--task translation \
|
||||
--warmup_steps 500 \
|
||||
--freeze_embeds \
|
||||
--model_name_or_path=facebook/mbart-large-cc25 \
|
||||
"$@"
|
||||
|
||||
@@ -299,7 +299,6 @@ if is_torch_available():
|
||||
BartModel,
|
||||
BartForConditionalGeneration,
|
||||
BartForQuestionAnswering,
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
)
|
||||
from .modeling_mbart import MBartForConditionalGeneration
|
||||
from .modeling_marian import MarianMTModel
|
||||
|
||||
@@ -51,15 +51,7 @@ _CONFIG_FOR_DOC = "BartConfig"
|
||||
_TOKENIZER_FOR_DOC = "BartTokenizer"
|
||||
|
||||
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_LIST = [
|
||||
"facebook/bart-base",
|
||||
"facebook/bart-large",
|
||||
"facebook/bart-large-mnli",
|
||||
"facebook/bart-large-cnn",
|
||||
"facebook/bart-large-xsum",
|
||||
"facebook/mbart-large-en-ro",
|
||||
# See all BART models at https://huggingface.co/models?filter=bart
|
||||
]
|
||||
# See all BART models at https://huggingface.co/models?filter=bart
|
||||
|
||||
|
||||
BART_START_DOCSTRING = r"""
|
||||
@@ -1023,6 +1015,7 @@ class BartForConditionalGeneration(PretrainedBartModel):
|
||||
|
||||
if labels is not None:
|
||||
use_cache = False
|
||||
decoder_input_ids = shift_tokens_right(labels, self.config.pad_token_id)
|
||||
|
||||
outputs = self.model(
|
||||
input_ids,
|
||||
|
||||
@@ -33,6 +33,7 @@ _all_bart_models = [
|
||||
"facebook/bart-large-cnn",
|
||||
"facebook/bart-large-xsum",
|
||||
"yjernite/bart_eli5",
|
||||
# This is not exhaustive: see https://huggingface.co/models?filter=bart
|
||||
]
|
||||
|
||||
|
||||
@@ -133,7 +134,7 @@ class BartTokenizer(RobertaTokenizer):
|
||||
# Process tgt_texts
|
||||
if max_target_length is None:
|
||||
max_target_length = max_length
|
||||
decoder_inputs: BatchEncoding = self(
|
||||
labels = self(
|
||||
tgt_texts,
|
||||
add_special_tokens=True,
|
||||
return_tensors=return_tensors,
|
||||
@@ -141,10 +142,8 @@ class BartTokenizer(RobertaTokenizer):
|
||||
max_length=max_target_length,
|
||||
truncation=truncation,
|
||||
**kwargs,
|
||||
)
|
||||
for k, v in decoder_inputs.items():
|
||||
model_inputs[f"decoder_{k}"] = v
|
||||
|
||||
)["input_ids"]
|
||||
model_inputs["labels"] = labels
|
||||
return model_inputs
|
||||
|
||||
|
||||
@@ -245,7 +244,7 @@ class BartTokenizerFast(RobertaTokenizerFast):
|
||||
# Process tgt_texts
|
||||
if max_target_length is None:
|
||||
max_target_length = max_length
|
||||
decoder_inputs: BatchEncoding = self(
|
||||
labels = self(
|
||||
tgt_texts,
|
||||
add_special_tokens=True,
|
||||
return_tensors=return_tensors,
|
||||
@@ -253,8 +252,6 @@ class BartTokenizerFast(RobertaTokenizerFast):
|
||||
max_length=max_target_length,
|
||||
truncation=truncation,
|
||||
**kwargs,
|
||||
)
|
||||
for k, v in decoder_inputs.items():
|
||||
model_inputs[f"decoder_{k}"] = v
|
||||
|
||||
)["input_ids"]
|
||||
model_inputs["labels"] = labels
|
||||
return model_inputs
|
||||
|
||||
@@ -56,6 +56,15 @@ FAIRSEQ_LANGUAGE_CODES = [
|
||||
]
|
||||
|
||||
|
||||
def shift_tokens_right(input_ids, pad_token_id):
|
||||
"""Shift input ids one token to the right, and wrap the last non pad token (usually <eos>)."""
|
||||
prev_output_tokens = input_ids.clone()
|
||||
index_of_eos = (input_ids.ne(pad_token_id).sum(dim=1) - 1).unsqueeze(-1)
|
||||
prev_output_tokens[:, 0] = input_ids.gather(1, index_of_eos).squeeze()
|
||||
prev_output_tokens[:, 1:] = input_ids[:, :-1]
|
||||
return prev_output_tokens
|
||||
|
||||
|
||||
class MBartTokenizer(XLMRobertaTokenizer):
|
||||
"""
|
||||
This inherits from XLMRobertaTokenizer. ``prepare_seq2seq_batch`` should be used to encode inputs.
|
||||
@@ -98,32 +107,6 @@ class MBartTokenizer(XLMRobertaTokenizer):
|
||||
self._additional_special_tokens = list(self.lang_code_to_id.keys())
|
||||
self.set_src_lang_special_tokens(kwargs.get("src_lang", "en_XX"))
|
||||
|
||||
def build_inputs_with_special_tokens(
|
||||
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
|
||||
) -> List[int]:
|
||||
"""
|
||||
Build model inputs from a sequence or a pair of sequence for sequence classification tasks
|
||||
by concatenating and adding special tokens. The special tokens depend on calling set_lang.
|
||||
An MBART sequence has the following format, where ``X`` represents the sequence:
|
||||
- ``input_ids`` (for encoder) ``X [eos, src_lang_code]``
|
||||
- ``decoder_input_ids``: (for decoder) ``[tgt_lang_code] X [eos]``
|
||||
BOS is never used.
|
||||
Pairs of sequences are not the expected use case, but they will be handled without a separator.
|
||||
|
||||
Args:
|
||||
token_ids_0 (:obj:`List[int]`):
|
||||
List of IDs to which the special tokens will be added
|
||||
token_ids_1 (:obj:`List[int]`, `optional`, defaults to :obj:`None`):
|
||||
Optional second list of IDs for sequence pairs.
|
||||
|
||||
Returns:
|
||||
:obj:`List[int]`: list of `input IDs <../glossary.html#input-ids>`__ with the appropriate special tokens.
|
||||
"""
|
||||
if token_ids_1 is None:
|
||||
return self.prefix_tokens + token_ids_0 + self.suffix_tokens
|
||||
# We don't expect to process pairs, but leave the pair logic for API consistency
|
||||
return self.prefix_tokens + token_ids_0 + token_ids_1 + self.suffix_tokens
|
||||
|
||||
def get_special_tokens_mask(
|
||||
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None, already_has_special_tokens: bool = False
|
||||
) -> List[int]:
|
||||
@@ -156,6 +139,32 @@ class MBartTokenizer(XLMRobertaTokenizer):
|
||||
return prefix_ones + ([0] * len(token_ids_0)) + suffix_ones
|
||||
return prefix_ones + ([0] * len(token_ids_0)) + ([0] * len(token_ids_1)) + suffix_ones
|
||||
|
||||
def build_inputs_with_special_tokens(
|
||||
self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
|
||||
) -> List[int]:
|
||||
"""
|
||||
Build model inputs from a sequence or a pair of sequence for sequence classification tasks
|
||||
by concatenating and adding special tokens. The special tokens depend on calling set_lang.
|
||||
An MBART sequence has the following format, where ``X`` represents the sequence:
|
||||
- ``input_ids`` (for encoder) ``X [eos, src_lang_code]``
|
||||
- ``decoder_input_ids``: (for decoder) ``[tgt_lang_code] X [eos]``
|
||||
BOS is never used.
|
||||
Pairs of sequences are not the expected use case, but they will be handled without a separator.
|
||||
|
||||
Args:
|
||||
token_ids_0 (:obj:`List[int]`):
|
||||
List of IDs to which the special tokens will be added
|
||||
token_ids_1 (:obj:`List[int]`, `optional`, defaults to :obj:`None`):
|
||||
Optional second list of IDs for sequence pairs.
|
||||
|
||||
Returns:
|
||||
:obj:`List[int]`: list of `input IDs <../glossary.html#input-ids>`__ with the appropriate special tokens.
|
||||
"""
|
||||
if token_ids_1 is None:
|
||||
return self.prefix_tokens + token_ids_0 + self.suffix_tokens
|
||||
# We don't expect to process pairs, but leave the pair logic for API consistency
|
||||
return self.prefix_tokens + token_ids_0 + token_ids_1 + self.suffix_tokens
|
||||
|
||||
@add_start_docstrings_to_callable(PREPARE_SEQ2SEQ_BATCH_DOCSTRING)
|
||||
def prepare_seq2seq_batch(
|
||||
self,
|
||||
@@ -251,7 +260,8 @@ class MBartTokenizer(XLMRobertaTokenizer):
|
||||
if max_target_length is None:
|
||||
max_target_length = max_length
|
||||
self.set_tgt_lang_special_tokens(tgt_lang)
|
||||
decoder_inputs: BatchEncoding = self(
|
||||
|
||||
labels = self(
|
||||
tgt_texts,
|
||||
add_special_tokens=True,
|
||||
return_tensors=return_tensors,
|
||||
@@ -259,10 +269,9 @@ class MBartTokenizer(XLMRobertaTokenizer):
|
||||
max_length=max_target_length,
|
||||
truncation=True,
|
||||
**kwargs,
|
||||
)
|
||||
for k, v in decoder_inputs.items():
|
||||
model_inputs[f"decoder_{k}"] = v
|
||||
|
||||
)["input_ids"]
|
||||
model_inputs["decoder_input_ids"] = shift_tokens_right(labels, self.pad_token_id)
|
||||
model_inputs["labels"] = labels
|
||||
self.set_src_lang_special_tokens(src_lang) # sets to src_lang
|
||||
return model_inputs
|
||||
|
||||
@@ -275,5 +284,5 @@ class MBartTokenizer(XLMRobertaTokenizer):
|
||||
def set_tgt_lang_special_tokens(self, lang: str) -> None:
|
||||
"""Reset the special tokens to the target language setting. Prefix [tgt_lang_code], suffix =[eos]."""
|
||||
self.cur_lang_code = self.lang_code_to_id[lang]
|
||||
self.prefix_tokens = [self.cur_lang_code]
|
||||
self.suffix_tokens = [self.eos_token_id]
|
||||
self.prefix_tokens = []
|
||||
self.suffix_tokens = [self.eos_token_id, self.cur_lang_code]
|
||||
|
||||
@@ -133,7 +133,8 @@ class PegasusTokenizer(ReformerTokenizer):
|
||||
return model_inputs
|
||||
if max_target_length is not None:
|
||||
tokenizer_kwargs["max_length"] = max_target_length
|
||||
decoder_inputs: BatchEncoding = self(tgt_texts, **tokenizer_kwargs)
|
||||
for k, v in decoder_inputs.items():
|
||||
model_inputs[f"decoder_{k}"] = v
|
||||
labels: BatchEncoding = self(tgt_texts, **tokenizer_kwargs)["input_ids"]
|
||||
model_inputs["labels"] = labels
|
||||
# for k, v in decoder_inputs.items():
|
||||
# model_inputs[f"decoder_{k}"] = v
|
||||
return model_inputs
|
||||
|
||||
@@ -303,7 +303,7 @@ class T5Tokenizer(PreTrainedTokenizer):
|
||||
if max_length is None:
|
||||
max_length = self.max_len
|
||||
self.prefix_tokens = []
|
||||
model_inputs: BatchEncoding = self(
|
||||
model_inputs = self(
|
||||
src_texts,
|
||||
add_special_tokens=True,
|
||||
return_tensors=return_tensors,
|
||||
@@ -319,7 +319,7 @@ class T5Tokenizer(PreTrainedTokenizer):
|
||||
max_target_length = max_length
|
||||
# set prefix_tokens for target text
|
||||
self.prefix_tokens = [self.pad_token_id]
|
||||
decoder_inputs: BatchEncoding = self(
|
||||
model_inputs["labels"] = self(
|
||||
tgt_texts,
|
||||
add_special_tokens=True,
|
||||
return_tensors=return_tensors,
|
||||
@@ -327,9 +327,7 @@ class T5Tokenizer(PreTrainedTokenizer):
|
||||
max_length=max_target_length,
|
||||
truncation=truncation,
|
||||
**kwargs,
|
||||
)
|
||||
for k, v in decoder_inputs.items():
|
||||
model_inputs[f"decoder_{k}"] = v
|
||||
)["input_ids"]
|
||||
|
||||
self.prefix_tokens = []
|
||||
return model_inputs
|
||||
|
||||
@@ -18,7 +18,7 @@ import unittest
|
||||
|
||||
import timeout_decorator # noqa
|
||||
|
||||
from transformers import BatchEncoding, is_torch_available
|
||||
from transformers import is_torch_available
|
||||
from transformers.file_utils import cached_property
|
||||
from transformers.testing_utils import require_torch, slow, torch_device
|
||||
|
||||
@@ -486,7 +486,7 @@ class BartModelIntegrationTests(unittest.TestCase):
|
||||
def test_xsum_summarization_same_as_fairseq(self):
|
||||
model = BartForConditionalGeneration.from_pretrained("facebook/bart-large-xsum").to(torch_device)
|
||||
self.assertFalse(model.config.is_valid_mbart())
|
||||
tok = BartTokenizer.from_pretrained("facebook/bart-large")
|
||||
tok = self.default_tokenizer
|
||||
|
||||
EXPECTED_SUMMARY = "California's largest power company has begun shutting off electricity to thousands of customers in the state."
|
||||
dct = tok.batch_encode_plus(
|
||||
@@ -568,84 +568,6 @@ class BartModelIntegrationTests(unittest.TestCase):
|
||||
# TODO(SS): run fairseq again with num_beams=2, min_len=20.
|
||||
# TODO(SS): add test case that hits max_length
|
||||
|
||||
def test_prepare_seq2seq_batch(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
"Another summary.",
|
||||
]
|
||||
expected_src_tokens = [0, 250, 251, 17818, 13, 32933, 21645, 1258, 4, 2]
|
||||
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_length=len(expected_src_tokens), return_tensors="pt"
|
||||
)
|
||||
self.assertIsInstance(batch, BatchEncoding)
|
||||
|
||||
self.assertEqual((2, 10), batch.input_ids.shape)
|
||||
self.assertEqual((2, 10), batch.attention_mask.shape)
|
||||
result = batch.input_ids.tolist()[0]
|
||||
self.assertListEqual(expected_src_tokens, result)
|
||||
# Test that special tokens are reset
|
||||
|
||||
def test_empty_target_text(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(src_text, return_tensors="pt")
|
||||
# check if input_ids are returned and no decoder_input_ids
|
||||
self.assertIn("input_ids", batch)
|
||||
self.assertIn("attention_mask", batch)
|
||||
self.assertNotIn("decoder_input_ids", batch)
|
||||
self.assertNotIn("decoder_attention_mask", batch)
|
||||
|
||||
def test_max_target_length(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
"Another summary.",
|
||||
]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_target_length=32, padding="max_length", return_tensors="pt"
|
||||
)
|
||||
self.assertEqual(32, batch["decoder_input_ids"].shape[1])
|
||||
self.assertEqual(32, batch["decoder_attention_mask"].shape[1])
|
||||
|
||||
# test None max_target_length
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_length=32, padding="max_length", return_tensors="pt"
|
||||
)
|
||||
self.assertEqual(32, batch["decoder_input_ids"].shape[1])
|
||||
self.assertEqual(32, batch["decoder_attention_mask"].shape[1])
|
||||
|
||||
def test_outputs_not_longer_than_maxlen(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
["I am a small frog" * 1024, "I am a small frog"], return_tensors="pt"
|
||||
)
|
||||
self.assertIsInstance(batch, BatchEncoding)
|
||||
self.assertEqual(batch.input_ids.shape, (2, 1024))
|
||||
|
||||
def test_special_tokens(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, return_tensors="pt")
|
||||
input_ids = batch["input_ids"]
|
||||
decoder_input_ids = batch["decoder_input_ids"]
|
||||
self.assertTrue((input_ids[:, 0] == tokenizer.bos_token_id).all().item())
|
||||
self.assertTrue((decoder_input_ids[:, 0] == tokenizer.bos_token_id).all().item())
|
||||
self.assertTrue((input_ids[:, -1] == tokenizer.eos_token_id).all().item())
|
||||
self.assertTrue((decoder_input_ids[:, -1] == tokenizer.eos_token_id).all().item())
|
||||
|
||||
|
||||
@require_torch
|
||||
class TestSinusoidalPositionalEmbeddings(unittest.TestCase):
|
||||
|
||||
@@ -31,9 +31,7 @@ class PegasusXSUMIntegrationTest(AbstractSeq2SeqIntegrationTest):
|
||||
@slow
|
||||
def test_pegasus_xsum_summary(self):
|
||||
assert self.tokenizer.model_max_length == 512
|
||||
inputs = self.tokenizer(self.src_text, return_tensors="pt", truncation=True, max_length=512, padding=True).to(
|
||||
torch_device
|
||||
)
|
||||
inputs = self.tokenizer(self.src_text, return_tensors="pt", truncation=True, padding=True).to(torch_device)
|
||||
assert inputs.input_ids.shape == (2, 421)
|
||||
translated_tokens = self.model.generate(**inputs)
|
||||
decoded = self.tokenizer.batch_decode(translated_tokens, skip_special_tokens=True)
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
import unittest
|
||||
|
||||
from transformers import BartTokenizer, BartTokenizerFast, BatchEncoding
|
||||
from transformers.file_utils import cached_property
|
||||
|
||||
|
||||
class TestTokenizationBart(unittest.TestCase):
|
||||
@cached_property
|
||||
def default_tokenizer(self):
|
||||
return BartTokenizer.from_pretrained("facebook/bart-large")
|
||||
|
||||
@cached_property
|
||||
def default_tokenizer_fast(self):
|
||||
return BartTokenizerFast.from_pretrained("facebook/bart-large")
|
||||
|
||||
def test_prepare_seq2seq_batch(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
"Another summary.",
|
||||
]
|
||||
expected_src_tokens = [0, 250, 251, 17818, 13, 32933, 21645, 1258, 4, 2]
|
||||
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_length=len(expected_src_tokens), return_tensors="pt"
|
||||
)
|
||||
self.assertIsInstance(batch, BatchEncoding)
|
||||
|
||||
self.assertEqual((2, 10), batch.input_ids.shape)
|
||||
self.assertEqual((2, 10), batch.attention_mask.shape)
|
||||
result = batch.input_ids.tolist()[0]
|
||||
self.assertListEqual(expected_src_tokens, result)
|
||||
# Test that special tokens are reset
|
||||
|
||||
def test_empty_target_text(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(src_text, return_tensors="pt")
|
||||
# check if input_ids are returned and no labels
|
||||
self.assertIn("input_ids", batch)
|
||||
self.assertIn("attention_mask", batch)
|
||||
self.assertNotIn("labels", batch)
|
||||
self.assertNotIn("decoder_attention_mask", batch)
|
||||
|
||||
def test_max_target_length(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization.", "Another paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
"Another summary.",
|
||||
]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_target_length=32, padding="max_length", return_tensors="pt"
|
||||
)
|
||||
self.assertEqual(32, batch["labels"].shape[1])
|
||||
|
||||
# test None max_target_length
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
src_text, tgt_texts=tgt_text, max_length=32, padding="max_length", return_tensors="pt"
|
||||
)
|
||||
self.assertEqual(32, batch["labels"].shape[1])
|
||||
|
||||
def test_outputs_not_longer_than_maxlen(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(
|
||||
["I am a small frog" * 1024, "I am a small frog"], return_tensors="pt"
|
||||
)
|
||||
self.assertIsInstance(batch, BatchEncoding)
|
||||
self.assertEqual(batch.input_ids.shape, (2, 1024))
|
||||
|
||||
def test_special_tokens(self):
|
||||
tokenizers = [self.default_tokenizer, self.default_tokenizer_fast]
|
||||
src_text = ["A long paragraph for summrization."]
|
||||
tgt_text = [
|
||||
"Summary of the text.",
|
||||
]
|
||||
for tokenizer in tokenizers:
|
||||
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, return_tensors="pt")
|
||||
input_ids = batch["input_ids"]
|
||||
labels = batch["labels"]
|
||||
self.assertTrue((input_ids[:, 0] == tokenizer.bos_token_id).all().item())
|
||||
self.assertTrue((labels[:, 0] == tokenizer.bos_token_id).all().item())
|
||||
self.assertTrue((input_ids[:, -1] == tokenizer.eos_token_id).all().item())
|
||||
self.assertTrue((labels[:, -1] == tokenizer.eos_token_id).all().item())
|
||||
@@ -1544,11 +1544,11 @@ class TokenizerTesterMixin:
|
||||
src_texts=src_text, tgt_texts=tgt_text, max_length=3, max_target_length=10, return_tensors="pt"
|
||||
)
|
||||
self.assertEqual(batch.input_ids.shape[1], 3)
|
||||
self.assertEqual(batch.decoder_input_ids.shape[1], 10)
|
||||
self.assertEqual(batch.labels.shape[1], 10)
|
||||
# max_target_length will default to max_length if not specified
|
||||
batch = tokenizer.prepare_seq2seq_batch(src_text, tgt_texts=tgt_text, max_length=3)
|
||||
self.assertEqual(batch.input_ids.shape[1], 3)
|
||||
self.assertEqual(batch.decoder_input_ids.shape[1], 3)
|
||||
self.assertEqual(batch.labels.shape[1], 3)
|
||||
|
||||
batch_encoder_only = tokenizer.prepare_seq2seq_batch(
|
||||
src_texts=src_text, max_length=3, max_target_length=10, return_tensors="pt"
|
||||
|
||||
@@ -182,3 +182,17 @@ class MBartEnroIntegrationTest(unittest.TestCase):
|
||||
self.tokenizer.save_pretrained(tmpdirname)
|
||||
new_tok = MBartTokenizer.from_pretrained(tmpdirname)
|
||||
self.assertDictEqual(new_tok.fairseq_tokens_to_ids, original_special_tokens)
|
||||
|
||||
def test_batch_fairseq_parity(self):
|
||||
batch: BatchEncoding = self.tokenizer.prepare_seq2seq_batch(
|
||||
self.src_text, tgt_texts=self.tgt_text, return_tensors="pt"
|
||||
)
|
||||
for k in batch:
|
||||
batch[k] = batch[k].tolist()
|
||||
# batch = {k: v.tolist() for k,v in batch.items()}
|
||||
# fairseq batch: https://gist.github.com/sshleifer/cba08bc2109361a74ac3760a7e30e4f4
|
||||
# batch.decoder_inputs_ids[0][0] ==
|
||||
assert batch.input_ids[1][-2:] == [2, EN_CODE]
|
||||
assert batch.decoder_input_ids[1][0] == RO_CODE
|
||||
assert batch.decoder_input_ids[1][-1] == 2
|
||||
assert batch.labels[1][-2:] == [2, RO_CODE]
|
||||
|
||||
@@ -63,7 +63,6 @@ class PegasusTokenizationTest(TokenizerTesterMixin, unittest.TestCase):
|
||||
batch = self.pegasus_large_tokenizer.prepare_seq2seq_batch(src_texts, tgt_texts=tgt_texts, max_target_length=5)
|
||||
assert batch.input_ids.shape == (2, 1024)
|
||||
assert batch.attention_mask.shape == (2, 1024)
|
||||
assert "decoder_input_ids" in batch # because tgt_texts was specified
|
||||
assert batch.decoder_input_ids.shape == (2, 5)
|
||||
assert batch.decoder_attention_mask.shape == (2, 5)
|
||||
assert len(batch) == 4 # no extra keys
|
||||
assert "labels" in batch # because tgt_texts was specified
|
||||
assert batch.labels.shape == (2, 5)
|
||||
assert len(batch) == 3 # input_ids, attention_mask, labels. Other things make by BartModel
|
||||
|
||||
Reference in New Issue
Block a user